From: "Theodore Ts'o" <tytso@mit.edu>
To: Ext4 Developers List <linux-ext4@vger.kernel.org>,
Linux Kernel Developers List <linux-kernel@vger.kernel.org>
Cc: Frederic Bohe <frederic.bohe@bull.net>, Mingming Cao <cmm@us.ibm.com>
Subject: [PATCH 31/52] ext4: fix online resize with mballoc
Date: Sat, 5 Jul 2008 13:35:57 -0400 [thread overview]
Message-ID: <1215279378-30504-32-git-send-email-tytso@mit.edu> (raw)
In-Reply-To: <1215279378-30504-31-git-send-email-tytso@mit.edu>
From: Frederic Bohe <frederic.bohe@bull.net>
Update group infos when updating a group's descriptor.
Add group infos when adding a group's descriptor.
Refresh cache pages used by mb_alloc when changes occur.
This will probably need modifications when META_BG resizing will be allowed.
Signed-off-by: Frederic Bohe <frederic.bohe@bull.net>
Signed-off-by: Mingming Cao <cmm@us.ibm.com>
---
fs/ext4/ext4.h | 4 +
fs/ext4/mballoc.c | 234 ++++++++++++++++++++++++++++++++++++++++-------------
fs/ext4/resize.c | 52 ++++++++++++-
3 files changed, 234 insertions(+), 56 deletions(-)
diff --git a/fs/ext4/ext4.h b/fs/ext4/ext4.h
index 37ce906..b8b0ae7 100644
--- a/fs/ext4/ext4.h
+++ b/fs/ext4/ext4.h
@@ -1033,6 +1033,10 @@ extern int __init init_ext4_mballoc(void);
extern void exit_ext4_mballoc(void);
extern void ext4_mb_free_blocks(handle_t *, struct inode *,
unsigned long, unsigned long, int, unsigned long *);
+extern int ext4_mb_add_more_groupinfo(struct super_block *sb,
+ ext4_group_t i, struct ext4_group_desc *desc);
+extern void ext4_mb_update_group_info(struct ext4_group_info *grp,
+ ext4_grpblk_t add);
/* inode.c */
diff --git a/fs/ext4/mballoc.c b/fs/ext4/mballoc.c
index 9ffad23..a1349b3 100644
--- a/fs/ext4/mballoc.c
+++ b/fs/ext4/mballoc.c
@@ -2231,21 +2231,192 @@ ext4_mb_store_history(struct ext4_allocation_context *ac)
#define ext4_mb_history_init(sb)
#endif
+
+/* Create and initialize ext4_group_info data for the given group. */
+int ext4_mb_add_groupinfo(struct super_block *sb, ext4_group_t group,
+ struct ext4_group_desc *desc)
+{
+ int i, len;
+ int metalen = 0;
+ struct ext4_sb_info *sbi = EXT4_SB(sb);
+ struct ext4_group_info **meta_group_info;
+
+ /*
+ * First check if this group is the first of a reserved block.
+ * If it's true, we have to allocate a new table of pointers
+ * to ext4_group_info structures
+ */
+ if (group % EXT4_DESC_PER_BLOCK(sb) == 0) {
+ metalen = sizeof(*meta_group_info) <<
+ EXT4_DESC_PER_BLOCK_BITS(sb);
+ meta_group_info = kmalloc(metalen, GFP_KERNEL);
+ if (meta_group_info == NULL) {
+ printk(KERN_ERR "EXT4-fs: can't allocate mem for a "
+ "buddy group\n");
+ goto exit_meta_group_info;
+ }
+ sbi->s_group_info[group >> EXT4_DESC_PER_BLOCK_BITS(sb)] =
+ meta_group_info;
+ }
+
+ /*
+ * calculate needed size. if change bb_counters size,
+ * don't forget about ext4_mb_generate_buddy()
+ */
+ len = offsetof(typeof(**meta_group_info),
+ bb_counters[sb->s_blocksize_bits + 2]);
+
+ meta_group_info =
+ sbi->s_group_info[group >> EXT4_DESC_PER_BLOCK_BITS(sb)];
+ i = group & (EXT4_DESC_PER_BLOCK(sb) - 1);
+
+ meta_group_info[i] = kzalloc(len, GFP_KERNEL);
+ if (meta_group_info[i] == NULL) {
+ printk(KERN_ERR "EXT4-fs: can't allocate buddy mem\n");
+ goto exit_group_info;
+ }
+ set_bit(EXT4_GROUP_INFO_NEED_INIT_BIT,
+ &(meta_group_info[i]->bb_state));
+
+ /*
+ * initialize bb_free to be able to skip
+ * empty groups without initialization
+ */
+ if (desc->bg_flags & cpu_to_le16(EXT4_BG_BLOCK_UNINIT)) {
+ meta_group_info[i]->bb_free =
+ ext4_free_blocks_after_init(sb, group, desc);
+ } else {
+ meta_group_info[i]->bb_free =
+ le16_to_cpu(desc->bg_free_blocks_count);
+ }
+
+ INIT_LIST_HEAD(&meta_group_info[i]->bb_prealloc_list);
+
+#ifdef DOUBLE_CHECK
+ {
+ struct buffer_head *bh;
+ meta_group_info[i]->bb_bitmap =
+ kmalloc(sb->s_blocksize, GFP_KERNEL);
+ BUG_ON(meta_group_info[i]->bb_bitmap == NULL);
+ bh = ext4_read_block_bitmap(sb, group);
+ BUG_ON(bh == NULL);
+ memcpy(meta_group_info[i]->bb_bitmap, bh->b_data,
+ sb->s_blocksize);
+ put_bh(bh);
+ }
+#endif
+
+ return 0;
+
+exit_group_info:
+ /* If a meta_group_info table has been allocated, release it now */
+ if (group % EXT4_DESC_PER_BLOCK(sb) == 0)
+ kfree(sbi->s_group_info[group >> EXT4_DESC_PER_BLOCK_BITS(sb)]);
+exit_meta_group_info:
+ return -ENOMEM;
+} /* ext4_mb_add_groupinfo */
+
+/*
+ * Add a group to the existing groups.
+ * This function is used for online resize
+ */
+int ext4_mb_add_more_groupinfo(struct super_block *sb, ext4_group_t group,
+ struct ext4_group_desc *desc)
+{
+ struct ext4_sb_info *sbi = EXT4_SB(sb);
+ struct inode *inode = sbi->s_buddy_cache;
+ int blocks_per_page;
+ int block;
+ int pnum;
+ struct page *page;
+ int err;
+
+ /* Add group based on group descriptor*/
+ err = ext4_mb_add_groupinfo(sb, group, desc);
+ if (err)
+ return err;
+
+ /*
+ * Cache pages containing dynamic mb_alloc datas (buddy and bitmap
+ * datas) are set not up to date so that they will be re-initilaized
+ * during the next call to ext4_mb_load_buddy
+ */
+
+ /* Set buddy page as not up to date */
+ blocks_per_page = PAGE_CACHE_SIZE / sb->s_blocksize;
+ block = group * 2;
+ pnum = block / blocks_per_page;
+ page = find_get_page(inode->i_mapping, pnum);
+ if (page != NULL) {
+ ClearPageUptodate(page);
+ page_cache_release(page);
+ }
+
+ /* Set bitmap page as not up to date */
+ block++;
+ pnum = block / blocks_per_page;
+ page = find_get_page(inode->i_mapping, pnum);
+ if (page != NULL) {
+ ClearPageUptodate(page);
+ page_cache_release(page);
+ }
+
+ return 0;
+}
+
+/*
+ * Update an existing group.
+ * This function is used for online resize
+ */
+void ext4_mb_update_group_info(struct ext4_group_info *grp, ext4_grpblk_t add)
+{
+ grp->bb_free += add;
+}
+
static int ext4_mb_init_backend(struct super_block *sb)
{
ext4_group_t i;
- int j, len, metalen;
+ int metalen;
struct ext4_sb_info *sbi = EXT4_SB(sb);
- int num_meta_group_infos =
- (sbi->s_groups_count + EXT4_DESC_PER_BLOCK(sb) - 1) >>
- EXT4_DESC_PER_BLOCK_BITS(sb);
+ struct ext4_super_block *es = sbi->s_es;
+ int num_meta_group_infos;
+ int num_meta_group_infos_max;
+ int array_size;
struct ext4_group_info **meta_group_info;
+ struct ext4_group_desc *desc;
+
+ /* This is the number of blocks used by GDT */
+ num_meta_group_infos = (sbi->s_groups_count + EXT4_DESC_PER_BLOCK(sb) -
+ 1) >> EXT4_DESC_PER_BLOCK_BITS(sb);
+ /*
+ * This is the total number of blocks used by GDT including
+ * the number of reserved blocks for GDT.
+ * The s_group_info array is allocated with this value
+ * to allow a clean online resize without a complex
+ * manipulation of pointer.
+ * The drawback is the unused memory when no resize
+ * occurs but it's very low in terms of pages
+ * (see comments below)
+ * Need to handle this properly when META_BG resizing is allowed
+ */
+ num_meta_group_infos_max = num_meta_group_infos +
+ le16_to_cpu(es->s_reserved_gdt_blocks);
+
+ /*
+ * array_size is the size of s_group_info array. We round it
+ * to the next power of two because this approximation is done
+ * internally by kmalloc so we can have some more memory
+ * for free here (e.g. may be used for META_BG resize).
+ */
+ array_size = 1;
+ while (array_size < sizeof(*sbi->s_group_info) *
+ num_meta_group_infos_max)
+ array_size = array_size << 1;
/* An 8TB filesystem with 64-bit pointers requires a 4096 byte
* kmalloc. A 128kb malloc should suffice for a 256TB filesystem.
* So a two level scheme suffices for now. */
- sbi->s_group_info = kmalloc(sizeof(*sbi->s_group_info) *
- num_meta_group_infos, GFP_KERNEL);
+ sbi->s_group_info = kmalloc(array_size, GFP_KERNEL);
if (sbi->s_group_info == NULL) {
printk(KERN_ERR "EXT4-fs: can't allocate buddy meta group\n");
return -ENOMEM;
@@ -2272,62 +2443,15 @@ static int ext4_mb_init_backend(struct super_block *sb)
sbi->s_group_info[i] = meta_group_info;
}
- /*
- * calculate needed size. if change bb_counters size,
- * don't forget about ext4_mb_generate_buddy()
- */
- len = sizeof(struct ext4_group_info);
- len += sizeof(unsigned short) * (sb->s_blocksize_bits + 2);
for (i = 0; i < sbi->s_groups_count; i++) {
- struct ext4_group_desc *desc;
-
- meta_group_info =
- sbi->s_group_info[i >> EXT4_DESC_PER_BLOCK_BITS(sb)];
- j = i & (EXT4_DESC_PER_BLOCK(sb) - 1);
-
- meta_group_info[j] = kzalloc(len, GFP_KERNEL);
- if (meta_group_info[j] == NULL) {
- printk(KERN_ERR "EXT4-fs: can't allocate buddy mem\n");
- goto err_freebuddy;
- }
desc = ext4_get_group_desc(sb, i, NULL);
if (desc == NULL) {
printk(KERN_ERR
"EXT4-fs: can't read descriptor %lu\n", i);
- i++;
goto err_freebuddy;
}
- set_bit(EXT4_GROUP_INFO_NEED_INIT_BIT,
- &(meta_group_info[j]->bb_state));
-
- /*
- * initialize bb_free to be able to skip
- * empty groups without initialization
- */
- if (desc->bg_flags & cpu_to_le16(EXT4_BG_BLOCK_UNINIT)) {
- meta_group_info[j]->bb_free =
- ext4_free_blocks_after_init(sb, i, desc);
- } else {
- meta_group_info[j]->bb_free =
- le16_to_cpu(desc->bg_free_blocks_count);
- }
-
- INIT_LIST_HEAD(&meta_group_info[j]->bb_prealloc_list);
-
-#ifdef DOUBLE_CHECK
- {
- struct buffer_head *bh;
- meta_group_info[j]->bb_bitmap =
- kmalloc(sb->s_blocksize, GFP_KERNEL);
- BUG_ON(meta_group_info[j]->bb_bitmap == NULL);
- bh = ext4_read_block_bitmap(sb, i);
- BUG_ON(bh == NULL);
- memcpy(meta_group_info[j]->bb_bitmap, bh->b_data,
- sb->s_blocksize);
- put_bh(bh);
- }
-#endif
-
+ if (ext4_mb_add_groupinfo(sb, i, desc) != 0)
+ goto err_freebuddy;
}
return 0;
diff --git a/fs/ext4/resize.c b/fs/ext4/resize.c
index 9ff7b1c..f000fbe 100644
--- a/fs/ext4/resize.c
+++ b/fs/ext4/resize.c
@@ -866,6 +866,15 @@ int ext4_group_add(struct super_block *sb, struct ext4_new_group_data *input)
gdp->bg_checksum = ext4_group_desc_csum(sbi, input->group, gdp);
/*
+ * We can allocate memory for mb_alloc based on the new group
+ * descriptor
+ */
+ if (test_opt(sb, MBALLOC)) {
+ err = ext4_mb_add_more_groupinfo(sb, input->group, gdp);
+ if (err)
+ goto exit_journal;
+ }
+ /*
* Make the new blocks and inodes valid next. We do this before
* increasing the group count so that once the group is enabled,
* all of its blocks and inodes are already valid.
@@ -957,6 +966,8 @@ int ext4_group_extend(struct super_block *sb, struct ext4_super_block *es,
handle_t *handle;
int err;
unsigned long freed_blocks;
+ ext4_group_t group;
+ struct ext4_group_info *grp;
/* We don't need to worry about locking wrt other resizers just
* yet: we're going to revalidate es->s_blocks_count after
@@ -988,7 +999,7 @@ int ext4_group_extend(struct super_block *sb, struct ext4_super_block *es,
}
/* Handle the remaining blocks in the last group only. */
- ext4_get_group_no_and_offset(sb, o_blocks_count, NULL, &last);
+ ext4_get_group_no_and_offset(sb, o_blocks_count, &group, &last);
if (last == 0) {
ext4_warning(sb, __func__,
@@ -1060,6 +1071,45 @@ int ext4_group_extend(struct super_block *sb, struct ext4_super_block *es,
o_blocks_count + add);
if ((err = ext4_journal_stop(handle)))
goto exit_put;
+
+ /*
+ * Mark mballoc pages as not up to date so that they will be updated
+ * next time they are loaded by ext4_mb_load_buddy.
+ */
+ if (test_opt(sb, MBALLOC)) {
+ struct ext4_sb_info *sbi = EXT4_SB(sb);
+ struct inode *inode = sbi->s_buddy_cache;
+ int blocks_per_page;
+ int block;
+ int pnum;
+ struct page *page;
+
+ /* Set buddy page as not up to date */
+ blocks_per_page = PAGE_CACHE_SIZE / sb->s_blocksize;
+ block = group * 2;
+ pnum = block / blocks_per_page;
+ page = find_get_page(inode->i_mapping, pnum);
+ if (page != NULL) {
+ ClearPageUptodate(page);
+ page_cache_release(page);
+ }
+
+ /* Set bitmap page as not up to date */
+ block++;
+ pnum = block / blocks_per_page;
+ page = find_get_page(inode->i_mapping, pnum);
+ if (page != NULL) {
+ ClearPageUptodate(page);
+ page_cache_release(page);
+ }
+
+ /* Get the info on the last group */
+ grp = ext4_get_group_info(sb, group);
+
+ /* Update free blocks in group info */
+ ext4_mb_update_group_info(grp, add);
+ }
+
if (test_opt(sb, DEBUG))
printk(KERN_DEBUG "EXT4-fs: extended group to %llu blocks\n",
ext4_blocks_count(es));
--
1.5.6.rc3.1.g36b7.dirty
next prev parent reply other threads:[~2008-07-05 17:42 UTC|newest]
Thread overview: 53+ messages / expand[flat|nested] mbox.gz Atom feed top
2008-07-05 17:35 Ext4 patches for the next merge window Theodore Ts'o
2008-07-05 17:35 ` [PATCH 01/52] ext4: fix comments to say "ext4" Theodore Ts'o
2008-07-05 17:35 ` [PATCH 02/52] ext4: start searching for the right extent from the goal group Theodore Ts'o
2008-07-05 17:35 ` [PATCH 03/52] ext4: Use BUG_ON() instead of BUG() Theodore Ts'o
2008-07-05 17:35 ` [PATCH 04/52] ext4: switch to seq_files Theodore Ts'o
2008-07-05 17:35 ` [PATCH 05/52] ext4: improve some code in rb tree part of dir.c Theodore Ts'o
2008-07-05 17:35 ` [PATCH 06/52] ext4: Fix ext4_mb_init_cache return error Theodore Ts'o
2008-07-05 17:35 ` [PATCH 07/52] ext4: add error processing when calling ext4_mb_init_cache in mballoc Theodore Ts'o
2008-07-05 17:35 ` [PATCH 08/52] ext4: miscellaneous error checks and coding cleanups for mballoc Theodore Ts'o
2008-07-05 17:35 ` [PATCH 09/52] ext4: remove double definitions of xattr macros Theodore Ts'o
2008-07-05 17:35 ` [PATCH 10/52] ext4: Rename read_block_bitmap() to ext4_read_block_bitmap() Theodore Ts'o
2008-07-05 17:35 ` [PATCH 11/52] ext4: Remove unused variable from ext4_show_options Theodore Ts'o
2008-07-05 17:35 ` [PATCH 12/52] jbd2: Add commit time into the commit block Theodore Ts'o
2008-07-05 17:35 ` [PATCH 13/52] ext4: New inode allocation for FLEX_BG meta-data groups Theodore Ts'o
2008-07-05 17:35 ` [PATCH 14/52] jbd2: fix race between jbd2_journal_try_to_free_buffers() and jbd2 commit transaction Theodore Ts'o
2008-07-05 17:35 ` [PATCH 15/52] ext4: remove redundant code in ext4_fill_super() Theodore Ts'o
2008-07-05 17:35 ` [PATCH 16/52] ext4: remove quota allocation when ext4_mb_new_blocks fails Theodore Ts'o
2008-07-05 17:35 ` [PATCH 17/52] ext4: Update i_disksize properly when allocating from fallocate area Theodore Ts'o
2008-07-05 17:35 ` [PATCH 18/52] ext4: return error when calling ext4_ext_split failed Theodore Ts'o
2008-07-05 17:35 ` [PATCH 19/52] ext4: Make ext4_ext_find_extent fills ext_path completely Theodore Ts'o
2008-07-05 17:35 ` [PATCH 20/52] ext4: Fix ext4_ext_journal_restart() to reflect errors up to the caller Theodore Ts'o
2008-07-05 17:35 ` [PATCH 21/52] ext4: cleanup never-used magic numbers from htree code Theodore Ts'o
2008-07-05 17:35 ` [PATCH 22/52] ext4: Fix sparse warning Theodore Ts'o
2008-07-05 17:35 ` [PATCH 23/52] ext4: fix ext4_init_block_bitmap() for metablock block group Theodore Ts'o
2008-07-05 17:35 ` [PATCH 24/52] ext4: Use inode preallocation with -o noextents Theodore Ts'o
2008-07-05 17:35 ` [PATCH 25/52] ext4: cleanup block allocator Theodore Ts'o
2008-07-05 17:35 ` [PATCH 26/52] ext4: call blkdev_issue_flush on fsync Theodore Ts'o
2008-07-05 17:35 ` [PATCH 27/52] ext4: mballoc avoid use root reserved blocks for non root allocation Theodore Ts'o
2008-07-05 17:35 ` [PATCH 28/52] ext4: Set journal pointer to NULL when journal is released Theodore Ts'o
2008-07-05 17:35 ` [PATCH 29/52] ext4: use atomic functions to set bh_state Theodore Ts'o
2008-07-05 17:35 ` [PATCH 30/52] ext4: Add missing unlock to an error path in ext4_quota_write() Theodore Ts'o
2008-07-05 17:35 ` Theodore Ts'o [this message]
2008-07-05 17:35 ` [PATCH 32/52] ext4: Documentation updates Theodore Ts'o
2008-07-05 17:35 ` [PATCH 33/52] ext4: Use page_mkwrite vma_operations to get mmap write notification Theodore Ts'o
2008-07-05 17:36 ` [PATCH 34/52] vfs: Move mark_inode_dirty() from under page lock in generic_write_end() Theodore Ts'o
2008-07-05 17:36 ` [PATCH 35/52] ext4: Invert the locking order of page_lock and transaction start Theodore Ts'o
2008-07-05 17:36 ` [PATCH 36/52] ext4: Fix lock inversion in ext4_ext_truncate() Theodore Ts'o
2008-07-05 17:36 ` [PATCH 37/52] vfs: export filemap_fdatawrite_range() Theodore Ts'o
2008-07-05 17:36 ` [PATCH 38/52] jbd2: Implement data=ordered mode handling via inodes Theodore Ts'o
2008-07-05 17:36 ` [PATCH 39/52] ext4: Use new framework for data=ordered mode in JBD2 Theodore Ts'o
2008-07-05 17:36 ` [PATCH 40/52] jbd2: Remove data=ordered mode support using jbd buffer heads Theodore Ts'o
2008-07-05 17:36 ` [PATCH 41/52] vfs: add basic delayed allocation support Theodore Ts'o
2008-07-05 17:36 ` [PATCH 42/52] ext4: Add " Theodore Ts'o
2008-07-05 17:36 ` [PATCH 43/52] percpu_counter: new function percpu_counter_sum_and_set Theodore Ts'o
2008-07-05 17:36 ` [PATCH 44/52] ext4: delayed allocation ENOSPC handling Theodore Ts'o
2008-07-05 17:36 ` [PATCH 45/52] mm: Add range_cont mode for writeback Theodore Ts'o
2008-07-05 17:36 ` [PATCH 46/52] ext4: Invert lock ordering of page_lock and transaction start in delalloc Theodore Ts'o
2008-07-05 17:36 ` [PATCH 47/52] ext4: Add ordered mode support for delalloc Theodore Ts'o
2008-07-05 17:36 ` [PATCH 48/52] ext4: Handle page without buffers in ext4_*_writepage() Theodore Ts'o
2008-07-05 17:36 ` [PATCH 49/52] ext4: fix delalloc i_disksize early update issue Theodore Ts'o
2008-07-05 17:36 ` [PATCH 50/52] ext4: Enable delalloc by default Theodore Ts'o
2008-07-05 17:36 ` [PATCH 51/52] ext4: Don't allow nonextenst mount option for large filesystem Theodore Ts'o
2008-07-05 17:36 ` [PATCH 52/52] ext4: Documention update for new ordered mode and delayed allocation Theodore Ts'o
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=1215279378-30504-32-git-send-email-tytso@mit.edu \
--to=tytso@mit.edu \
--cc=cmm@us.ibm.com \
--cc=frederic.bohe@bull.net \
--cc=linux-ext4@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®