Viewing: ext4-mballoc-for-hybrid.patch

commit 05e487a01b40da4e9f952eafa68bfde28074a385
Author:     Bobi Jam <bobijam@whamcloud.com>
AuthorDate: Mon Jul 10 19:40:34 2023 +0800

With LVM it is possible to create an LV with SSD storage at the
beginning of the LV and HDD storage at the end of the LV, and use that
to separate ext4 metadata allocations (that need small random IOs)
from data allocations (that are better suited for large sequential
IOs) depending on the type of underlying storage.  Between 0.5-1.0% of
the filesystem capacity would need to be high-IOPS storage in order to
hold all of the internal metadata.

This would improve performance for inode and other metadata access,
such as ls, find, e2fsck, and in general improve file access latency,
modification, truncate, unlink, transaction commit, etc.

This patch split largest free order group lists and average fragment
size lists into other two lists for IOPS/fast storage groups, and
CR_POWER2_ALIGNED / CR_GOAL_LEN_FAST group scanning for metadata
block allocation in following order:

if (allocate metadata blocks)
      if (cr == CR_POWER2_ALIGNED)
              try to find group in largest free order IOPS group list
      if (cr == CR_GOAL_LEN_FAST)
              try to find group in fragment size IOPS group list
      if (above two find failed)
              fall through normal group lists as before
if (allocate data blocks)
      try to find group in normal group lists as before
      if (failed to find group in normal group && mb_enable_iops_data)
              try to find group in IOPS groups

Non-metadata block allocation does not allocate from the IOPS groups
if non-IOPS groups are not used up.

Add for mke2fs an option to mark which blocks are in the IOPS region
of storage at format time:

  -E iops=0-1024G,4096-8192G

so the ext4 mballoc code can then use the EXT4_BG_IOPS flag in the
group descriptors to decide which groups to allocate dynamic
filesystem metadata.

--
v2->v3: add sysfs mb_enable_iops_data to enable data block allocation
        from IOPS groups.
v1->v2: for metadata block allocation, search in IOPS list then normal
        list.

Signed-off-by: Bobi Jam <bobijam@whamcloud.com>
Reviewed-by: Li Dongyang <dongyangli@ddn.com>
Reviewed-by: Andreas Dilger <adilger@whamcloud.com>
Reviewed-by: Oleg Drokin <green@whamcloud.com>
Change-Id: Ice2d25b8db19f67e70690f9ccebc419f253b12bd
Reviewed-on: https://review.whamcloud.com/51625


Index: linux-stage/fs/ext4/ext4.h
===================================================================
--- linux-stage.orig/fs/ext4/ext4.h
+++ linux-stage/fs/ext4/ext4.h
@@ -435,6 +435,7 @@ struct flex_groups {
 #define EXT4_BG_BLOCK_UNINIT	0x0002 /* Block bitmap not in use */
 #define EXT4_BG_INODE_ZEROED	0x0004 /* On-disk itable initialized to zero */
 #define EXT4_BG_TRIMMED		0x0008 /* block group was trimmed */
+#define EXT4_BG_IOPS		0x0010 /* In IOPS/fast storage */
 
 /*
  * Macro-instructions used to manage group descriptors
@@ -1184,6 +1185,8 @@ struct ext4_inode_info {
 	void *i_dirdata;
 };
 
+#define EXT2_FLAGS_HAS_IOPS		0x0080	/* has IOPS storage */
+
 /*
  * File system states
  */
@@ -1618,6 +1621,8 @@ struct ext4_sb_info {
 	atomic_t s_retry_alloc_pending;
 	struct xarray *s_mb_avg_fragment_size;
 	struct xarray *s_mb_largest_free_orders;
+	struct xarray *s_avg_fragment_size_iops;   /* avg_fragment_size for IOPS groups */
+	struct xarray *s_largest_free_orders_iops; /* largest_free_orders for IOPS groups */
 
 	/* tunables */
 	unsigned long s_stripe;
@@ -1638,6 +1643,7 @@ struct ext4_sb_info {
 	unsigned int s_mb_prefetch;
 	unsigned int s_mb_prefetch_limit;
 	unsigned int s_mb_best_avail_max_trim_order;
+	unsigned int s_mb_enable_iops_data;
 
 	/* stats for buddy allocator */
 	atomic_t s_bal_reqs;	/* number of reqs with len > 1 */
@@ -3686,6 +3692,7 @@ struct ext4_group_info {
 #define EXT4_GROUP_INFO_IBITMAP_CORRUPT		\
 	(1 << EXT4_GROUP_INFO_IBITMAP_CORRUPT_BIT)
 #define EXT4_GROUP_INFO_BBITMAP_READ_BIT	4
+#define EXT4_GROUP_INFO_IOPS_BIT		5
 
 #define EXT4_MB_GRP_NEED_INIT(grp)	\
 	(test_bit(EXT4_GROUP_INFO_NEED_INIT_BIT, &((grp)->bb_state)))
@@ -3695,6 +3702,10 @@ struct ext4_group_info {
 	(test_bit(EXT4_GROUP_INFO_IBITMAP_CORRUPT_BIT, &((grp)->bb_state)))
 #define EXT4_MB_GRP_TEST_AND_SET_READ(grp)	\
 	(test_and_set_bit(EXT4_GROUP_INFO_BBITMAP_READ_BIT, &((grp)->bb_state)))
+#define EXT4_MB_GRP_TEST_IOPS(grp)	\
+	(test_bit(EXT4_GROUP_INFO_IOPS_BIT, &((grp)->bb_state)))
+#define EXT4_MB_GRP_SET_IOPS(grp)	\
+	(set_bit(EXT4_GROUP_INFO_IOPS_BIT, &((grp)->bb_state)))
 
 #define EXT4_MAX_CONTENTION		8
 #define EXT4_CONTENTION_THRESHOLD	2
Index: linux-stage/fs/ext4/mballoc.c
===================================================================
--- linux-stage.orig/fs/ext4/mballoc.c
+++ linux-stage/fs/ext4/mballoc.c
@@ -863,6 +863,7 @@ static void
 mb_update_avg_fragment_size(struct super_block *sb, struct ext4_group_info *grp)
 {
 	struct ext4_sb_info *sbi = EXT4_SB(sb);
+	struct xarray *xa;
 	int new, old;
 
 	if (!test_opt2(sb, MB_OPTIMIZE_SCAN))
@@ -874,8 +875,12 @@ mb_update_avg_fragment_size(struct super
 	if (new == old)
 		return;
 
+	xa = sbi->s_es->s_flags & cpu_to_le32(EXT2_FLAGS_HAS_IOPS) &&
+	     EXT4_MB_GRP_TEST_IOPS(grp) ?
+	     sbi->s_avg_fragment_size_iops : sbi->s_mb_avg_fragment_size;
+
 	if (old >= 0)
-		xa_erase(&sbi->s_mb_avg_fragment_size[old], grp->bb_group);
+		xa_erase(&xa[old], grp->bb_group);
 
 	grp->bb_avg_fragment_size_order = new;
 	if (new >= 0) {
@@ -884,8 +889,7 @@ mb_update_avg_fragment_size(struct super
 		 * Although allocation for insertion may fails, it's not fatal
 		 * as we have linear traversal to fall back on.
 		 */
-		int err = xa_insert(&sbi->s_mb_avg_fragment_size[new],
-				    grp->bb_group, grp, GFP_ATOMIC);
+		int err = xa_insert(&xa[new], grp->bb_group, grp, GFP_ATOMIC);
 		if (err)
 			mb_debug(sb, "insert group: %u to s_mb_avg_fragment_size[%d] failed, err %d",
 				 grp->bb_group, new, err);
@@ -1142,6 +1146,71 @@ wrap_around:
 	return ret;
 }
 
+/*
+ * Choose next group from the IOPS largest-free-order xarrays, for metadata
+ * block allocation under CR_POWER2_ALIGNED.
+ */
+static int ext4_mb_scan_groups_iops_p2_aligned(struct ext4_allocation_context *ac,
+					       ext4_group_t group)
+{
+	struct ext4_sb_info *sbi = EXT4_SB(ac->ac_sb);
+	int i, ret = 0;
+	ext4_group_t start, end;
+
+	start = group;
+	end = ext4_get_allocation_groups_count(ac);
+wrap_around:
+	for (i = ac->ac_2order; i < MB_NUM_ORDERS(ac->ac_sb); i++) {
+		struct xarray *xa = &sbi->s_largest_free_orders_iops[i];
+
+		if (xa_empty(xa))
+			continue;
+		ret = ext4_mb_scan_groups_xa_range(ac, xa, start, end);
+		if (ret || ac->ac_status != AC_STATUS_CONTINUE)
+			return ret;
+	}
+	if (start) {
+		end = start;
+		start = 0;
+		goto wrap_around;
+	}
+
+	return ret;
+}
+
+/*
+ * Choose next group from the IOPS average-fragment-size xarrays, for
+ * metadata block allocation under CR_GOAL_LEN_FAST.
+ */
+static int ext4_mb_scan_groups_iops_goal_fast(struct ext4_allocation_context *ac,
+					      ext4_group_t group)
+{
+	struct ext4_sb_info *sbi = EXT4_SB(ac->ac_sb);
+	int i, ret = 0;
+	ext4_group_t start, end;
+
+	start = group;
+	end = ext4_get_allocation_groups_count(ac);
+wrap_around:
+	i = mb_avg_fragment_size_order(ac->ac_sb, ac->ac_g_ex.fe_len);
+	for (; i < MB_NUM_ORDERS(ac->ac_sb); i++) {
+		struct xarray *xa = &sbi->s_avg_fragment_size_iops[i];
+
+		if (xa_empty(xa))
+			continue;
+		ret = ext4_mb_scan_groups_xa_range(ac, xa, start, end);
+		if (ret || ac->ac_status != AC_STATUS_CONTINUE)
+			return ret;
+	}
+	if (start) {
+		end = start;
+		start = 0;
+		goto wrap_around;
+	}
+
+	return ret;
+}
+
 static inline int should_optimize_scan(struct ext4_allocation_context *ac)
 {
 	if (unlikely(!test_opt2(ac->ac_sb, MB_OPTIMIZE_SCAN)))
@@ -1218,6 +1287,17 @@ static int ext4_mb_scan_groups(struct ex
 	if (ret || ac->ac_status != AC_STATUS_CONTINUE)
 		return ret;
 
+	if (sbi->s_es->s_flags & cpu_to_le32(EXT2_FLAGS_HAS_IOPS) &&
+	    (ac->ac_flags & EXT4_MB_HINT_METADATA)) {
+		if (ac->ac_criteria == CR_POWER2_ALIGNED)
+			ret = ext4_mb_scan_groups_iops_p2_aligned(ac, start);
+		else if (ac->ac_criteria == CR_GOAL_LEN_FAST)
+			ret = ext4_mb_scan_groups_iops_goal_fast(ac, start);
+		if (ret || ac->ac_status != AC_STATUS_CONTINUE)
+			return ret;
+		cond_resched();
+	}
+
 	switch (ac->ac_criteria) {
 	case CR_POWER2_ALIGNED:
 		return ext4_mb_scan_groups_p2_aligned(ac, start);
@@ -1245,6 +1325,7 @@ static void
 mb_set_largest_free_order(struct super_block *sb, struct ext4_group_info *grp)
 {
 	struct ext4_sb_info *sbi = EXT4_SB(sb);
+	struct xarray *lfo;
 	int new, old = grp->bb_largest_free_order;
 
 	for (new = MB_NUM_ORDERS(sb) - 1; new >= 0; new--)
@@ -1255,8 +1336,12 @@ mb_set_largest_free_order(struct super_b
 	if (new == old)
 		return;
 
+	lfo = sbi->s_es->s_flags & cpu_to_le32(EXT2_FLAGS_HAS_IOPS) &&
+	      EXT4_MB_GRP_TEST_IOPS(grp) ?
+	      sbi->s_largest_free_orders_iops : sbi->s_mb_largest_free_orders;
+
 	if (old >= 0) {
-		struct xarray *xa = &sbi->s_mb_largest_free_orders[old];
+		struct xarray *xa = &lfo[old];
 
 		if (!xa_empty(xa) && xa_load(xa, grp->bb_group))
 			xa_erase(xa, grp->bb_group);
@@ -1269,8 +1354,7 @@ mb_set_largest_free_order(struct super_b
 		 * Although allocation for insertion may fails, it's not fatal
 		 * as we have linear traversal to fall back on.
 		 */
-		int err = xa_insert(&sbi->s_mb_largest_free_orders[new],
-				    grp->bb_group, grp, GFP_ATOMIC);
+		int err = xa_insert(&lfo[new], grp->bb_group, grp, GFP_ATOMIC);
 		if (err)
 			mb_debug(sb, "insert group: %u to s_mb_largest_free_orders[%d] failed, err %d",
 				 grp->bb_group, new, err);
@@ -2810,6 +2894,10 @@ static int ext4_mb_good_group_nolock(str
 		goto out;
 	if (unlikely(EXT4_MB_GRP_BBITMAP_CORRUPT(grp)))
 		goto out;
+	if (sbi->s_es->s_flags & cpu_to_le32(EXT2_FLAGS_HAS_IOPS) &&
+	    (ac->ac_flags & EXT4_MB_HINT_DATA) && EXT4_MB_GRP_TEST_IOPS(grp) &&
+	    !sbi->s_mb_enable_iops_data)
+		goto out;
 	if (should_lock) {
 		__acquire(ext4_group_lock_ptr(sb, group));
 		ext4_unlock_group(sb, group);
@@ -3640,6 +3728,9 @@ int ext4_mb_add_groupinfo(struct super_b
 	INIT_LIST_HEAD(&meta_group_info[i]->bb_prealloc_list);
 	init_rwsem(&meta_group_info[i]->alloc_sem);
 	meta_group_info[i]->bb_free_root = RB_ROOT;
+	if (sbi->s_es->s_flags & cpu_to_le32(EXT2_FLAGS_HAS_IOPS) &&
+	    desc->bg_flags & cpu_to_le16(EXT4_BG_IOPS))
+		EXT4_MB_GRP_SET_IOPS(meta_group_info[i]);
 	meta_group_info[i]->bb_largest_free_order = -1;  /* uninit */
 	meta_group_info[i]->bb_avg_fragment_size_order = -1;  /* uninit */
 	meta_group_info[i]->bb_group = group;
@@ -3872,6 +3963,30 @@ static inline void ext4_mb_largest_free_
 	sbi->s_mb_largest_free_orders = NULL;
 }
 
+static inline void ext4_mb_avg_fragment_size_iops_destroy(struct ext4_sb_info *sbi)
+{
+	if (!sbi->s_avg_fragment_size_iops)
+		return;
+
+	for (int i = 0; i < MB_NUM_ORDERS(sbi->s_sb); i++)
+		xa_destroy(&sbi->s_avg_fragment_size_iops[i]);
+
+	kfree(sbi->s_avg_fragment_size_iops);
+	sbi->s_avg_fragment_size_iops = NULL;
+}
+
+static inline void ext4_mb_largest_free_orders_iops_destroy(struct ext4_sb_info *sbi)
+{
+	if (!sbi->s_largest_free_orders_iops)
+		return;
+
+	for (int i = 0; i < MB_NUM_ORDERS(sbi->s_sb); i++)
+		xa_destroy(&sbi->s_largest_free_orders_iops[i]);
+
+	kfree(sbi->s_largest_free_orders_iops);
+	sbi->s_largest_free_orders_iops = NULL;
+}
+
 int ext4_mb_init(struct super_block *sb)
 {
 	struct ext4_sb_info *sbi = EXT4_SB(sb);
@@ -3926,6 +4041,18 @@ int ext4_mb_init(struct super_block *sb)
 	for (i = 0; i < MB_NUM_ORDERS(sb); i++)
 		xa_init(&sbi->s_mb_avg_fragment_size[i]);
 
+	if (sbi->s_es->s_flags & cpu_to_le32(EXT2_FLAGS_HAS_IOPS)) {
+		sbi->s_avg_fragment_size_iops =
+			kmalloc_array(MB_NUM_ORDERS(sb), sizeof(struct xarray),
+				GFP_KERNEL);
+		if (!sbi->s_avg_fragment_size_iops) {
+			ret = -ENOMEM;
+			goto out;
+		}
+		for (i = 0; i < MB_NUM_ORDERS(sb); i++)
+			xa_init(&sbi->s_avg_fragment_size_iops[i]);
+	}
+
 	sbi->s_mb_largest_free_orders =
 		kmalloc_array(MB_NUM_ORDERS(sb), sizeof(struct xarray),
 			GFP_KERNEL);
@@ -3936,6 +4063,18 @@ int ext4_mb_init(struct super_block *sb)
 	for (i = 0; i < MB_NUM_ORDERS(sb); i++)
 		xa_init(&sbi->s_mb_largest_free_orders[i]);
 
+	if (sbi->s_es->s_flags & cpu_to_le32(EXT2_FLAGS_HAS_IOPS)) {
+		sbi->s_largest_free_orders_iops =
+			kmalloc_array(MB_NUM_ORDERS(sb), sizeof(struct xarray),
+				GFP_KERNEL);
+		if (!sbi->s_largest_free_orders_iops) {
+			ret = -ENOMEM;
+			goto out;
+		}
+		for (i = 0; i < MB_NUM_ORDERS(sb); i++)
+			xa_init(&sbi->s_largest_free_orders_iops[i]);
+	}
+
 	spin_lock_init(&sbi->s_md_lock);
 	sbi->s_mb_free_pending = 0;
 	INIT_LIST_HEAD(&sbi->s_freed_data_list[0]);
@@ -3949,6 +4088,7 @@ int ext4_mb_init(struct super_block *sb)
 	sbi->s_mb_stats = MB_DEFAULT_STATS;
 	sbi->s_mb_order2_reqs = MB_DEFAULT_ORDER2_REQS;
 	sbi->s_mb_best_avail_max_trim_order = MB_DEFAULT_BEST_AVAIL_TRIM_ORDER;
+	sbi->s_mb_enable_iops_data = 0;
 
 	/*
 	 * The default group preallocation is 512, which for 4k block
@@ -4029,7 +4169,9 @@ out_free_locality_groups:
 	sbi->s_locality_groups = NULL;
 out:
 	ext4_mb_avg_fragment_size_destroy(sbi);
+	ext4_mb_avg_fragment_size_iops_destroy(sbi);
 	ext4_mb_largest_free_orders_destroy(sbi);
+	ext4_mb_largest_free_orders_iops_destroy(sbi);
 	kfree(sbi->s_mb_prealloc_table);
 	kfree(sbi->s_mb_offsets);
 	sbi->s_mb_offsets = NULL;
@@ -4094,7 +4236,9 @@ void ext4_mb_release(struct super_block
 		kvfree(group_info);
 	}
 	ext4_mb_avg_fragment_size_destroy(sbi);
+	ext4_mb_avg_fragment_size_iops_destroy(sbi);
 	ext4_mb_largest_free_orders_destroy(sbi);
+	ext4_mb_largest_free_orders_iops_destroy(sbi);
 	kfree(sbi->s_mb_prealloc_table);
 	kfree(sbi->s_mb_offsets);
 	kfree(sbi->s_mb_maxs);
Index: linux-stage/fs/ext4/balloc.c
===================================================================
--- linux-stage.orig/fs/ext4/balloc.c
+++ linux-stage/fs/ext4/balloc.c
@@ -737,7 +737,7 @@ ext4_fsblk_t ext4_new_meta_blocks(handle
 	ar.inode = inode;
 	ar.goal = goal;
 	ar.len = count ? *count : 1;
-	ar.flags = flags;
+	ar.flags = flags | EXT4_MB_HINT_METADATA;

 	ret = ext4_mb_new_blocks(handle, &ar, errp);
 	if (count)
Index: linux-stage/fs/ext4/extents.c
===================================================================
--- linux-stage.orig/fs/ext4/extents.c
+++ linux-stage/fs/ext4/extents.c
@@ -4675,11 +4675,12 @@ int ext4_ext_map_blocks(handle_t *handle
 	ar.len = EXT4_NUM_B2C(sbi, offset+allocated);
 	ar.goal -= offset;
 	ar.logical -= offset;
-	if (S_ISREG(inode->i_mode))
+	if (S_ISREG(inode->i_mode) &&
+	    !(EXT4_I(inode)->i_flags & EXT4_EA_INODE_FL))
 		ar.flags = EXT4_MB_HINT_DATA;
 	else
 		/* disable in-core preallocation for non-regular files */
-		ar.flags = 0;
+		ar.flags = EXT4_MB_HINT_METADATA;
 	if (flags & EXT4_GET_BLOCKS_NO_NORMALIZE)
 		ar.flags |= EXT4_MB_HINT_NOPREALLOC;
 	if (flags & EXT4_GET_BLOCKS_DELALLOC_RESERVE)
Index: linux-stage/fs/ext4/sysfs.c
===================================================================
--- linux-stage.orig/fs/ext4/sysfs.c
+++ linux-stage/fs/ext4/sysfs.c
@@ -250,6 +250,7 @@ EXT4_ATTR(journal_task, 0444, journal_ta
 EXT4_RW_ATTR_SBI_UI(mb_prefetch, s_mb_prefetch);
 EXT4_RW_ATTR_SBI_UI(mb_prefetch_limit, s_mb_prefetch_limit);
 EXT4_RW_ATTR_SBI_UL(last_trim_minblks, s_last_trim_minblks);
+EXT4_RW_ATTR_SBI_UI(mb_enable_iops_data, s_mb_enable_iops_data);

 static unsigned int old_bump_val = 128;
 EXT4_ATTR_PTR(max_writeback_mb_bump, 0444, pointer_ui, &old_bump_val);
@@ -305,6 +306,7 @@ static struct attribute *ext4_attrs[] =
 	ATTR_LIST(mb_prefetch),
 	ATTR_LIST(mb_prefetch_limit),
 	ATTR_LIST(last_trim_minblks),
+	ATTR_LIST(mb_enable_iops_data),
 	NULL,
 };
 ATTRIBUTE_GROUPS(ext4);
@@ -328,6 +330,7 @@ EXT4_ATTR_FEATURE(fast_commit);
 #if defined(CONFIG_UNICODE) && defined(CONFIG_FS_ENCRYPTION)
 EXT4_ATTR_FEATURE(encrypted_casefold);
 #endif
+EXT4_ATTR_FEATURE(iops);

 static struct attribute *ext4_feat_attrs[] = {
 	ATTR_LIST(lazy_itable_init),
@@ -348,6 +351,7 @@ static struct attribute *ext4_feat_attrs
 #if defined(CONFIG_UNICODE) && defined(CONFIG_FS_ENCRYPTION)
 	ATTR_LIST(encrypted_casefold),
 #endif
+	ATTR_LIST(iops),
 	NULL,
 };
 ATTRIBUTE_GROUPS(ext4_feat);
Index: linux-stage/fs/ext4/indirect.c
===================================================================
--- linux-stage.orig/fs/ext4/indirect.c
+++ linux-stage/fs/ext4/indirect.c
@@ -610,8 +610,11 @@ int ext4_ind_map_blocks(handle_t *handle
 	memset(&ar, 0, sizeof(ar));
 	ar.inode = inode;
 	ar.logical = map->m_lblk;
-	if (S_ISREG(inode->i_mode))
+	if (S_ISREG(inode->i_mode) &&
+	    !(EXT4_I(inode)->i_flags & EXT4_EA_INODE_FL))
 		ar.flags = EXT4_MB_HINT_DATA;
+	else
+		ar.flags = EXT4_MB_HINT_METADATA;
 	if (flags & EXT4_GET_BLOCKS_DELALLOC_RESERVE)
 		ar.flags |= EXT4_MB_DELALLOC_RESERVED;
 	if (flags & EXT4_GET_BLOCKS_METADATA_NOFAIL)