2017-12-18 06:00:59 +03:00
// SPDX-License-Identifier: GPL-2.0
2008-04-30 06:01:31 +04:00
/*
* fs / ext4 / mballoc . h
*
* Written by : Alex Tomas < alex @ clusterfs . com >
*
*/
# ifndef _EXT4_MBALLOC_H
# define _EXT4_MBALLOC_H
# include <linux/time.h>
# include <linux/fs.h>
# include <linux/namei.h>
# include <linux/quotaops.h>
# include <linux/buffer_head.h>
# include <linux/module.h>
# include <linux/swap.h>
# include <linux/proc_fs.h>
# include <linux/pagemap.h>
# include <linux/seq_file.h>
2008-10-16 18:06:27 +04:00
# include <linux/blkdev.h>
2009-01-06 05:36:19 +03:00
# include <linux/mutex.h>
2008-04-30 06:01:31 +04:00
# include "ext4_jbd2.h"
# include "ext4.h"
/*
2020-05-10 09:24:54 +03:00
* mb_debug ( ) dynamic printk msgs could be used to debug mballoc code .
2008-04-30 06:01:31 +04:00
*/
2009-09-18 21:38:55 +04:00
# ifdef CONFIG_EXT4_DEBUG
2020-05-10 09:24:54 +03:00
# define mb_debug(sb, fmt, ...) \
pr_debug ( " [%s/%d] EXT4-fs (%s): (%s, %d): %s: " fmt , \
current - > comm , task_pid_nr ( current ) , sb - > s_id , \
__FILE__ , __LINE__ , __func__ , # # __VA_ARGS__ )
2008-04-30 06:01:31 +04:00
# else
2020-05-10 09:24:54 +03:00
# define mb_debug(sb, fmt, ...) no_printk(fmt, ##__VA_ARGS__)
2008-04-30 06:01:31 +04:00
# endif
# define EXT4_MB_HISTORY_ALLOC 1 /* allocation */
# define EXT4_MB_HISTORY_PREALLOC 2 /* preallocated blocks used */
/*
* How long mballoc can look for a best extent ( in found extents )
*/
# define MB_DEFAULT_MAX_TO_SCAN 200
/*
* How long mballoc must look for a best extent
*/
# define MB_DEFAULT_MIN_TO_SCAN 10
/*
* with ' ext4_mb_stats ' allocator will collect stats that will be
* shown at umount . The collecting costs though !
*/
2009-09-29 23:51:30 +04:00
# define MB_DEFAULT_STATS 0
2008-04-30 06:01:31 +04:00
/*
* files smaller than MB_DEFAULT_STREAM_THRESHOLD are served
* by the stream allocator , which purpose is to pack requests
* as close each to other as possible to produce smooth I / O traffic
* We use locality group prealloc space for stream request .
* We can tune the same via / proc / fs / ext4 / < parition > / stream_req
*/
# define MB_DEFAULT_STREAM_THRESHOLD 16 /* 64K */
/*
* for which requests use 2 ^ N search using buddies
*/
# define MB_DEFAULT_ORDER2_REQS 2
/*
* default group prealloc size 512 blocks
*/
# define MB_DEFAULT_GROUP_PREALLOC 512
2020-08-17 10:36:15 +03:00
/*
* maximum length of inode prealloc list
*/
# define MB_DEFAULT_MAX_INODE_PREALLOC 512
2008-04-30 06:01:31 +04:00
ext4: improve cr 0 / cr 1 group scanning
Instead of traversing through groups linearly, scan groups in specific
orders at cr 0 and cr 1. At cr 0, we want to find groups that have the
largest free order >= the order of the request. So, with this patch,
we maintain lists for each possible order and insert each group into a
list based on the largest free order in its buddy bitmap. During cr 0
allocation, we traverse these lists in the increasing order of largest
free orders. This allows us to find a group with the best available cr
0 match in constant time. If nothing can be found, we fallback to cr 1
immediately.
At CR1, the story is slightly different. We want to traverse in the
order of increasing average fragment size. For CR1, we maintain a rb
tree of groupinfos which is sorted by average fragment size. Instead
of traversing linearly, at CR1, we traverse in the order of increasing
average fragment size, starting at the most optimal group. This brings
down cr 1 search complexity to log(num groups).
For cr >= 2, we just perform the linear search as before. Also, in
case of lock contention, we intermittently fallback to linear search
even in CR 0 and CR 1 cases. This allows us to proceed during the
allocation path even in case of high contention.
There is an opportunity to do optimization at CR2 too. That's because
at CR2 we only consider groups where bb_free counter (number of free
blocks) is greater than the request extent size. That's left as future
work.
All the changes introduced in this patch are protected under a new
mount option "mb_optimize_scan".
With this patchset, following experiment was performed:
Created a highly fragmented disk of size 65TB. The disk had no
contiguous 2M regions. Following command was run consecutively for 3
times:
time dd if=/dev/urandom of=file bs=2M count=10
Here are the results with and without cr 0/1 optimizations introduced
in this patch:
|---------+------------------------------+---------------------------|
| | Without CR 0/1 Optimizations | With CR 0/1 Optimizations |
|---------+------------------------------+---------------------------|
| 1st run | 5m1.871s | 2m47.642s |
| 2nd run | 2m28.390s | 0m0.611s |
| 3rd run | 2m26.530s | 0m1.255s |
|---------+------------------------------+---------------------------|
Signed-off-by: Harshad Shirwadkar <harshadshirwadkar@gmail.com>
Reported-by: kernel test robot <lkp@intel.com>
Reported-by: Dan Carpenter <dan.carpenter@oracle.com>
Reviewed-by: Andreas Dilger <adilger@dilger.ca>
Link: https://lore.kernel.org/r/20210401172129.189766-6-harshadshirwadkar@gmail.com
Signed-off-by: Theodore Ts'o <tytso@mit.edu>
2021-04-01 20:21:27 +03:00
/*
* Number of groups to search linearly before performing group scanning
* optimization .
*/
# define MB_DEFAULT_LINEAR_LIMIT 4
/*
* Minimum number of groups that should be present in the file system to perform
* group scanning optimizations .
*/
# define MB_DEFAULT_LINEAR_SCAN_THRESHOLD 16
2021-04-01 20:21:26 +03:00
/*
* Number of valid buddy orders
*/
# define MB_NUM_ORDERS(sb) ((sb)->s_blocksize_bits + 2)
2008-10-16 18:14:27 +04:00
struct ext4_free_data {
2017-06-23 06:54:33 +03:00
/* this links the free block information from sb_info */
struct list_head efd_list ;
2008-04-30 06:01:31 +04:00
2012-02-21 02:53:02 +04:00
/* this links the free block information from group_info */
struct rb_node efd_node ;
2008-10-16 18:14:27 +04:00
/* group which free block extent belongs */
2012-02-21 02:53:02 +04:00
ext4_group_t efd_group ;
2008-10-16 18:14:27 +04:00
/* free block extent */
2012-02-21 02:53:02 +04:00
ext4_grpblk_t efd_start_cluster ;
ext4_grpblk_t efd_count ;
2008-10-16 18:14:27 +04:00
/* transaction which freed this extent */
2012-02-21 02:53:02 +04:00
tid_t efd_tid ;
2008-04-30 06:01:31 +04:00
} ;
struct ext4_prealloc_space {
struct list_head pa_inode_list ;
struct list_head pa_group_list ;
union {
struct list_head pa_tmp_list ;
struct rcu_head pa_rcu ;
} u ;
spinlock_t pa_lock ;
atomic_t pa_count ;
unsigned pa_deleted ;
ext4_fsblk_t pa_pstart ; /* phys. block */
ext4_lblk_t pa_lstart ; /* log. block */
2009-08-26 06:36:45 +04:00
ext4_grpblk_t pa_len ; /* len of preallocated chunk */
ext4_grpblk_t pa_free ; /* how many blocks are free */
2009-03-28 00:16:58 +03:00
unsigned short pa_type ; /* pa type. inode or group */
2008-04-30 06:01:31 +04:00
spinlock_t * pa_obj_lock ;
struct inode * pa_inode ; /* hack, for history only */
} ;
2009-03-28 00:16:58 +03:00
enum {
MB_INODE_PA = 0 ,
MB_GROUP_PA = 1
} ;
2008-04-30 06:01:31 +04:00
struct ext4_free_extent {
ext4_lblk_t fe_logical ;
2011-09-10 02:48:51 +04:00
ext4_grpblk_t fe_start ; /* In cluster units */
2008-04-30 06:01:31 +04:00
ext4_group_t fe_group ;
2011-09-10 02:48:51 +04:00
ext4_grpblk_t fe_len ; /* In cluster units */
2008-04-30 06:01:31 +04:00
} ;
/*
* Locality group :
* we try to group all related changes together
* so that writeback can flush / allocate them together as well
2008-07-23 22:14:05 +04:00
* Size of lg_prealloc_list hash is determined by MB_DEFAULT_GROUP_PREALLOC
* ( 512 ) . We store prealloc space into the hash based on the pa_free blocks
* order value . ie , fls ( pa_free ) - 1 ;
2008-04-30 06:01:31 +04:00
*/
2008-07-23 22:14:05 +04:00
# define PREALLOC_TB_SIZE 10
2008-04-30 06:01:31 +04:00
struct ext4_locality_group {
/* for allocator */
2008-07-23 22:14:05 +04:00
/* to serialize allocates */
struct mutex lg_mutex ;
/* list of preallocations */
struct list_head lg_prealloc_list [ PREALLOC_TB_SIZE ] ;
2008-04-30 06:01:31 +04:00
spinlock_t lg_prealloc_lock ;
} ;
struct ext4_allocation_context {
struct inode * ac_inode ;
struct super_block * ac_sb ;
/* original request */
struct ext4_free_extent ac_o_ex ;
2011-02-24 22:10:00 +03:00
/* goal request (normalized ac_o_ex) */
2008-04-30 06:01:31 +04:00
struct ext4_free_extent ac_g_ex ;
/* the best found extent */
struct ext4_free_extent ac_b_ex ;
2011-11-01 02:55:50 +04:00
/* copy of the best found extent taken before preallocation efforts */
2008-04-30 06:01:31 +04:00
struct ext4_free_extent ac_f_ex ;
ext4: improve cr 0 / cr 1 group scanning
Instead of traversing through groups linearly, scan groups in specific
orders at cr 0 and cr 1. At cr 0, we want to find groups that have the
largest free order >= the order of the request. So, with this patch,
we maintain lists for each possible order and insert each group into a
list based on the largest free order in its buddy bitmap. During cr 0
allocation, we traverse these lists in the increasing order of largest
free orders. This allows us to find a group with the best available cr
0 match in constant time. If nothing can be found, we fallback to cr 1
immediately.
At CR1, the story is slightly different. We want to traverse in the
order of increasing average fragment size. For CR1, we maintain a rb
tree of groupinfos which is sorted by average fragment size. Instead
of traversing linearly, at CR1, we traverse in the order of increasing
average fragment size, starting at the most optimal group. This brings
down cr 1 search complexity to log(num groups).
For cr >= 2, we just perform the linear search as before. Also, in
case of lock contention, we intermittently fallback to linear search
even in CR 0 and CR 1 cases. This allows us to proceed during the
allocation path even in case of high contention.
There is an opportunity to do optimization at CR2 too. That's because
at CR2 we only consider groups where bb_free counter (number of free
blocks) is greater than the request extent size. That's left as future
work.
All the changes introduced in this patch are protected under a new
mount option "mb_optimize_scan".
With this patchset, following experiment was performed:
Created a highly fragmented disk of size 65TB. The disk had no
contiguous 2M regions. Following command was run consecutively for 3
times:
time dd if=/dev/urandom of=file bs=2M count=10
Here are the results with and without cr 0/1 optimizations introduced
in this patch:
|---------+------------------------------+---------------------------|
| | Without CR 0/1 Optimizations | With CR 0/1 Optimizations |
|---------+------------------------------+---------------------------|
| 1st run | 5m1.871s | 2m47.642s |
| 2nd run | 2m28.390s | 0m0.611s |
| 3rd run | 2m26.530s | 0m1.255s |
|---------+------------------------------+---------------------------|
Signed-off-by: Harshad Shirwadkar <harshadshirwadkar@gmail.com>
Reported-by: kernel test robot <lkp@intel.com>
Reported-by: Dan Carpenter <dan.carpenter@oracle.com>
Reviewed-by: Andreas Dilger <adilger@dilger.ca>
Link: https://lore.kernel.org/r/20210401172129.189766-6-harshadshirwadkar@gmail.com
Signed-off-by: Theodore Ts'o <tytso@mit.edu>
2021-04-01 20:21:27 +03:00
ext4_group_t ac_last_optimal_group ;
__u32 ac_groups_considered ;
__u32 ac_flags ; /* allocation hints */
2008-04-30 06:01:31 +04:00
__u16 ac_groups_scanned ;
ext4: improve cr 0 / cr 1 group scanning
Instead of traversing through groups linearly, scan groups in specific
orders at cr 0 and cr 1. At cr 0, we want to find groups that have the
largest free order >= the order of the request. So, with this patch,
we maintain lists for each possible order and insert each group into a
list based on the largest free order in its buddy bitmap. During cr 0
allocation, we traverse these lists in the increasing order of largest
free orders. This allows us to find a group with the best available cr
0 match in constant time. If nothing can be found, we fallback to cr 1
immediately.
At CR1, the story is slightly different. We want to traverse in the
order of increasing average fragment size. For CR1, we maintain a rb
tree of groupinfos which is sorted by average fragment size. Instead
of traversing linearly, at CR1, we traverse in the order of increasing
average fragment size, starting at the most optimal group. This brings
down cr 1 search complexity to log(num groups).
For cr >= 2, we just perform the linear search as before. Also, in
case of lock contention, we intermittently fallback to linear search
even in CR 0 and CR 1 cases. This allows us to proceed during the
allocation path even in case of high contention.
There is an opportunity to do optimization at CR2 too. That's because
at CR2 we only consider groups where bb_free counter (number of free
blocks) is greater than the request extent size. That's left as future
work.
All the changes introduced in this patch are protected under a new
mount option "mb_optimize_scan".
With this patchset, following experiment was performed:
Created a highly fragmented disk of size 65TB. The disk had no
contiguous 2M regions. Following command was run consecutively for 3
times:
time dd if=/dev/urandom of=file bs=2M count=10
Here are the results with and without cr 0/1 optimizations introduced
in this patch:
|---------+------------------------------+---------------------------|
| | Without CR 0/1 Optimizations | With CR 0/1 Optimizations |
|---------+------------------------------+---------------------------|
| 1st run | 5m1.871s | 2m47.642s |
| 2nd run | 2m28.390s | 0m0.611s |
| 3rd run | 2m26.530s | 0m1.255s |
|---------+------------------------------+---------------------------|
Signed-off-by: Harshad Shirwadkar <harshadshirwadkar@gmail.com>
Reported-by: kernel test robot <lkp@intel.com>
Reported-by: Dan Carpenter <dan.carpenter@oracle.com>
Reviewed-by: Andreas Dilger <adilger@dilger.ca>
Link: https://lore.kernel.org/r/20210401172129.189766-6-harshadshirwadkar@gmail.com
Signed-off-by: Theodore Ts'o <tytso@mit.edu>
2021-04-01 20:21:27 +03:00
__u16 ac_groups_linear_remaining ;
2008-04-30 06:01:31 +04:00
__u16 ac_found ;
__u16 ac_tail ;
__u16 ac_buddy ;
__u8 ac_status ;
__u8 ac_criteria ;
__u8 ac_2order ; /* if request is to allocate 2^N blocks and
* N > 0 , the field stores N , otherwise 0 */
__u8 ac_op ; /* operation, for history only */
struct page * ac_bitmap_page ;
struct page * ac_buddy_page ;
struct ext4_prealloc_space * ac_pa ;
struct ext4_locality_group * ac_lg ;
} ;
# define AC_STATUS_CONTINUE 1
# define AC_STATUS_FOUND 2
# define AC_STATUS_BREAK 3
struct ext4_buddy {
struct page * bd_buddy_page ;
void * bd_buddy ;
struct page * bd_bitmap_page ;
void * bd_bitmap ;
struct ext4_group_info * bd_info ;
struct super_block * bd_sb ;
__u16 bd_blkbits ;
ext4_group_t bd_group ;
} ;
2008-11-25 23:11:52 +03:00
static inline ext4_fsblk_t ext4_grp_offs_to_block ( struct super_block * sb ,
2008-04-30 06:01:31 +04:00
struct ext4_free_extent * fex )
{
2011-09-10 02:46:51 +04:00
return ext4_group_first_block_no ( sb , fex - > fe_group ) +
( fex - > fe_start < < EXT4_SB ( sb ) - > s_cluster_bits ) ;
2008-04-30 06:01:31 +04:00
}
2017-04-30 07:36:53 +03:00
typedef int ( * ext4_mballoc_query_range_fn ) (
struct super_block * sb ,
ext4_group_t agno ,
ext4_grpblk_t start ,
ext4_grpblk_t len ,
void * priv ) ;
int
ext4_mballoc_query_range (
struct super_block * sb ,
ext4_group_t agno ,
ext4_grpblk_t start ,
ext4_grpblk_t end ,
ext4_mballoc_query_range_fn formatter ,
void * priv ) ;
2008-04-30 06:01:31 +04:00
# endif