extent_io.h 17.8 KB
Newer Older
1
/* SPDX-License-Identifier: GPL-2.0 */
2 3 4

#ifndef BTRFS_EXTENT_IO_H
#define BTRFS_EXTENT_IO_H
5 6

#include <linux/rbtree.h>
7
#include <linux/refcount.h>
8
#include "ulist.h"
9 10

/* bits for the extent state */
11
#define EXTENT_DIRTY		(1U << 0)
N
Nikolay Borisov 已提交
12 13 14 15 16 17 18 19 20 21 22 23 24 25
#define EXTENT_UPTODATE		(1U << 1)
#define EXTENT_LOCKED		(1U << 2)
#define EXTENT_NEW		(1U << 3)
#define EXTENT_DELALLOC		(1U << 4)
#define EXTENT_DEFRAG		(1U << 5)
#define EXTENT_BOUNDARY		(1U << 6)
#define EXTENT_NODATASUM	(1U << 7)
#define EXTENT_CLEAR_META_RESV	(1U << 8)
#define EXTENT_NEED_WAIT	(1U << 9)
#define EXTENT_DAMAGED		(1U << 10)
#define EXTENT_NORESERVE	(1U << 11)
#define EXTENT_QGROUP_RESERVED	(1U << 12)
#define EXTENT_CLEAR_DATA_RESV	(1U << 13)
#define EXTENT_DELALLOC_NEW	(1U << 14)
26 27
#define EXTENT_DO_ACCOUNTING    (EXTENT_CLEAR_META_RESV | \
				 EXTENT_CLEAR_DATA_RESV)
28
#define EXTENT_CTLBITS		(EXTENT_DO_ACCOUNTING)
29

30 31 32
/* Redefined bits above which are used only in the device allocation tree */
#define CHUNK_ALLOCATED EXTENT_DIRTY

33 34 35 36
/*
 * flags for bio submission. The high bits indicate the compression
 * type for this bio
 */
C
Chris Mason 已提交
37
#define EXTENT_BIO_COMPRESSED 1
38
#define EXTENT_BIO_FLAG_SHIFT 16
C
Chris Mason 已提交
39

40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55
enum {
	EXTENT_BUFFER_UPTODATE,
	EXTENT_BUFFER_DIRTY,
	EXTENT_BUFFER_CORRUPT,
	/* this got triggered by readahead */
	EXTENT_BUFFER_READAHEAD,
	EXTENT_BUFFER_TREE_REF,
	EXTENT_BUFFER_STALE,
	EXTENT_BUFFER_WRITEBACK,
	/* read IO error */
	EXTENT_BUFFER_READ_ERR,
	EXTENT_BUFFER_UNMAPPED,
	EXTENT_BUFFER_IN_TREE,
	/* write IO error */
	EXTENT_BUFFER_WRITE_ERR,
};
56

57
/* these are flags for __process_pages_contig */
58 59 60 61 62
#define PAGE_UNLOCK		(1 << 0)
#define PAGE_CLEAR_DIRTY	(1 << 1)
#define PAGE_SET_WRITEBACK	(1 << 2)
#define PAGE_END_WRITEBACK	(1 << 3)
#define PAGE_SET_PRIVATE2	(1 << 4)
63
#define PAGE_SET_ERROR		(1 << 5)
64
#define PAGE_LOCK		(1 << 6)
65

66 67 68 69 70 71
/*
 * page->private values.  Every page that is controlled by the extent
 * map has page->private set to one.
 */
#define EXTENT_PAGE_PRIVATE 1

72 73 74 75 76 77 78 79 80 81 82 83 84 85
/*
 * The extent buffer bitmap operations are done with byte granularity instead of
 * word granularity for two reasons:
 * 1. The bitmaps must be little-endian on disk.
 * 2. Bitmap items are not guaranteed to be aligned to a word and therefore a
 *    single word in a bitmap may straddle two pages in the extent buffer.
 */
#define BIT_BYTE(nr) ((nr) / BITS_PER_BYTE)
#define BYTE_MASK ((1 << BITS_PER_BYTE) - 1)
#define BITMAP_FIRST_BYTE_MASK(start) \
	((BYTE_MASK << ((start) & (BITS_PER_BYTE - 1))) & BYTE_MASK)
#define BITMAP_LAST_BYTE_MASK(nbits) \
	(BYTE_MASK >> (-(nbits) & (BITS_PER_BYTE - 1)))

86
struct extent_state;
87
struct btrfs_root;
88
struct btrfs_inode;
89
struct btrfs_io_bio;
90
struct io_failure_record;
91

92
typedef	blk_status_t (extent_submit_bio_hook_t)(void *private_data, struct bio *bio,
93 94
				       int mirror_num, unsigned long bio_flags,
				       u64 bio_offset);
95 96

typedef blk_status_t (extent_submit_bio_start_t)(void *private_data,
97
		struct bio *bio, u64 bio_offset);
98

99
struct extent_io_ops {
100
	/*
101
	 * The following callbacks must be always defined, the function
102 103
	 * pointer will be called unconditionally.
	 */
104
	extent_submit_bio_hook_t *submit_bio_hook;
105 106 107
	int (*readpage_end_io_hook)(struct btrfs_io_bio *io_bio, u64 phy_offset,
				    struct page *page, u64 start, u64 end,
				    int mirror);
108 109
};

110 111 112 113 114 115 116 117 118 119 120
enum {
	IO_TREE_FS_INFO_FREED_EXTENTS0,
	IO_TREE_FS_INFO_FREED_EXTENTS1,
	IO_TREE_INODE_IO,
	IO_TREE_INODE_IO_FAILURE,
	IO_TREE_RELOC_BLOCKS,
	IO_TREE_TRANS_DIRTY_PAGES,
	IO_TREE_ROOT_DIRTY_LOG_PAGES,
	IO_TREE_SELFTEST,
};

121 122
struct extent_io_tree {
	struct rb_root state;
123
	struct btrfs_fs_info *fs_info;
124
	void *private_data;
125
	u64 dirty_bytes;
126
	bool track_uptodate;
127 128 129 130

	/* Who owns this io tree, should be one of IO_TREE_* */
	u8 owner;

131
	spinlock_t lock;
132
	const struct extent_io_ops *ops;
133 134 135 136 137 138
};

struct extent_state {
	u64 start;
	u64 end; /* inclusive */
	struct rb_node rb_node;
J
Josef Bacik 已提交
139 140

	/* ADD NEW ELEMENTS AFTER THIS */
141
	wait_queue_head_t wq;
142
	refcount_t refs;
143
	unsigned state;
144

145
	struct io_failure_record *failrec;
146

147
#ifdef CONFIG_BTRFS_DEBUG
148
	struct list_head leak_list;
149
#endif
150 151
};

152
#define INLINE_EXTENT_BUFFER_PAGES 16
153
#define MAX_INLINE_EXTENT_BUFFER_SIZE (INLINE_EXTENT_BUFFER_PAGES * PAGE_SIZE)
154 155 156
struct extent_buffer {
	u64 start;
	unsigned long len;
157
	unsigned long bflags;
158
	struct btrfs_fs_info *fs_info;
159
	spinlock_t refs_lock;
160
	atomic_t refs;
161
	atomic_t io_pages;
162
	int read_mirror;
163
	struct rcu_head rcu_head;
164
	pid_t lock_owner;
165

166 167
	atomic_t blocking_writers;
	atomic_t blocking_readers;
168
	bool lock_nested;
169 170
	/* >= 0 if eb belongs to a log tree, -1 otherwise */
	short log_index;
171 172 173 174 175 176 177 178

	/* protects write locks */
	rwlock_t lock;

	/* readers use lock_wq while they wait for the write
	 * lock holders to unlock
	 */
	wait_queue_head_t write_lock_wq;
179

180 181
	/* writers use read_lock_wq while they wait for readers
	 * to unlock
182
	 */
183
	wait_queue_head_t read_lock_wq;
184
	struct page *pages[INLINE_EXTENT_BUFFER_PAGES];
185
#ifdef CONFIG_BTRFS_DEBUG
186
	atomic_t spinning_writers;
187
	atomic_t spinning_readers;
188
	atomic_t read_locks;
189
	atomic_t write_locks;
190 191
	struct list_head leak_list;
#endif
192 193
};

194 195 196 197 198
/*
 * Structure to record how many bytes and which ranges are set/cleared
 */
struct extent_changeset {
	/* How many bytes are set/cleared in this operation */
199
	unsigned int bytes_changed;
200 201

	/* Changed ranges */
202
	struct ulist range_changed;
203 204
};

205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238
static inline void extent_changeset_init(struct extent_changeset *changeset)
{
	changeset->bytes_changed = 0;
	ulist_init(&changeset->range_changed);
}

static inline struct extent_changeset *extent_changeset_alloc(void)
{
	struct extent_changeset *ret;

	ret = kmalloc(sizeof(*ret), GFP_KERNEL);
	if (!ret)
		return NULL;

	extent_changeset_init(ret);
	return ret;
}

static inline void extent_changeset_release(struct extent_changeset *changeset)
{
	if (!changeset)
		return;
	changeset->bytes_changed = 0;
	ulist_release(&changeset->range_changed);
}

static inline void extent_changeset_free(struct extent_changeset *changeset)
{
	if (!changeset)
		return;
	extent_changeset_release(changeset);
	kfree(changeset);
}

239 240 241 242 243 244 245 246 247 248 249
static inline void extent_set_compress_type(unsigned long *bio_flags,
					    int compress_type)
{
	*bio_flags |= compress_type << EXTENT_BIO_FLAG_SHIFT;
}

static inline int extent_compress_type(unsigned long bio_flags)
{
	return bio_flags >> EXTENT_BIO_FLAG_SHIFT;
}

250 251
struct extent_map_tree;

252
typedef struct extent_map *(get_extent_t)(struct btrfs_inode *inode,
253
					  struct page *page,
254
					  size_t pg_offset,
255 256 257
					  u64 start, u64 len,
					  int create);

258
void extent_io_tree_init(struct btrfs_fs_info *fs_info,
259 260
			 struct extent_io_tree *tree, unsigned int owner,
			 void *private_data);
261
void extent_io_tree_release(struct extent_io_tree *tree);
262
int try_release_extent_mapping(struct page *page, gfp_t mask);
263
int try_release_extent_buffer(struct page *page);
264
int lock_extent_bits(struct extent_io_tree *tree, u64 start, u64 end,
265
		     struct extent_state **cached);
266 267 268 269 270 271

static inline int lock_extent(struct extent_io_tree *tree, u64 start, u64 end)
{
	return lock_extent_bits(tree, start, end, NULL);
}

272
int try_lock_extent(struct extent_io_tree *tree, u64 start, u64 end);
273
int extent_read_full_page(struct extent_io_tree *tree, struct page *page,
274
			  get_extent_t *get_extent, int mirror_num);
275
int __init extent_io_init(void);
276
void __cold extent_io_exit(void);
277 278 279

u64 count_range_bits(struct extent_io_tree *tree,
		     u64 *start, u64 search_end,
280
		     u64 max_bytes, unsigned bits, int contig);
281

282
void free_extent_state(struct extent_state *state);
283
int test_range_bit(struct extent_io_tree *tree, u64 start, u64 end,
284
		   unsigned bits, int filled,
285
		   struct extent_state *cached_state);
286
int clear_record_extent_bits(struct extent_io_tree *tree, u64 start, u64 end,
287
		unsigned bits, struct extent_changeset *changeset);
288
int clear_extent_bit(struct extent_io_tree *tree, u64 start, u64 end,
289
		     unsigned bits, int wake, int delete,
290
		     struct extent_state **cached);
291 292 293 294
int __clear_extent_bit(struct extent_io_tree *tree, u64 start, u64 end,
		     unsigned bits, int wake, int delete,
		     struct extent_state **cached, gfp_t mask,
		     struct extent_changeset *changeset);
295

296 297
static inline int unlock_extent(struct extent_io_tree *tree, u64 start, u64 end)
{
298
	return clear_extent_bit(tree, start, end, EXTENT_LOCKED, 1, 0, NULL);
299 300 301
}

static inline int unlock_extent_cached(struct extent_io_tree *tree, u64 start,
302
		u64 end, struct extent_state **cached)
303
{
304
	return __clear_extent_bit(tree, start, end, EXTENT_LOCKED, 1, 0, cached,
305
				GFP_NOFS, NULL);
306 307
}

308 309
static inline int unlock_extent_cached_atomic(struct extent_io_tree *tree,
		u64 start, u64 end, struct extent_state **cached)
310
{
311 312
	return __clear_extent_bit(tree, start, end, EXTENT_LOCKED, 1, 0, cached,
				GFP_ATOMIC, NULL);
313 314 315
}

static inline int clear_extent_bits(struct extent_io_tree *tree, u64 start,
316
		u64 end, unsigned bits)
317 318 319 320 321 322
{
	int wake = 0;

	if (bits & EXTENT_LOCKED)
		wake = 1;

323
	return clear_extent_bit(tree, start, end, bits, wake, 0, NULL);
324 325
}

326
int set_record_extent_bits(struct extent_io_tree *tree, u64 start, u64 end,
327
			   unsigned bits, struct extent_changeset *changeset);
328
int set_extent_bit(struct extent_io_tree *tree, u64 start, u64 end,
329
		   unsigned bits, u64 *failed_start,
330
		   struct extent_state **cached_state, gfp_t mask);
331 332
int set_extent_bits_nowait(struct extent_io_tree *tree, u64 start, u64 end,
			   unsigned bits);
333 334

static inline int set_extent_bits(struct extent_io_tree *tree, u64 start,
335
		u64 end, unsigned bits)
336
{
337
	return set_extent_bit(tree, start, end, bits, NULL, NULL, GFP_NOFS);
338 339
}

340
static inline int clear_extent_uptodate(struct extent_io_tree *tree, u64 start,
341
		u64 end, struct extent_state **cached_state)
342
{
343
	return __clear_extent_bit(tree, start, end, EXTENT_UPTODATE, 0, 0,
344
				cached_state, GFP_NOFS, NULL);
345
}
346 347 348 349 350 351 352 353

static inline int set_extent_dirty(struct extent_io_tree *tree, u64 start,
		u64 end, gfp_t mask)
{
	return set_extent_bit(tree, start, end, EXTENT_DIRTY, NULL,
			      NULL, mask);
}

354
static inline int clear_extent_dirty(struct extent_io_tree *tree, u64 start,
355
				     u64 end, struct extent_state **cached)
356 357 358
{
	return clear_extent_bit(tree, start, end,
				EXTENT_DIRTY | EXTENT_DELALLOC |
359
				EXTENT_DO_ACCOUNTING, 0, 0, cached);
360 361
}

J
Josef Bacik 已提交
362
int convert_extent_bit(struct extent_io_tree *tree, u64 start, u64 end,
363
		       unsigned bits, unsigned clear_bits,
364
		       struct extent_state **cached_state);
365 366

static inline int set_extent_delalloc(struct extent_io_tree *tree, u64 start,
367 368
				      u64 end, unsigned int extra_bits,
				      struct extent_state **cached_state)
369 370
{
	return set_extent_bit(tree, start, end,
371
			      EXTENT_DELALLOC | EXTENT_UPTODATE | extra_bits,
372
			      NULL, cached_state, GFP_NOFS);
373 374 375
}

static inline int set_extent_defrag(struct extent_io_tree *tree, u64 start,
376
		u64 end, struct extent_state **cached_state)
377 378 379
{
	return set_extent_bit(tree, start, end,
			      EXTENT_DELALLOC | EXTENT_UPTODATE | EXTENT_DEFRAG,
380
			      NULL, cached_state, GFP_NOFS);
381 382 383
}

static inline int set_extent_new(struct extent_io_tree *tree, u64 start,
384
		u64 end)
385
{
386 387
	return set_extent_bit(tree, start, end, EXTENT_NEW, NULL, NULL,
			GFP_NOFS);
388 389 390 391 392 393 394 395 396
}

static inline int set_extent_uptodate(struct extent_io_tree *tree, u64 start,
		u64 end, struct extent_state **cached_state, gfp_t mask)
{
	return set_extent_bit(tree, start, end, EXTENT_UPTODATE, NULL,
			      cached_state, mask);
}

397
int find_first_extent_bit(struct extent_io_tree *tree, u64 start,
398
			  u64 *start_ret, u64 *end_ret, unsigned bits,
399
			  struct extent_state **cached_state);
400 401
int extent_invalidatepage(struct extent_io_tree *tree,
			  struct page *page, unsigned long offset);
402
int extent_write_full_page(struct page *page, struct writeback_control *wbc);
403
int extent_write_locked_range(struct inode *inode, u64 start, u64 end,
404
			      int mode);
405
int extent_writepages(struct address_space *mapping,
406
		      struct writeback_control *wbc);
407 408
int btree_write_cache_pages(struct address_space *mapping,
			    struct writeback_control *wbc);
409 410
int extent_readpages(struct address_space *mapping, struct list_head *pages,
		     unsigned nr_pages);
Y
Yehuda Sadeh 已提交
411
int extent_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo,
412
		__u64 start, __u64 len);
413 414
void set_page_extent_mapped(struct page *page);

415
struct extent_buffer *alloc_extent_buffer(struct btrfs_fs_info *fs_info,
416
					  u64 start);
417 418
struct extent_buffer *__alloc_dummy_extent_buffer(struct btrfs_fs_info *fs_info,
						  u64 start, unsigned long len);
419
struct extent_buffer *alloc_dummy_extent_buffer(struct btrfs_fs_info *fs_info,
420
						u64 start);
421
struct extent_buffer *btrfs_clone_extent_buffer(struct extent_buffer *src);
422
struct extent_buffer *find_extent_buffer(struct btrfs_fs_info *fs_info,
423
					 u64 start);
424
void free_extent_buffer(struct extent_buffer *eb);
425
void free_extent_buffer_stale(struct extent_buffer *eb);
426 427 428
#define WAIT_NONE	0
#define WAIT_COMPLETE	1
#define WAIT_PAGE_LOCK	2
429
int read_extent_buffer_pages(struct extent_io_tree *tree,
430
			     struct extent_buffer *eb, int wait,
431
			     int mirror_num);
432
void wait_on_extent_buffer_writeback(struct extent_buffer *eb);
433

434
static inline int num_extent_pages(const struct extent_buffer *eb)
435
{
436 437
	return (round_up(eb->start + eb->len, PAGE_SIZE) >> PAGE_SHIFT) -
	       (eb->start >> PAGE_SHIFT);
438 439
}

440 441 442 443 444
static inline void extent_buffer_get(struct extent_buffer *eb)
{
	atomic_inc(&eb->refs);
}

445 446 447 448 449
static inline int extent_buffer_uptodate(struct extent_buffer *eb)
{
	return test_bit(EXTENT_BUFFER_UPTODATE, &eb->bflags);
}

450 451 452
int memcmp_extent_buffer(const struct extent_buffer *eb, const void *ptrv,
			 unsigned long start, unsigned long len);
void read_extent_buffer(const struct extent_buffer *eb, void *dst,
453 454
			unsigned long start,
			unsigned long len);
455 456
int read_extent_buffer_to_user(const struct extent_buffer *eb,
			       void __user *dst, unsigned long start,
457
			       unsigned long len);
458 459 460
void write_extent_buffer_fsid(struct extent_buffer *eb, const void *src);
void write_extent_buffer_chunk_tree_uuid(struct extent_buffer *eb,
		const void *src);
461 462
void write_extent_buffer(struct extent_buffer *eb, const void *src,
			 unsigned long start, unsigned long len);
463 464
void copy_extent_buffer_full(struct extent_buffer *dst,
			     struct extent_buffer *src);
465 466 467 468 469 470 471
void copy_extent_buffer(struct extent_buffer *dst, struct extent_buffer *src,
			unsigned long dst_offset, unsigned long src_offset,
			unsigned long len);
void memcpy_extent_buffer(struct extent_buffer *dst, unsigned long dst_offset,
			   unsigned long src_offset, unsigned long len);
void memmove_extent_buffer(struct extent_buffer *dst, unsigned long dst_offset,
			   unsigned long src_offset, unsigned long len);
472 473
void memzero_extent_buffer(struct extent_buffer *eb, unsigned long start,
			   unsigned long len);
474 475 476 477 478 479
int extent_buffer_test_bit(struct extent_buffer *eb, unsigned long start,
			   unsigned long pos);
void extent_buffer_bitmap_set(struct extent_buffer *eb, unsigned long start,
			      unsigned long pos, unsigned long len);
void extent_buffer_bitmap_clear(struct extent_buffer *eb, unsigned long start,
				unsigned long pos, unsigned long len);
480
void clear_extent_buffer_dirty(struct extent_buffer *eb);
481
bool set_extent_buffer_dirty(struct extent_buffer *eb);
482
void set_extent_buffer_uptodate(struct extent_buffer *eb);
483
void clear_extent_buffer_uptodate(struct extent_buffer *eb);
484
int extent_buffer_under_io(struct extent_buffer *eb);
485 486 487 488
int map_private_extent_buffer(const struct extent_buffer *eb,
			      unsigned long offset, unsigned long min_len,
			      char **map, unsigned long *map_start,
			      unsigned long *map_len);
489
void extent_range_clear_dirty_for_io(struct inode *inode, u64 start, u64 end);
490
void extent_range_redirty_for_io(struct inode *inode, u64 start, u64 end);
491
void extent_clear_unlock_delalloc(struct inode *inode, u64 start, u64 end,
492
				 u64 delalloc_end, struct page *locked_page,
493
				 unsigned bits_to_clear,
494
				 unsigned long page_ops);
495
struct bio *btrfs_bio_alloc(struct block_device *bdev, u64 first_byte);
496
struct bio *btrfs_io_bio_alloc(unsigned int nr_iovecs);
497
struct bio *btrfs_bio_clone(struct bio *bio);
498
struct bio *btrfs_bio_clone_partial(struct bio *orig, int offset, int size);
499

500
struct btrfs_fs_info;
501
struct btrfs_inode;
502

503 504 505
int repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 start,
		      u64 length, u64 logical, struct page *page,
		      unsigned int pg_offset, int mirror_num);
506 507 508 509
int clean_io_failure(struct btrfs_fs_info *fs_info,
		     struct extent_io_tree *failure_tree,
		     struct extent_io_tree *io_tree, u64 start,
		     struct page *page, u64 ino, unsigned int pg_offset);
510
void end_extent_writepage(struct page *page, int err, u64 start, u64 end);
511
int btrfs_repair_eb_io_failure(struct extent_buffer *eb, int mirror_num);
512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531

/*
 * When IO fails, either with EIO or csum verification fails, we
 * try other mirrors that might have a good copy of the data.  This
 * io_failure_record is used to record state as we go through all the
 * mirrors.  If another mirror has good data, the page is set up to date
 * and things continue.  If a good mirror can't be found, the original
 * bio end_io callback is called to indicate things have failed.
 */
struct io_failure_record {
	struct page *page;
	u64 start;
	u64 len;
	u64 logical;
	unsigned long bio_flags;
	int this_mirror;
	int failed_mirror;
	int in_validation;
};

532

533 534
void btrfs_free_io_failure_record(struct btrfs_inode *inode, u64 start,
		u64 end);
535 536
int btrfs_get_io_failure_record(struct inode *inode, u64 start, u64 end,
				struct io_failure_record **failrec_ret);
537
bool btrfs_check_repairable(struct inode *inode, unsigned failed_bio_pages,
538
			    struct io_failure_record *failrec, int fail_mirror);
539 540 541
struct bio *btrfs_create_repair_bio(struct inode *inode, struct bio *failed_bio,
				    struct io_failure_record *failrec,
				    struct page *page, int pg_offset, int icsum,
542
				    bio_end_io_t *endio_func, void *data);
543 544 545
int free_io_failure(struct extent_io_tree *failure_tree,
		    struct extent_io_tree *io_tree,
		    struct io_failure_record *rec);
546
#ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS
547
bool find_lock_delalloc_range(struct inode *inode, struct extent_io_tree *tree,
548 549
			     struct page *locked_page, u64 *start,
			     u64 *end);
550
#endif
551
struct extent_buffer *alloc_test_extent_buffer(struct btrfs_fs_info *fs_info,
552
					       u64 start);
553

554
#endif