inode.c 30.0 KB
Newer Older
R
Ryusuke Konishi 已提交
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24
/*
 * inode.c - NILFS inode operations.
 *
 * Copyright (C) 2005-2008 Nippon Telegraph and Telephone Corporation.
 *
 * This program is free software; you can redistribute it and/or modify
 * it under the terms of the GNU General Public License as published by
 * the Free Software Foundation; either version 2 of the License, or
 * (at your option) any later version.
 *
 * This program is distributed in the hope that it will be useful,
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
 * GNU General Public License for more details.
 *
 * You should have received a copy of the GNU General Public License
 * along with this program; if not, write to the Free Software
 * Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA  02110-1301  USA
 *
 * Written by Ryusuke Konishi <ryusuke@osrg.net>
 *
 */

#include <linux/buffer_head.h>
25
#include <linux/gfp.h>
R
Ryusuke Konishi 已提交
26
#include <linux/mpage.h>
27
#include <linux/pagemap.h>
R
Ryusuke Konishi 已提交
28
#include <linux/writeback.h>
29
#include <linux/uio.h>
R
Ryusuke Konishi 已提交
30
#include "nilfs.h"
A
Al Viro 已提交
31
#include "btnode.h"
R
Ryusuke Konishi 已提交
32 33 34 35 36 37
#include "segment.h"
#include "page.h"
#include "mdt.h"
#include "cpfile.h"
#include "ifile.h"

38 39 40 41 42 43 44
/**
 * struct nilfs_iget_args - arguments used during comparison between inodes
 * @ino: inode number
 * @cno: checkpoint number
 * @root: pointer on NILFS root object (mounted checkpoint)
 * @for_gc: inode for GC flag
 */
45 46 47
struct nilfs_iget_args {
	u64 ino;
	__u64 cno;
48
	struct nilfs_root *root;
49 50
	int for_gc;
};
R
Ryusuke Konishi 已提交
51

52 53
static int nilfs_iget_test(struct inode *inode, void *opaque);

54 55 56 57 58 59
void nilfs_inode_add_blocks(struct inode *inode, int n)
{
	struct nilfs_root *root = NILFS_I(inode)->i_root;

	inode_add_bytes(inode, (1 << inode->i_blkbits) * n);
	if (root)
60
		atomic64_add(n, &root->blocks_count);
61 62 63 64 65 66 67 68
}

void nilfs_inode_sub_blocks(struct inode *inode, int n)
{
	struct nilfs_root *root = NILFS_I(inode)->i_root;

	inode_sub_bytes(inode, (1 << inode->i_blkbits) * n);
	if (root)
69
		atomic64_sub(n, &root->blocks_count);
70 71
}

R
Ryusuke Konishi 已提交
72 73 74 75 76 77 78 79 80 81 82 83 84 85 86
/**
 * nilfs_get_block() - get a file block on the filesystem (callback function)
 * @inode - inode struct of the target file
 * @blkoff - file block number
 * @bh_result - buffer head to be mapped on
 * @create - indicate whether allocating the block or not when it has not
 *      been allocated yet.
 *
 * This function does not issue actual read request of the specified data
 * block. It is done by VFS.
 */
int nilfs_get_block(struct inode *inode, sector_t blkoff,
		    struct buffer_head *bh_result, int create)
{
	struct nilfs_inode_info *ii = NILFS_I(inode);
87
	struct the_nilfs *nilfs = inode->i_sb->s_fs_info;
88
	__u64 blknum = 0;
R
Ryusuke Konishi 已提交
89
	int err = 0, ret;
90
	unsigned maxblocks = bh_result->b_size >> inode->i_blkbits;
R
Ryusuke Konishi 已提交
91

92
	down_read(&NILFS_MDT(nilfs->ns_dat)->mi_sem);
93
	ret = nilfs_bmap_lookup_contig(ii->i_bmap, blkoff, &blknum, maxblocks);
94
	up_read(&NILFS_MDT(nilfs->ns_dat)->mi_sem);
95
	if (ret >= 0) {	/* found */
R
Ryusuke Konishi 已提交
96
		map_bh(bh_result, inode->i_sb, blknum);
97 98
		if (ret > 0)
			bh_result->b_size = (ret << inode->i_blkbits);
R
Ryusuke Konishi 已提交
99 100 101 102 103 104 105 106 107 108
		goto out;
	}
	/* data block was not found */
	if (ret == -ENOENT && create) {
		struct nilfs_transaction_info ti;

		bh_result->b_blocknr = 0;
		err = nilfs_transaction_begin(inode->i_sb, &ti, 1);
		if (unlikely(err))
			goto out;
109
		err = nilfs_bmap_insert(ii->i_bmap, blkoff,
R
Ryusuke Konishi 已提交
110 111 112 113 114 115 116 117 118
					(unsigned long)bh_result);
		if (unlikely(err != 0)) {
			if (err == -EEXIST) {
				/*
				 * The get_block() function could be called
				 * from multiple callers for an inode.
				 * However, the page having this block must
				 * be locked in this case.
				 */
119
				printk(KERN_WARNING
R
Ryusuke Konishi 已提交
120 121 122 123 124 125
				       "nilfs_get_block: a race condition "
				       "while inserting a data block. "
				       "(inode number=%lu, file block "
				       "offset=%llu)\n",
				       inode->i_ino,
				       (unsigned long long)blkoff);
126
				err = 0;
R
Ryusuke Konishi 已提交
127
			}
128
			nilfs_transaction_abort(inode->i_sb);
R
Ryusuke Konishi 已提交
129 130
			goto out;
		}
131
		nilfs_mark_inode_dirty_sync(inode);
132
		nilfs_transaction_commit(inode->i_sb); /* never fails */
R
Ryusuke Konishi 已提交
133 134
		/* Error handling should be detailed */
		set_buffer_new(bh_result);
135
		set_buffer_delay(bh_result);
R
Ryusuke Konishi 已提交
136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177
		map_bh(bh_result, inode->i_sb, 0); /* dbn must be changed
						      to proper value */
	} else if (ret == -ENOENT) {
		/* not found is not error (e.g. hole); must return without
		   the mapped state flag. */
		;
	} else {
		err = ret;
	}

 out:
	return err;
}

/**
 * nilfs_readpage() - implement readpage() method of nilfs_aops {}
 * address_space_operations.
 * @file - file struct of the file to be read
 * @page - the page to be read
 */
static int nilfs_readpage(struct file *file, struct page *page)
{
	return mpage_readpage(page, nilfs_get_block);
}

/**
 * nilfs_readpages() - implement readpages() method of nilfs_aops {}
 * address_space_operations.
 * @file - file struct of the file to be read
 * @mapping - address_space struct used for reading multiple pages
 * @pages - the pages to be read
 * @nr_pages - number of pages to be read
 */
static int nilfs_readpages(struct file *file, struct address_space *mapping,
			   struct list_head *pages, unsigned nr_pages)
{
	return mpage_readpages(mapping, pages, nr_pages, nilfs_get_block);
}

static int nilfs_writepages(struct address_space *mapping,
			    struct writeback_control *wbc)
{
178 179 180
	struct inode *inode = mapping->host;
	int err = 0;

181 182 183 184 185
	if (inode->i_sb->s_flags & MS_RDONLY) {
		nilfs_clear_dirty_pages(mapping, false);
		return -EROFS;
	}

186 187 188 189 190
	if (wbc->sync_mode == WB_SYNC_ALL)
		err = nilfs_construct_dsync_segment(inode->i_sb, inode,
						    wbc->range_start,
						    wbc->range_end);
	return err;
R
Ryusuke Konishi 已提交
191 192 193 194 195 196 197
}

static int nilfs_writepage(struct page *page, struct writeback_control *wbc)
{
	struct inode *inode = page->mapping->host;
	int err;

198
	if (inode->i_sb->s_flags & MS_RDONLY) {
199 200 201 202 203 204 205 206 207 208 209
		/*
		 * It means that filesystem was remounted in read-only
		 * mode because of error or metadata corruption. But we
		 * have dirty pages that try to be flushed in background.
		 * So, here we simply discard this dirty page.
		 */
		nilfs_clear_dirty_page(page, false);
		unlock_page(page);
		return -EROFS;
	}

R
Ryusuke Konishi 已提交
210 211 212 213 214 215 216 217 218 219 220 221 222 223 224
	redirty_page_for_writepage(wbc, page);
	unlock_page(page);

	if (wbc->sync_mode == WB_SYNC_ALL) {
		err = nilfs_construct_segment(inode->i_sb);
		if (unlikely(err))
			return err;
	} else if (wbc->for_reclaim)
		nilfs_flush_segment(inode->i_sb, inode->i_ino);

	return 0;
}

static int nilfs_set_page_dirty(struct page *page)
{
225
	struct inode *inode = page->mapping->host;
226
	int ret = __set_page_dirty_nobuffers(page);
R
Ryusuke Konishi 已提交
227

228 229 230
	if (page_has_buffers(page)) {
		unsigned nr_dirty = 0;
		struct buffer_head *bh, *head;
R
Ryusuke Konishi 已提交
231

232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250
		/*
		 * This page is locked by callers, and no other thread
		 * concurrently marks its buffers dirty since they are
		 * only dirtied through routines in fs/buffer.c in
		 * which call sites of mark_buffer_dirty are protected
		 * by page lock.
		 */
		bh = head = page_buffers(page);
		do {
			/* Do not mark hole blocks dirty */
			if (buffer_dirty(bh) || !buffer_mapped(bh))
				continue;

			set_buffer_dirty(bh);
			nr_dirty++;
		} while (bh = bh->b_this_page, bh != head);

		if (nr_dirty)
			nilfs_set_file_dirty(inode, nr_dirty);
251 252 253 254
	} else if (ret) {
		unsigned nr_dirty = 1 << (PAGE_CACHE_SHIFT - inode->i_blkbits);

		nilfs_set_file_dirty(inode, nr_dirty);
R
Ryusuke Konishi 已提交
255 256 257 258
	}
	return ret;
}

M
Marco Stornelli 已提交
259 260 261 262 263
void nilfs_write_failed(struct address_space *mapping, loff_t to)
{
	struct inode *inode = mapping->host;

	if (to > inode->i_size) {
264
		truncate_pagecache(inode, inode->i_size);
M
Marco Stornelli 已提交
265 266 267 268
		nilfs_truncate(inode);
	}
}

R
Ryusuke Konishi 已提交
269 270 271 272 273 274 275 276 277 278 279
static int nilfs_write_begin(struct file *file, struct address_space *mapping,
			     loff_t pos, unsigned len, unsigned flags,
			     struct page **pagep, void **fsdata)

{
	struct inode *inode = mapping->host;
	int err = nilfs_transaction_begin(inode->i_sb, NULL, 1);

	if (unlikely(err))
		return err;

280 281 282
	err = block_write_begin(mapping, pos, len, flags, pagep,
				nilfs_get_block);
	if (unlikely(err)) {
M
Marco Stornelli 已提交
283
		nilfs_write_failed(mapping, pos + len);
284
		nilfs_transaction_abort(inode->i_sb);
285
	}
R
Ryusuke Konishi 已提交
286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301
	return err;
}

static int nilfs_write_end(struct file *file, struct address_space *mapping,
			   loff_t pos, unsigned len, unsigned copied,
			   struct page *page, void *fsdata)
{
	struct inode *inode = mapping->host;
	unsigned start = pos & (PAGE_CACHE_SIZE - 1);
	unsigned nr_dirty;
	int err;

	nr_dirty = nilfs_page_count_clean_buffers(page, start,
						  start + copied);
	copied = generic_write_end(file, mapping, pos, len, copied, page,
				   fsdata);
302
	nilfs_set_file_dirty(inode, nr_dirty);
303
	err = nilfs_transaction_commit(inode->i_sb);
R
Ryusuke Konishi 已提交
304 305 306 307
	return err ? : copied;
}

static ssize_t
A
Al Viro 已提交
308 309
nilfs_direct_IO(int rw, struct kiocb *iocb, struct iov_iter *iter,
		loff_t offset)
R
Ryusuke Konishi 已提交
310 311
{
	struct file *file = iocb->ki_filp;
M
Marco Stornelli 已提交
312
	struct address_space *mapping = file->f_mapping;
R
Ryusuke Konishi 已提交
313
	struct inode *inode = file->f_mapping->host;
314
	size_t count = iov_iter_count(iter);
R
Ryusuke Konishi 已提交
315 316 317 318 319 320
	ssize_t size;

	if (rw == WRITE)
		return 0;

	/* Needs synchronization with the cleaner */
321 322
	size = blockdev_direct_IO(rw, iocb, inode, iter, offset,
				  nilfs_get_block);
323 324 325 326 327 328 329

	/*
	 * In case of error extending write may have instantiated a few
	 * blocks outside i_size. Trim these off again.
	 */
	if (unlikely((rw & WRITE) && size < 0)) {
		loff_t isize = i_size_read(inode);
330
		loff_t end = offset + count;
331 332

		if (end > isize)
M
Marco Stornelli 已提交
333
			nilfs_write_failed(mapping, end);
334 335
	}

R
Ryusuke Konishi 已提交
336 337 338
	return size;
}

339
const struct address_space_operations nilfs_aops = {
R
Ryusuke Konishi 已提交
340 341 342 343 344 345 346 347 348 349
	.writepage		= nilfs_writepage,
	.readpage		= nilfs_readpage,
	.writepages		= nilfs_writepages,
	.set_page_dirty		= nilfs_set_page_dirty,
	.readpages		= nilfs_readpages,
	.write_begin		= nilfs_write_begin,
	.write_end		= nilfs_write_end,
	/* .releasepage		= nilfs_releasepage, */
	.invalidatepage		= block_invalidatepage,
	.direct_IO		= nilfs_direct_IO,
350
	.is_partially_uptodate  = block_is_partially_uptodate,
R
Ryusuke Konishi 已提交
351 352
};

353 354 355 356 357 358 359 360 361 362 363
static int nilfs_insert_inode_locked(struct inode *inode,
				     struct nilfs_root *root,
				     unsigned long ino)
{
	struct nilfs_iget_args args = {
		.ino = ino, .root = root, .cno = 0, .for_gc = 0
	};

	return insert_inode_locked4(inode, ino, nilfs_iget_test, &args);
}

A
Al Viro 已提交
364
struct inode *nilfs_new_inode(struct inode *dir, umode_t mode)
R
Ryusuke Konishi 已提交
365 366
{
	struct super_block *sb = dir->i_sb;
367
	struct the_nilfs *nilfs = sb->s_fs_info;
R
Ryusuke Konishi 已提交
368 369
	struct inode *inode;
	struct nilfs_inode_info *ii;
370
	struct nilfs_root *root;
R
Ryusuke Konishi 已提交
371 372 373 374 375 376 377 378 379 380
	int err = -ENOMEM;
	ino_t ino;

	inode = new_inode(sb);
	if (unlikely(!inode))
		goto failed;

	mapping_set_gfp_mask(inode->i_mapping,
			     mapping_gfp_mask(inode->i_mapping) & ~__GFP_FS);

381
	root = NILFS_I(dir)->i_root;
R
Ryusuke Konishi 已提交
382 383
	ii = NILFS_I(inode);
	ii->i_state = 1 << NILFS_I_NEW;
384
	ii->i_root = root;
R
Ryusuke Konishi 已提交
385

386
	err = nilfs_ifile_create_inode(root->ifile, &ino, &ii->i_bh);
R
Ryusuke Konishi 已提交
387 388 389 390
	if (unlikely(err))
		goto failed_ifile_create_inode;
	/* reference count of i_bh inherits from nilfs_mdt_read_block() */

391
	atomic64_inc(&root->inodes_count);
392
	inode_init_owner(inode, dir, mode);
R
Ryusuke Konishi 已提交
393 394 395 396 397 398
	inode->i_ino = ino;
	inode->i_mtime = inode->i_atime = inode->i_ctime = CURRENT_TIME;

	if (S_ISREG(mode) || S_ISDIR(mode) || S_ISLNK(mode)) {
		err = nilfs_bmap_read(ii->i_bmap, NULL);
		if (err < 0)
399
			goto failed_after_creation;
R
Ryusuke Konishi 已提交
400 401 402 403 404

		set_bit(NILFS_I_BMAP, &ii->i_state);
		/* No lock is needed; iget() ensures it. */
	}

405 406
	ii->i_flags = nilfs_mask_flags(
		mode, NILFS_I(dir)->i_flags & NILFS_FL_INHERITED);
R
Ryusuke Konishi 已提交
407 408 409 410 411

	/* ii->i_file_acl = 0; */
	/* ii->i_dir_acl = 0; */
	ii->i_dir_start_lookup = 0;
	nilfs_set_inode_flags(inode);
412 413 414
	spin_lock(&nilfs->ns_next_gen_lock);
	inode->i_generation = nilfs->ns_next_generation++;
	spin_unlock(&nilfs->ns_next_gen_lock);
415 416 417 418
	if (nilfs_insert_inode_locked(inode, root, ino) < 0) {
		err = -EIO;
		goto failed_after_creation;
	}
R
Ryusuke Konishi 已提交
419 420 421

	err = nilfs_init_acl(inode, dir);
	if (unlikely(err))
422
		goto failed_after_creation; /* never occur. When supporting
R
Ryusuke Konishi 已提交
423 424 425 426 427
				    nilfs_init_acl(), proper cancellation of
				    above jobs should be considered */

	return inode;

428
 failed_after_creation:
429
	clear_nlink(inode);
430
	unlock_new_inode(inode);
R
Ryusuke Konishi 已提交
431
	iput(inode);  /* raw_inode will be deleted through
432
			 nilfs_evict_inode() */
R
Ryusuke Konishi 已提交
433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448
	goto failed;

 failed_ifile_create_inode:
	make_bad_inode(inode);
	iput(inode);  /* if i_nlink == 1, generic_forget_inode() will be
			 called */
 failed:
	return ERR_PTR(err);
}

void nilfs_set_inode_flags(struct inode *inode)
{
	unsigned int flags = NILFS_I(inode)->i_flags;

	inode->i_flags &= ~(S_SYNC | S_APPEND | S_IMMUTABLE | S_NOATIME |
			    S_DIRSYNC);
449
	if (flags & FS_SYNC_FL)
R
Ryusuke Konishi 已提交
450
		inode->i_flags |= S_SYNC;
451
	if (flags & FS_APPEND_FL)
R
Ryusuke Konishi 已提交
452
		inode->i_flags |= S_APPEND;
453
	if (flags & FS_IMMUTABLE_FL)
R
Ryusuke Konishi 已提交
454
		inode->i_flags |= S_IMMUTABLE;
455
	if (flags & FS_NOATIME_FL)
R
Ryusuke Konishi 已提交
456
		inode->i_flags |= S_NOATIME;
457
	if (flags & FS_DIRSYNC_FL)
R
Ryusuke Konishi 已提交
458 459 460 461 462 463 464 465 466 467 468 469
		inode->i_flags |= S_DIRSYNC;
	mapping_set_gfp_mask(inode->i_mapping,
			     mapping_gfp_mask(inode->i_mapping) & ~__GFP_FS);
}

int nilfs_read_inode_common(struct inode *inode,
			    struct nilfs_inode *raw_inode)
{
	struct nilfs_inode_info *ii = NILFS_I(inode);
	int err;

	inode->i_mode = le16_to_cpu(raw_inode->i_mode);
470 471
	i_uid_write(inode, le32_to_cpu(raw_inode->i_uid));
	i_gid_write(inode, le32_to_cpu(raw_inode->i_gid));
M
Miklos Szeredi 已提交
472
	set_nlink(inode, le16_to_cpu(raw_inode->i_links_count));
R
Ryusuke Konishi 已提交
473 474 475 476
	inode->i_size = le64_to_cpu(raw_inode->i_size);
	inode->i_atime.tv_sec = le64_to_cpu(raw_inode->i_mtime);
	inode->i_ctime.tv_sec = le64_to_cpu(raw_inode->i_ctime);
	inode->i_mtime.tv_sec = le64_to_cpu(raw_inode->i_mtime);
477 478 479
	inode->i_atime.tv_nsec = le32_to_cpu(raw_inode->i_mtime_nsec);
	inode->i_ctime.tv_nsec = le32_to_cpu(raw_inode->i_ctime_nsec);
	inode->i_mtime.tv_nsec = le32_to_cpu(raw_inode->i_mtime_nsec);
480 481
	if (inode->i_nlink == 0)
		return -ESTALE; /* this inode is deleted */
R
Ryusuke Konishi 已提交
482 483 484 485 486 487 488 489

	inode->i_blocks = le64_to_cpu(raw_inode->i_blocks);
	ii->i_flags = le32_to_cpu(raw_inode->i_flags);
#if 0
	ii->i_file_acl = le32_to_cpu(raw_inode->i_file_acl);
	ii->i_dir_acl = S_ISREG(inode->i_mode) ?
		0 : le32_to_cpu(raw_inode->i_dir_acl);
#endif
490
	ii->i_dir_start_lookup = 0;
R
Ryusuke Konishi 已提交
491 492 493 494 495 496 497 498 499 500 501 502 503
	inode->i_generation = le32_to_cpu(raw_inode->i_generation);

	if (S_ISREG(inode->i_mode) || S_ISDIR(inode->i_mode) ||
	    S_ISLNK(inode->i_mode)) {
		err = nilfs_bmap_read(ii->i_bmap, raw_inode);
		if (err < 0)
			return err;
		set_bit(NILFS_I_BMAP, &ii->i_state);
		/* No lock is needed; iget() ensures it. */
	}
	return 0;
}

504 505
static int __nilfs_read_inode(struct super_block *sb,
			      struct nilfs_root *root, unsigned long ino,
R
Ryusuke Konishi 已提交
506 507
			      struct inode *inode)
{
508
	struct the_nilfs *nilfs = sb->s_fs_info;
R
Ryusuke Konishi 已提交
509 510 511 512
	struct buffer_head *bh;
	struct nilfs_inode *raw_inode;
	int err;

513
	down_read(&NILFS_MDT(nilfs->ns_dat)->mi_sem);
514
	err = nilfs_ifile_get_inode_block(root->ifile, ino, &bh);
R
Ryusuke Konishi 已提交
515 516 517
	if (unlikely(err))
		goto bad_inode;

518
	raw_inode = nilfs_ifile_map_inode(root->ifile, ino, bh);
R
Ryusuke Konishi 已提交
519

520 521
	err = nilfs_read_inode_common(inode, raw_inode);
	if (err)
R
Ryusuke Konishi 已提交
522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538
		goto failed_unmap;

	if (S_ISREG(inode->i_mode)) {
		inode->i_op = &nilfs_file_inode_operations;
		inode->i_fop = &nilfs_file_operations;
		inode->i_mapping->a_ops = &nilfs_aops;
	} else if (S_ISDIR(inode->i_mode)) {
		inode->i_op = &nilfs_dir_inode_operations;
		inode->i_fop = &nilfs_dir_operations;
		inode->i_mapping->a_ops = &nilfs_aops;
	} else if (S_ISLNK(inode->i_mode)) {
		inode->i_op = &nilfs_symlink_inode_operations;
		inode->i_mapping->a_ops = &nilfs_aops;
	} else {
		inode->i_op = &nilfs_special_inode_operations;
		init_special_inode(
			inode, inode->i_mode,
539
			huge_decode_dev(le64_to_cpu(raw_inode->i_device_code)));
R
Ryusuke Konishi 已提交
540
	}
541
	nilfs_ifile_unmap_inode(root->ifile, ino, bh);
R
Ryusuke Konishi 已提交
542
	brelse(bh);
543
	up_read(&NILFS_MDT(nilfs->ns_dat)->mi_sem);
R
Ryusuke Konishi 已提交
544 545 546 547
	nilfs_set_inode_flags(inode);
	return 0;

 failed_unmap:
548
	nilfs_ifile_unmap_inode(root->ifile, ino, bh);
R
Ryusuke Konishi 已提交
549 550 551
	brelse(bh);

 bad_inode:
552
	up_read(&NILFS_MDT(nilfs->ns_dat)->mi_sem);
R
Ryusuke Konishi 已提交
553 554 555
	return err;
}

556 557 558 559 560
static int nilfs_iget_test(struct inode *inode, void *opaque)
{
	struct nilfs_iget_args *args = opaque;
	struct nilfs_inode_info *ii;

561
	if (args->ino != inode->i_ino || args->root != NILFS_I(inode)->i_root)
562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578
		return 0;

	ii = NILFS_I(inode);
	if (!test_bit(NILFS_I_GCINODE, &ii->i_state))
		return !args->for_gc;

	return args->for_gc && args->cno == ii->i_cno;
}

static int nilfs_iget_set(struct inode *inode, void *opaque)
{
	struct nilfs_iget_args *args = opaque;

	inode->i_ino = args->ino;
	if (args->for_gc) {
		NILFS_I(inode)->i_state = 1 << NILFS_I_GCINODE;
		NILFS_I(inode)->i_cno = args->cno;
579 580 581 582 583
		NILFS_I(inode)->i_root = NULL;
	} else {
		if (args->root && args->ino == NILFS_ROOT_INO)
			nilfs_get_root(args->root);
		NILFS_I(inode)->i_root = args->root;
584 585 586 587
	}
	return 0;
}

588 589 590 591 592 593 594 595 596 597
struct inode *nilfs_ilookup(struct super_block *sb, struct nilfs_root *root,
			    unsigned long ino)
{
	struct nilfs_iget_args args = {
		.ino = ino, .root = root, .cno = 0, .for_gc = 0
	};

	return ilookup5(sb, ino, nilfs_iget_test, &args);
}

598 599
struct inode *nilfs_iget_locked(struct super_block *sb, struct nilfs_root *root,
				unsigned long ino)
R
Ryusuke Konishi 已提交
600
{
601 602 603
	struct nilfs_iget_args args = {
		.ino = ino, .root = root, .cno = 0, .for_gc = 0
	};
604 605 606 607 608 609 610

	return iget5_locked(sb, ino, nilfs_iget_test, nilfs_iget_set, &args);
}

struct inode *nilfs_iget(struct super_block *sb, struct nilfs_root *root,
			 unsigned long ino)
{
R
Ryusuke Konishi 已提交
611 612 613
	struct inode *inode;
	int err;

614
	inode = nilfs_iget_locked(sb, root, ino);
R
Ryusuke Konishi 已提交
615 616 617 618 619
	if (unlikely(!inode))
		return ERR_PTR(-ENOMEM);
	if (!(inode->i_state & I_NEW))
		return inode;

620
	err = __nilfs_read_inode(sb, root, ino, inode);
R
Ryusuke Konishi 已提交
621 622 623 624 625 626 627 628
	if (unlikely(err)) {
		iget_failed(inode);
		return ERR_PTR(err);
	}
	unlock_new_inode(inode);
	return inode;
}

629 630 631
struct inode *nilfs_iget_for_gc(struct super_block *sb, unsigned long ino,
				__u64 cno)
{
632 633 634
	struct nilfs_iget_args args = {
		.ino = ino, .root = NULL, .cno = cno, .for_gc = 1
	};
635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652
	struct inode *inode;
	int err;

	inode = iget5_locked(sb, ino, nilfs_iget_test, nilfs_iget_set, &args);
	if (unlikely(!inode))
		return ERR_PTR(-ENOMEM);
	if (!(inode->i_state & I_NEW))
		return inode;

	err = nilfs_init_gcinode(inode);
	if (unlikely(err)) {
		iget_failed(inode);
		return ERR_PTR(err);
	}
	unlock_new_inode(inode);
	return inode;
}

R
Ryusuke Konishi 已提交
653 654 655 656 657 658
void nilfs_write_inode_common(struct inode *inode,
			      struct nilfs_inode *raw_inode, int has_bmap)
{
	struct nilfs_inode_info *ii = NILFS_I(inode);

	raw_inode->i_mode = cpu_to_le16(inode->i_mode);
659 660
	raw_inode->i_uid = cpu_to_le32(i_uid_read(inode));
	raw_inode->i_gid = cpu_to_le32(i_gid_read(inode));
R
Ryusuke Konishi 已提交
661 662 663 664
	raw_inode->i_links_count = cpu_to_le16(inode->i_nlink);
	raw_inode->i_size = cpu_to_le64(inode->i_size);
	raw_inode->i_ctime = cpu_to_le64(inode->i_ctime.tv_sec);
	raw_inode->i_mtime = cpu_to_le64(inode->i_mtime.tv_sec);
665 666
	raw_inode->i_ctime_nsec = cpu_to_le32(inode->i_ctime.tv_nsec);
	raw_inode->i_mtime_nsec = cpu_to_le32(inode->i_mtime.tv_nsec);
R
Ryusuke Konishi 已提交
667 668 669 670 671
	raw_inode->i_blocks = cpu_to_le64(inode->i_blocks);

	raw_inode->i_flags = cpu_to_le32(ii->i_flags);
	raw_inode->i_generation = cpu_to_le32(inode->i_generation);

672 673 674 675 676 677 678 679 680 681
	if (NILFS_ROOT_METADATA_FILE(inode->i_ino)) {
		struct the_nilfs *nilfs = inode->i_sb->s_fs_info;

		/* zero-fill unused portion in the case of super root block */
		raw_inode->i_xattr = 0;
		raw_inode->i_pad = 0;
		memset((void *)raw_inode + sizeof(*raw_inode), 0,
		       nilfs->ns_inode_size - sizeof(*raw_inode));
	}

R
Ryusuke Konishi 已提交
682 683 684 685
	if (has_bmap)
		nilfs_bmap_write(ii->i_bmap, raw_inode);
	else if (S_ISCHR(inode->i_mode) || S_ISBLK(inode->i_mode))
		raw_inode->i_device_code =
686
			cpu_to_le64(huge_encode_dev(inode->i_rdev));
R
Ryusuke Konishi 已提交
687 688 689 690
	/* When extending inode, nilfs->ns_inode_size should be checked
	   for substitutions of appended fields */
}

691
void nilfs_update_inode(struct inode *inode, struct buffer_head *ibh, int flags)
R
Ryusuke Konishi 已提交
692 693 694
{
	ino_t ino = inode->i_ino;
	struct nilfs_inode_info *ii = NILFS_I(inode);
695
	struct inode *ifile = ii->i_root->ifile;
R
Ryusuke Konishi 已提交
696 697
	struct nilfs_inode *raw_inode;

698
	raw_inode = nilfs_ifile_map_inode(ifile, ino, ibh);
R
Ryusuke Konishi 已提交
699 700

	if (test_and_clear_bit(NILFS_I_NEW, &ii->i_state))
701
		memset(raw_inode, 0, NILFS_MDT(ifile)->mi_entry_size);
702 703
	if (flags & I_DIRTY_DATASYNC)
		set_bit(NILFS_I_INODE_SYNC, &ii->i_state);
R
Ryusuke Konishi 已提交
704 705 706 707 708

	nilfs_write_inode_common(inode, raw_inode, 0);
		/* XXX: call with has_bmap = 0 is a workaround to avoid
		   deadlock of bmap. This delays update of i_bmap to just
		   before writing */
709
	nilfs_ifile_unmap_inode(ifile, ino, ibh);
R
Ryusuke Konishi 已提交
710 711 712 713 714 715 716
}

#define NILFS_MAX_TRUNCATE_BLOCKS	16384  /* 64MB for 4KB block */

static void nilfs_truncate_bmap(struct nilfs_inode_info *ii,
				unsigned long from)
{
717
	__u64 b;
R
Ryusuke Konishi 已提交
718 719 720 721
	int ret;

	if (!test_bit(NILFS_I_BMAP, &ii->i_state))
		return;
722
repeat:
R
Ryusuke Konishi 已提交
723 724 725 726 727 728 729 730 731
	ret = nilfs_bmap_last_key(ii->i_bmap, &b);
	if (ret == -ENOENT)
		return;
	else if (ret < 0)
		goto failed;

	if (b < from)
		return;

732
	b -= min_t(__u64, NILFS_MAX_TRUNCATE_BLOCKS, b - from);
R
Ryusuke Konishi 已提交
733 734 735 736 737 738
	ret = nilfs_bmap_truncate(ii->i_bmap, b);
	nilfs_relax_pressure_in_lock(ii->vfs_inode.i_sb);
	if (!ret || (ret == -ENOMEM &&
		     nilfs_bmap_truncate(ii->i_bmap, b) == 0))
		goto repeat;

739 740 741 742
failed:
	nilfs_warning(ii->vfs_inode.i_sb, __func__,
		      "failed to truncate bmap (ino=%lu, err=%d)",
		      ii->vfs_inode.i_ino, ret);
R
Ryusuke Konishi 已提交
743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759
}

void nilfs_truncate(struct inode *inode)
{
	unsigned long blkoff;
	unsigned int blocksize;
	struct nilfs_transaction_info ti;
	struct super_block *sb = inode->i_sb;
	struct nilfs_inode_info *ii = NILFS_I(inode);

	if (!test_bit(NILFS_I_BMAP, &ii->i_state))
		return;
	if (IS_APPEND(inode) || IS_IMMUTABLE(inode))
		return;

	blocksize = sb->s_blocksize;
	blkoff = (inode->i_size + blocksize - 1) >> sb->s_blocksize_bits;
760
	nilfs_transaction_begin(sb, &ti, 0); /* never fails */
R
Ryusuke Konishi 已提交
761 762 763 764 765 766 767 768 769

	block_truncate_page(inode->i_mapping, inode->i_size, nilfs_get_block);

	nilfs_truncate_bmap(ii, blkoff);

	inode->i_mtime = inode->i_ctime = CURRENT_TIME;
	if (IS_SYNC(inode))
		nilfs_set_transaction_flag(NILFS_TI_SYNC);

770
	nilfs_mark_inode_dirty(inode);
771
	nilfs_set_file_dirty(inode, 0);
772
	nilfs_transaction_commit(sb);
R
Ryusuke Konishi 已提交
773 774 775 776
	/* May construct a logical segment and may fail in sync mode.
	   But truncate has no return value. */
}

A
Al Viro 已提交
777 778 779
static void nilfs_clear_inode(struct inode *inode)
{
	struct nilfs_inode_info *ii = NILFS_I(inode);
780
	struct nilfs_mdt_info *mdi = NILFS_MDT(inode);
A
Al Viro 已提交
781 782 783 784 785 786 787 788

	/*
	 * Free resources allocated in nilfs_read_inode(), here.
	 */
	BUG_ON(!list_empty(&ii->i_dirty));
	brelse(ii->i_bh);
	ii->i_bh = NULL;

789 790 791
	if (mdi && mdi->mi_palloc_cache)
		nilfs_palloc_destroy_cache(inode);

A
Al Viro 已提交
792 793 794 795
	if (test_bit(NILFS_I_BMAP, &ii->i_state))
		nilfs_bmap_clear(ii->i_bmap);

	nilfs_btnode_cache_clear(&ii->i_btnode_cache);
796 797 798

	if (ii->i_root && inode->i_ino == NILFS_ROOT_INO)
		nilfs_put_root(ii->i_root);
A
Al Viro 已提交
799 800 801
}

void nilfs_evict_inode(struct inode *inode)
R
Ryusuke Konishi 已提交
802 803 804 805
{
	struct nilfs_transaction_info ti;
	struct super_block *sb = inode->i_sb;
	struct nilfs_inode_info *ii = NILFS_I(inode);
806
	int ret;
R
Ryusuke Konishi 已提交
807

808
	if (inode->i_nlink || !ii->i_root || unlikely(is_bad_inode(inode))) {
809
		truncate_inode_pages_final(&inode->i_data);
810
		clear_inode(inode);
A
Al Viro 已提交
811
		nilfs_clear_inode(inode);
R
Ryusuke Konishi 已提交
812 813
		return;
	}
814 815
	nilfs_transaction_begin(sb, &ti, 0); /* never fails */

816
	truncate_inode_pages_final(&inode->i_data);
R
Ryusuke Konishi 已提交
817

818
	/* TODO: some of the following operations may fail.  */
R
Ryusuke Konishi 已提交
819
	nilfs_truncate_bmap(ii, 0);
820
	nilfs_mark_inode_dirty(inode);
821
	clear_inode(inode);
822

823 824
	ret = nilfs_ifile_delete_inode(ii->i_root->ifile, inode->i_ino);
	if (!ret)
825
		atomic64_dec(&ii->i_root->inodes_count);
826

A
Al Viro 已提交
827
	nilfs_clear_inode(inode);
828

R
Ryusuke Konishi 已提交
829 830
	if (IS_SYNC(inode))
		nilfs_set_transaction_flag(NILFS_TI_SYNC);
831
	nilfs_transaction_commit(sb);
R
Ryusuke Konishi 已提交
832 833 834 835 836 837 838 839 840
	/* May construct a logical segment and may fail in sync mode.
	   But delete_inode has no return value. */
}

int nilfs_setattr(struct dentry *dentry, struct iattr *iattr)
{
	struct nilfs_transaction_info ti;
	struct inode *inode = dentry->d_inode;
	struct super_block *sb = inode->i_sb;
841
	int err;
R
Ryusuke Konishi 已提交
842 843 844 845 846 847 848 849

	err = inode_change_ok(inode, iattr);
	if (err)
		return err;

	err = nilfs_transaction_begin(sb, &ti, 0);
	if (unlikely(err))
		return err;
C
Christoph Hellwig 已提交
850 851 852

	if ((iattr->ia_valid & ATTR_SIZE) &&
	    iattr->ia_size != i_size_read(inode)) {
853
		inode_dio_wait(inode);
M
Marco Stornelli 已提交
854 855
		truncate_setsize(inode, iattr->ia_size);
		nilfs_truncate(inode);
C
Christoph Hellwig 已提交
856 857 858 859 860 861
	}

	setattr_copy(inode, iattr);
	mark_inode_dirty(inode);

	if (iattr->ia_valid & ATTR_MODE) {
R
Ryusuke Konishi 已提交
862
		err = nilfs_acl_chmod(inode);
C
Christoph Hellwig 已提交
863 864 865 866 867
		if (unlikely(err))
			goto out_err;
	}

	return nilfs_transaction_commit(sb);
868

C
Christoph Hellwig 已提交
869 870
out_err:
	nilfs_transaction_abort(sb);
871
	return err;
R
Ryusuke Konishi 已提交
872 873
}

874
int nilfs_permission(struct inode *inode, int mask)
875
{
876
	struct nilfs_root *root = NILFS_I(inode)->i_root;
877 878 879 880
	if ((mask & MAY_WRITE) && root &&
	    root->cno != NILFS_CPTREE_CURRENT_CNO)
		return -EROFS; /* snapshot is not writable */

881
	return generic_permission(inode, mask);
882 883
}

884
int nilfs_load_inode_block(struct inode *inode, struct buffer_head **pbh)
R
Ryusuke Konishi 已提交
885
{
886
	struct the_nilfs *nilfs = inode->i_sb->s_fs_info;
R
Ryusuke Konishi 已提交
887 888 889
	struct nilfs_inode_info *ii = NILFS_I(inode);
	int err;

890
	spin_lock(&nilfs->ns_inode_lock);
R
Ryusuke Konishi 已提交
891
	if (ii->i_bh == NULL) {
892
		spin_unlock(&nilfs->ns_inode_lock);
893 894
		err = nilfs_ifile_get_inode_block(ii->i_root->ifile,
						  inode->i_ino, pbh);
R
Ryusuke Konishi 已提交
895 896
		if (unlikely(err))
			return err;
897
		spin_lock(&nilfs->ns_inode_lock);
R
Ryusuke Konishi 已提交
898 899 900 901 902 903 904 905 906 907
		if (ii->i_bh == NULL)
			ii->i_bh = *pbh;
		else {
			brelse(*pbh);
			*pbh = ii->i_bh;
		}
	} else
		*pbh = ii->i_bh;

	get_bh(*pbh);
908
	spin_unlock(&nilfs->ns_inode_lock);
R
Ryusuke Konishi 已提交
909 910 911 912 913 914
	return 0;
}

int nilfs_inode_dirty(struct inode *inode)
{
	struct nilfs_inode_info *ii = NILFS_I(inode);
915
	struct the_nilfs *nilfs = inode->i_sb->s_fs_info;
R
Ryusuke Konishi 已提交
916 917 918
	int ret = 0;

	if (!list_empty(&ii->i_dirty)) {
919
		spin_lock(&nilfs->ns_inode_lock);
R
Ryusuke Konishi 已提交
920 921
		ret = test_bit(NILFS_I_DIRTY, &ii->i_state) ||
			test_bit(NILFS_I_BUSY, &ii->i_state);
922
		spin_unlock(&nilfs->ns_inode_lock);
R
Ryusuke Konishi 已提交
923 924 925 926
	}
	return ret;
}

927
int nilfs_set_file_dirty(struct inode *inode, unsigned nr_dirty)
R
Ryusuke Konishi 已提交
928 929
{
	struct nilfs_inode_info *ii = NILFS_I(inode);
930
	struct the_nilfs *nilfs = inode->i_sb->s_fs_info;
R
Ryusuke Konishi 已提交
931

932
	atomic_add(nr_dirty, &nilfs->ns_ndirtyblks);
R
Ryusuke Konishi 已提交
933

R
Ryusuke Konishi 已提交
934
	if (test_and_set_bit(NILFS_I_DIRTY, &ii->i_state))
R
Ryusuke Konishi 已提交
935 936
		return 0;

937
	spin_lock(&nilfs->ns_inode_lock);
R
Ryusuke Konishi 已提交
938 939 940 941 942 943 944
	if (!test_bit(NILFS_I_QUEUED, &ii->i_state) &&
	    !test_bit(NILFS_I_BUSY, &ii->i_state)) {
		/* Because this routine may race with nilfs_dispose_list(),
		   we have to check NILFS_I_QUEUED here, too. */
		if (list_empty(&ii->i_dirty) && igrab(inode) == NULL) {
			/* This will happen when somebody is freeing
			   this inode. */
945
			nilfs_warning(inode->i_sb, __func__,
R
Ryusuke Konishi 已提交
946 947
				      "cannot get inode (ino=%lu)\n",
				      inode->i_ino);
948
			spin_unlock(&nilfs->ns_inode_lock);
R
Ryusuke Konishi 已提交
949 950 951
			return -EINVAL; /* NILFS_I_DIRTY may remain for
					   freeing inode */
		}
952
		list_move_tail(&ii->i_dirty, &nilfs->ns_dirty_files);
R
Ryusuke Konishi 已提交
953 954
		set_bit(NILFS_I_QUEUED, &ii->i_state);
	}
955
	spin_unlock(&nilfs->ns_inode_lock);
R
Ryusuke Konishi 已提交
956 957 958
	return 0;
}

959
int __nilfs_mark_inode_dirty(struct inode *inode, int flags)
R
Ryusuke Konishi 已提交
960 961 962 963
{
	struct buffer_head *ibh;
	int err;

964
	err = nilfs_load_inode_block(inode, &ibh);
R
Ryusuke Konishi 已提交
965 966 967 968 969
	if (unlikely(err)) {
		nilfs_warning(inode->i_sb, __func__,
			      "failed to reget inode block.\n");
		return err;
	}
970
	nilfs_update_inode(inode, ibh, flags);
971
	mark_buffer_dirty(ibh);
972
	nilfs_mdt_mark_dirty(NILFS_I(inode)->i_root->ifile);
R
Ryusuke Konishi 已提交
973 974 975 976 977 978 979 980 981 982 983 984 985 986
	brelse(ibh);
	return 0;
}

/**
 * nilfs_dirty_inode - reflect changes on given inode to an inode block.
 * @inode: inode of the file to be registered.
 *
 * nilfs_dirty_inode() loads a inode block containing the specified
 * @inode and copies data from a nilfs_inode to a corresponding inode
 * entry in the inode block. This operation is excluded from the segment
 * construction. This function can be called both as a single operation
 * and as a part of indivisible file operations.
 */
987
void nilfs_dirty_inode(struct inode *inode, int flags)
R
Ryusuke Konishi 已提交
988 989
{
	struct nilfs_transaction_info ti;
990
	struct nilfs_mdt_info *mdi = NILFS_MDT(inode);
R
Ryusuke Konishi 已提交
991 992 993 994 995 996 997

	if (is_bad_inode(inode)) {
		nilfs_warning(inode->i_sb, __func__,
			      "tried to mark bad_inode dirty. ignored.\n");
		dump_stack();
		return;
	}
998 999 1000 1001
	if (mdi) {
		nilfs_mdt_mark_dirty(inode);
		return;
	}
R
Ryusuke Konishi 已提交
1002
	nilfs_transaction_begin(inode->i_sb, &ti, 0);
1003
	__nilfs_mark_inode_dirty(inode, flags);
1004
	nilfs_transaction_commit(inode->i_sb); /* never fails */
R
Ryusuke Konishi 已提交
1005
}
R
Ryusuke Konishi 已提交
1006 1007 1008 1009

int nilfs_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo,
		 __u64 start, __u64 len)
{
1010
	struct the_nilfs *nilfs = inode->i_sb->s_fs_info;
R
Ryusuke Konishi 已提交
1011 1012 1013 1014 1015 1016 1017 1018 1019 1020 1021 1022 1023 1024 1025 1026 1027 1028 1029 1030 1031 1032 1033 1034 1035 1036 1037 1038 1039 1040 1041 1042 1043 1044 1045 1046 1047 1048 1049 1050 1051 1052 1053 1054 1055 1056 1057 1058 1059 1060 1061 1062 1063 1064 1065 1066 1067 1068 1069 1070 1071 1072 1073 1074 1075 1076 1077 1078 1079 1080 1081 1082 1083 1084 1085 1086 1087 1088 1089 1090 1091 1092 1093 1094 1095 1096 1097 1098 1099 1100 1101 1102 1103 1104 1105 1106 1107 1108 1109 1110 1111 1112 1113 1114 1115 1116 1117 1118 1119 1120 1121 1122 1123 1124 1125 1126 1127 1128 1129 1130 1131 1132 1133 1134 1135 1136
	__u64 logical = 0, phys = 0, size = 0;
	__u32 flags = 0;
	loff_t isize;
	sector_t blkoff, end_blkoff;
	sector_t delalloc_blkoff;
	unsigned long delalloc_blklen;
	unsigned int blkbits = inode->i_blkbits;
	int ret, n;

	ret = fiemap_check_flags(fieinfo, FIEMAP_FLAG_SYNC);
	if (ret)
		return ret;

	mutex_lock(&inode->i_mutex);

	isize = i_size_read(inode);

	blkoff = start >> blkbits;
	end_blkoff = (start + len - 1) >> blkbits;

	delalloc_blklen = nilfs_find_uncommitted_extent(inode, blkoff,
							&delalloc_blkoff);

	do {
		__u64 blkphy;
		unsigned int maxblocks;

		if (delalloc_blklen && blkoff == delalloc_blkoff) {
			if (size) {
				/* End of the current extent */
				ret = fiemap_fill_next_extent(
					fieinfo, logical, phys, size, flags);
				if (ret)
					break;
			}
			if (blkoff > end_blkoff)
				break;

			flags = FIEMAP_EXTENT_MERGED | FIEMAP_EXTENT_DELALLOC;
			logical = blkoff << blkbits;
			phys = 0;
			size = delalloc_blklen << blkbits;

			blkoff = delalloc_blkoff + delalloc_blklen;
			delalloc_blklen = nilfs_find_uncommitted_extent(
				inode, blkoff, &delalloc_blkoff);
			continue;
		}

		/*
		 * Limit the number of blocks that we look up so as
		 * not to get into the next delayed allocation extent.
		 */
		maxblocks = INT_MAX;
		if (delalloc_blklen)
			maxblocks = min_t(sector_t, delalloc_blkoff - blkoff,
					  maxblocks);
		blkphy = 0;

		down_read(&NILFS_MDT(nilfs->ns_dat)->mi_sem);
		n = nilfs_bmap_lookup_contig(
			NILFS_I(inode)->i_bmap, blkoff, &blkphy, maxblocks);
		up_read(&NILFS_MDT(nilfs->ns_dat)->mi_sem);

		if (n < 0) {
			int past_eof;

			if (unlikely(n != -ENOENT))
				break; /* error */

			/* HOLE */
			blkoff++;
			past_eof = ((blkoff << blkbits) >= isize);

			if (size) {
				/* End of the current extent */

				if (past_eof)
					flags |= FIEMAP_EXTENT_LAST;

				ret = fiemap_fill_next_extent(
					fieinfo, logical, phys, size, flags);
				if (ret)
					break;
				size = 0;
			}
			if (blkoff > end_blkoff || past_eof)
				break;
		} else {
			if (size) {
				if (phys && blkphy << blkbits == phys + size) {
					/* The current extent goes on */
					size += n << blkbits;
				} else {
					/* Terminate the current extent */
					ret = fiemap_fill_next_extent(
						fieinfo, logical, phys, size,
						flags);
					if (ret || blkoff > end_blkoff)
						break;

					/* Start another extent */
					flags = FIEMAP_EXTENT_MERGED;
					logical = blkoff << blkbits;
					phys = blkphy << blkbits;
					size = n << blkbits;
				}
			} else {
				/* Start a new extent */
				flags = FIEMAP_EXTENT_MERGED;
				logical = blkoff << blkbits;
				phys = blkphy << blkbits;
				size = n << blkbits;
			}
			blkoff += n;
		}
		cond_resched();
	} while (true);

	/* If ret is 1 then we just hit the end of the extent array */
	if (ret == 1)
		ret = 0;

	mutex_unlock(&inode->i_mutex);
	return ret;
}