file.c 40.0 KB
Newer Older
1
// SPDX-License-Identifier: GPL-2.0-only
D
David Teigland 已提交
2 3
/*
 * Copyright (C) Sistina Software, Inc.  1997-2003 All rights reserved.
4
 * Copyright (C) 2004-2006 Red Hat, Inc.  All rights reserved.
D
David Teigland 已提交
5 6 7 8
 */

#include <linux/slab.h>
#include <linux/spinlock.h>
A
Arnd Bergmann 已提交
9
#include <linux/compat.h>
D
David Teigland 已提交
10 11 12 13 14 15
#include <linux/completion.h>
#include <linux/buffer_head.h>
#include <linux/pagemap.h>
#include <linux/uio.h>
#include <linux/blkdev.h>
#include <linux/mm.h>
M
Miklos Szeredi 已提交
16
#include <linux/mount.h>
17
#include <linux/fs.h>
18
#include <linux/gfs2_ondisk.h>
19 20
#include <linux/falloc.h>
#include <linux/swap.h>
21
#include <linux/crc32.h>
22
#include <linux/writeback.h>
23
#include <linux/uaccess.h>
24 25
#include <linux/dlm.h>
#include <linux/dlm_plock.h>
26
#include <linux/delay.h>
27
#include <linux/backing-dev.h>
M
Miklos Szeredi 已提交
28
#include <linux/fileattr.h>
D
David Teigland 已提交
29 30

#include "gfs2.h"
31
#include "incore.h"
D
David Teigland 已提交
32
#include "bmap.h"
33
#include "aops.h"
D
David Teigland 已提交
34 35 36 37 38 39 40 41 42
#include "dir.h"
#include "glock.h"
#include "glops.h"
#include "inode.h"
#include "log.h"
#include "meta_io.h"
#include "quota.h"
#include "rgrp.h"
#include "trans.h"
43
#include "util.h"
D
David Teigland 已提交
44 45 46 47 48

/**
 * gfs2_llseek - seek to a location in a file
 * @file: the file
 * @offset: the offset
49
 * @whence: Where to seek from (SEEK_SET, SEEK_CUR, or SEEK_END)
D
David Teigland 已提交
50 51 52 53 54 55 56
 *
 * SEEK_END requires the glock for the file because it references the
 * file's size.
 *
 * Returns: The new offset, or errno
 */

57
static loff_t gfs2_llseek(struct file *file, loff_t offset, int whence)
D
David Teigland 已提交
58
{
59
	struct gfs2_inode *ip = GFS2_I(file->f_mapping->host);
D
David Teigland 已提交
60 61 62
	struct gfs2_holder i_gh;
	loff_t error;

63
	switch (whence) {
64
	case SEEK_END:
D
David Teigland 已提交
65 66 67
		error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY,
					   &i_gh);
		if (!error) {
68
			error = generic_file_llseek(file, offset, whence);
D
David Teigland 已提交
69 70
			gfs2_glock_dq_uninit(&i_gh);
		}
71
		break;
72 73 74 75 76 77 78 79 80

	case SEEK_DATA:
		error = gfs2_seek_data(file, offset);
		break;

	case SEEK_HOLE:
		error = gfs2_seek_hole(file, offset);
		break;

81 82
	case SEEK_CUR:
	case SEEK_SET:
83 84 85 86
		/*
		 * These don't reference inode->i_size and don't depend on the
		 * block mapping, so we don't need the glock.
		 */
87
		error = generic_file_llseek(file, offset, whence);
88 89 90 91
		break;
	default:
		error = -EINVAL;
	}
D
David Teigland 已提交
92 93 94 95 96

	return error;
}

/**
A
Al Viro 已提交
97
 * gfs2_readdir - Iterator for a directory
D
David Teigland 已提交
98
 * @file: The directory to read from
A
Al Viro 已提交
99
 * @ctx: What to feed directory entries to
D
David Teigland 已提交
100 101 102 103
 *
 * Returns: errno
 */

A
Al Viro 已提交
104
static int gfs2_readdir(struct file *file, struct dir_context *ctx)
D
David Teigland 已提交
105
{
106
	struct inode *dir = file->f_mapping->host;
107
	struct gfs2_inode *dip = GFS2_I(dir);
D
David Teigland 已提交
108 109 110
	struct gfs2_holder d_gh;
	int error;

A
Al Viro 已提交
111 112
	error = gfs2_glock_nq_init(dip->i_gl, LM_ST_SHARED, 0, &d_gh);
	if (error)
D
David Teigland 已提交
113 114
		return error;

A
Al Viro 已提交
115
	error = gfs2_dir_read(dir, ctx, &file->f_ra);
D
David Teigland 已提交
116 117 118 119 120 121

	gfs2_glock_dq_uninit(&d_gh);

	return error;
}

122 123
/*
 * struct fsflag_gfs2flag
124
 *
125 126
 * The FS_JOURNAL_DATA_FL flag maps to GFS2_DIF_INHERIT_JDATA for directories,
 * and to GFS2_DIF_JDATA for non-directories.
127
 */
128 129 130 131 132 133 134 135 136 137 138
static struct {
	u32 fsflag;
	u32 gfsflag;
} fsflag_gfs2flag[] = {
	{FS_SYNC_FL, GFS2_DIF_SYNC},
	{FS_IMMUTABLE_FL, GFS2_DIF_IMMUTABLE},
	{FS_APPEND_FL, GFS2_DIF_APPENDONLY},
	{FS_NOATIME_FL, GFS2_DIF_NOATIME},
	{FS_INDEX_FL, GFS2_DIF_EXHASH},
	{FS_TOPDIR_FL, GFS2_DIF_TOPDIR},
	{FS_JOURNAL_DATA_FL, GFS2_DIF_JDATA | GFS2_DIF_INHERIT_JDATA},
139
};
140

141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156
static inline u32 gfs2_gfsflags_to_fsflags(struct inode *inode, u32 gfsflags)
{
	int i;
	u32 fsflags = 0;

	if (S_ISDIR(inode->i_mode))
		gfsflags &= ~GFS2_DIF_JDATA;
	else
		gfsflags &= ~GFS2_DIF_INHERIT_JDATA;

	for (i = 0; i < ARRAY_SIZE(fsflag_gfs2flag); i++)
		if (gfsflags & fsflag_gfs2flag[i].gfsflag)
			fsflags |= fsflag_gfs2flag[i].fsflag;
	return fsflags;
}

M
Miklos Szeredi 已提交
157
int gfs2_fileattr_get(struct dentry *dentry, struct fileattr *fa)
158
{
M
Miklos Szeredi 已提交
159
	struct inode *inode = d_inode(dentry);
160
	struct gfs2_inode *ip = GFS2_I(inode);
161
	struct gfs2_holder gh;
162 163
	int error;
	u32 fsflags;
164

M
Miklos Szeredi 已提交
165 166 167
	if (d_is_special(dentry))
		return -ENOTTY;

168 169
	gfs2_holder_init(ip->i_gl, LM_ST_SHARED, 0, &gh);
	error = gfs2_glock_nq(&gh);
170
	if (error)
171
		goto out_uninit;
172

173
	fsflags = gfs2_gfsflags_to_fsflags(inode, ip->i_diskflags);
174

M
Miklos Szeredi 已提交
175
	fileattr_fill_flags(fa, fsflags);
176

177
	gfs2_glock_dq(&gh);
178
out_uninit:
179 180 181 182
	gfs2_holder_uninit(&gh);
	return error;
}

183 184 185 186 187
void gfs2_set_inode_flags(struct inode *inode)
{
	struct gfs2_inode *ip = GFS2_I(inode);
	unsigned int flags = inode->i_flags;

S
Steven Whitehouse 已提交
188 189
	flags &= ~(S_SYNC|S_APPEND|S_IMMUTABLE|S_NOATIME|S_DIRSYNC|S_NOSEC);
	if ((ip->i_eattr == 0) && !is_sxid(inode->i_mode))
190
		flags |= S_NOSEC;
191
	if (ip->i_diskflags & GFS2_DIF_IMMUTABLE)
192
		flags |= S_IMMUTABLE;
193
	if (ip->i_diskflags & GFS2_DIF_APPENDONLY)
194
		flags |= S_APPEND;
195
	if (ip->i_diskflags & GFS2_DIF_NOATIME)
196
		flags |= S_NOATIME;
197
	if (ip->i_diskflags & GFS2_DIF_SYNC)
198 199 200 201
		flags |= S_SYNC;
	inode->i_flags = flags;
}

202 203 204 205 206 207
/* Flags that can be set by user space */
#define GFS2_FLAGS_USER_SET (GFS2_DIF_JDATA|			\
			     GFS2_DIF_IMMUTABLE|		\
			     GFS2_DIF_APPENDONLY|		\
			     GFS2_DIF_NOATIME|			\
			     GFS2_DIF_SYNC|			\
208
			     GFS2_DIF_TOPDIR|			\
209 210 211
			     GFS2_DIF_INHERIT_JDATA)

/**
212
 * do_gfs2_set_flags - set flags on an inode
213
 * @inode: The inode
214
 * @reqflags: The flags to set
215 216 217
 * @mask: Indicates which flags are valid
 *
 */
218
static int do_gfs2_set_flags(struct inode *inode, u32 reqflags, u32 mask)
219
{
220 221
	struct gfs2_inode *ip = GFS2_I(inode);
	struct gfs2_sbd *sdp = GFS2_SB(inode);
222 223 224
	struct buffer_head *bh;
	struct gfs2_holder gh;
	int error;
M
Miklos Szeredi 已提交
225
	u32 new_flags, flags;
226

M
Miklos Szeredi 已提交
227 228
	error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh);
	if (error)
M
Miklos Szeredi 已提交
229
		return error;
230 231

	error = 0;
232
	flags = ip->i_diskflags;
233
	new_flags = (flags & ~mask) | (reqflags & mask);
234 235 236
	if ((new_flags ^ flags) == 0)
		goto out;

237
	if (!IS_IMMUTABLE(inode)) {
238
		error = gfs2_permission(&init_user_ns, inode, MAY_WRITE);
239 240 241
		if (error)
			goto out;
	}
242
	if ((flags ^ new_flags) & GFS2_DIF_JDATA) {
243
		if (new_flags & GFS2_DIF_JDATA)
244
			gfs2_log_flush(sdp, ip->i_gl,
245 246
				       GFS2_LOG_HEAD_FLUSH_NORMAL |
				       GFS2_LFC_SET_FLAGS);
247 248 249 250 251 252
		error = filemap_fdatawrite(inode->i_mapping);
		if (error)
			goto out;
		error = filemap_fdatawait(inode->i_mapping);
		if (error)
			goto out;
253 254
		if (new_flags & GFS2_DIF_JDATA)
			gfs2_ordered_del_inode(ip);
255
	}
256
	error = gfs2_trans_begin(sdp, RES_DINODE, 0);
257 258
	if (error)
		goto out;
259 260 261
	error = gfs2_meta_inode_buffer(ip, &bh);
	if (error)
		goto out_trans_end;
262
	inode->i_ctime = current_time(inode);
263
	gfs2_trans_add_meta(ip->i_gl, bh);
264
	ip->i_diskflags = new_flags;
265
	gfs2_dinode_out(ip, bh->b_data);
266
	brelse(bh);
267
	gfs2_set_inode_flags(inode);
268
	gfs2_set_aops(inode);
269 270
out_trans_end:
	gfs2_trans_end(sdp);
271 272 273 274 275
out:
	gfs2_glock_dq_uninit(&gh);
	return error;
}

M
Miklos Szeredi 已提交
276 277
int gfs2_fileattr_set(struct user_namespace *mnt_userns,
		      struct dentry *dentry, struct fileattr *fa)
278
{
M
Miklos Szeredi 已提交
279 280
	struct inode *inode = d_inode(dentry);
	u32 fsflags = fa->flags, gfsflags = 0;
281 282
	u32 mask;
	int i;
283

M
Miklos Szeredi 已提交
284 285 286 287 288
	if (d_is_special(dentry))
		return -ENOTTY;

	if (fileattr_has_fsx(fa))
		return -EOPNOTSUPP;
289

290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306
	for (i = 0; i < ARRAY_SIZE(fsflag_gfs2flag); i++) {
		if (fsflags & fsflag_gfs2flag[i].fsflag) {
			fsflags &= ~fsflag_gfs2flag[i].fsflag;
			gfsflags |= fsflag_gfs2flag[i].gfsflag;
		}
	}
	if (fsflags || gfsflags & ~GFS2_FLAGS_USER_SET)
		return -EINVAL;

	mask = GFS2_FLAGS_USER_SET;
	if (S_ISDIR(inode->i_mode)) {
		mask &= ~GFS2_DIF_JDATA;
	} else {
		/* The GFS2_DIF_TOPDIR flag is only valid for directories. */
		if (gfsflags & GFS2_DIF_TOPDIR)
			return -EINVAL;
		mask &= ~(GFS2_DIF_TOPDIR | GFS2_DIF_INHERIT_JDATA);
307
	}
308

309
	return do_gfs2_set_flags(inode, gfsflags, mask);
310 311
}

S
Steve Whitehouse 已提交
312 313 314 315 316 317 318 319 320 321 322
static int gfs2_getlabel(struct file *filp, char __user *label)
{
	struct inode *inode = file_inode(filp);
	struct gfs2_sbd *sdp = GFS2_SB(inode);

	if (copy_to_user(label, sdp->sd_sb.sb_locktable, GFS2_LOCKNAME_LEN))
		return -EFAULT;

	return 0;
}

323
static long gfs2_ioctl(struct file *filp, unsigned int cmd, unsigned long arg)
324 325
{
	switch(cmd) {
S
Steven Whitehouse 已提交
326 327
	case FITRIM:
		return gfs2_fitrim(filp, (void __user *)arg);
S
Steve Whitehouse 已提交
328 329
	case FS_IOC_GETFSLABEL:
		return gfs2_getlabel(filp, (char __user *)arg);
330
	}
S
Steve Whitehouse 已提交
331

332 333 334
	return -ENOTTY;
}

A
Arnd Bergmann 已提交
335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352
#ifdef CONFIG_COMPAT
static long gfs2_compat_ioctl(struct file *filp, unsigned int cmd, unsigned long arg)
{
	switch(cmd) {
	/* Keep this list in sync with gfs2_ioctl */
	case FITRIM:
	case FS_IOC_GETFSLABEL:
		break;
	default:
		return -ENOIOCTLCMD;
	}

	return gfs2_ioctl(filp, cmd, (unsigned long)compat_ptr(arg));
}
#else
#define gfs2_compat_ioctl NULL
#endif

353 354
/**
 * gfs2_size_hint - Give a hint to the size of a write request
355
 * @filep: The struct file
356 357 358 359 360 361 362 363 364 365 366
 * @offset: The file offset of the write
 * @size: The length of the write
 *
 * When we are about to do a write, this function records the total
 * write size in order to provide a suitable hint to the lower layers
 * about how many blocks will be required.
 *
 */

static void gfs2_size_hint(struct file *filep, loff_t offset, size_t size)
{
A
Al Viro 已提交
367
	struct inode *inode = file_inode(filep);
368 369 370 371 372
	struct gfs2_sbd *sdp = GFS2_SB(inode);
	struct gfs2_inode *ip = GFS2_I(inode);
	size_t blks = (size + sdp->sd_sb.sb_bsize - 1) >> sdp->sd_sb.sb_bsize_shift;
	int hint = min_t(size_t, INT_MAX, blks);

373 374
	if (hint > atomic_read(&ip->i_sizehint))
		atomic_set(&ip->i_sizehint, hint);
375 376
}

377
/**
378
 * gfs2_allocate_page_backing - Allocate blocks for a write fault
379
 * @page: The (locked) page to allocate backing for
380
 * @length: Size of the allocation
381
 *
382 383 384 385
 * We try to allocate all the blocks required for the page in one go.  This
 * might fail for various reasons, so we keep trying until all the blocks to
 * back this page are allocated.  If some of the blocks are already allocated,
 * that is ok too.
386
 */
387
static int gfs2_allocate_page_backing(struct page *page, unsigned int length)
388
{
389
	u64 pos = page_offset(page);
390 391

	do {
392 393
		struct iomap iomap = { };

394
		if (gfs2_iomap_alloc(page->mapping->host, pos, length, &iomap))
395
			return -EIO;
396

397 398 399
		if (length < iomap.length)
			iomap.length = length;
		length -= iomap.length;
400
		pos += iomap.length;
401
	} while (length > 0);
402

403 404 405 406 407
	return 0;
}

/**
 * gfs2_page_mkwrite - Make a shared, mmap()ed, page writable
408
 * @vmf: The virtual memory fault containing the page to become writable
409 410 411 412 413
 *
 * When the page becomes writable, we need to ensure that we have
 * blocks allocated on disk to back that page.
 */

414
static vm_fault_t gfs2_page_mkwrite(struct vm_fault *vmf)
415
{
416
	struct page *page = vmf->page;
417
	struct inode *inode = file_inode(vmf->vma->vm_file);
418 419
	struct gfs2_inode *ip = GFS2_I(inode);
	struct gfs2_sbd *sdp = GFS2_SB(inode);
420
	struct gfs2_alloc_parms ap = { .aflags = 0, };
421
	u64 offset = page_offset(page);
422
	unsigned int data_blocks, ind_blocks, rblocks;
423
	vm_fault_t ret = VM_FAULT_LOCKED;
424
	struct gfs2_holder gh;
425
	unsigned int length;
S
Steven Whitehouse 已提交
426
	loff_t size;
427
	int err;
428

429
	sb_start_pagefault(inode->i_sb);
S
Steven Whitehouse 已提交
430

431
	gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh);
432 433 434
	err = gfs2_glock_nq(&gh);
	if (err) {
		ret = block_page_mkwrite_return(err);
435
		goto out_uninit;
436
	}
437

438 439 440
	/* Check page index against inode size */
	size = i_size_read(inode);
	if (offset >= size) {
441
		ret = VM_FAULT_SIGBUS;
442 443 444
		goto out_unlock;
	}

445
	/* Update file times before taking page lock */
446
	file_update_time(vmf->vma->vm_file);
447

448
	/* page is wholly or partially inside EOF */
449 450
	if (size - offset < PAGE_SIZE)
		length = size - offset;
451 452 453 454 455
	else
		length = PAGE_SIZE;

	gfs2_size_hint(vmf->vma->vm_file, offset, length);

456 457 458
	set_bit(GLF_DIRTY, &ip->i_gl->gl_flags);
	set_bit(GIF_SW_PAGED, &ip->i_flags);

459 460 461 462 463 464 465
	/*
	 * iomap_writepage / iomap_writepages currently don't support inline
	 * files, so always unstuff here.
	 */

	if (!gfs2_is_stuffed(ip) &&
	    !gfs2_write_alloc_required(ip, offset, length)) {
S
Steven Whitehouse 已提交
466 467
		lock_page(page);
		if (!PageUptodate(page) || page->mapping != inode->i_mapping) {
468
			ret = VM_FAULT_NOPAGE;
S
Steven Whitehouse 已提交
469 470
			unlock_page(page);
		}
471
		goto out_unlock;
S
Steven Whitehouse 已提交
472 473
	}

474 475 476
	err = gfs2_rindex_update(sdp);
	if (err) {
		ret = block_page_mkwrite_return(err);
477
		goto out_unlock;
478
	}
479

480
	gfs2_write_calc_reserv(ip, length, &data_blocks, &ind_blocks);
481
	ap.target = data_blocks + ind_blocks;
482 483 484
	err = gfs2_quota_lock_check(ip, &ap);
	if (err) {
		ret = block_page_mkwrite_return(err);
485
		goto out_unlock;
486 487 488 489
	}
	err = gfs2_inplace_reserve(ip, &ap);
	if (err) {
		ret = block_page_mkwrite_return(err);
490
		goto out_quota_unlock;
491
	}
492 493 494 495

	rblocks = RES_DINODE + ind_blocks;
	if (gfs2_is_jdata(ip))
		rblocks += data_blocks ? data_blocks : 1;
496
	if (ind_blocks || data_blocks) {
497
		rblocks += RES_STATFS + RES_QUOTA;
498
		rblocks += gfs2_rg_blocks(ip, data_blocks + ind_blocks);
499
	}
500 501 502
	err = gfs2_trans_begin(sdp, rblocks, 0);
	if (err) {
		ret = block_page_mkwrite_return(err);
503
		goto out_trans_fail;
504
	}
505

506 507
	/* Unstuff, if required, and allocate backing blocks for page */
	if (gfs2_is_stuffed(ip)) {
508
		err = gfs2_unstuff_dinode(ip);
509 510 511 512 513 514
		if (err) {
			ret = block_page_mkwrite_return(err);
			goto out_trans_end;
		}
	}

515
	lock_page(page);
S
Steven Whitehouse 已提交
516 517 518
	/* If truncated, we must retry the operation, we may have raced
	 * with the glock demotion code.
	 */
519 520
	if (!PageUptodate(page) || page->mapping != inode->i_mapping) {
		ret = VM_FAULT_NOPAGE;
521
		goto out_page_locked;
522
	}
S
Steven Whitehouse 已提交
523

524 525 526
	err = gfs2_allocate_page_backing(page, length);
	if (err)
		ret = block_page_mkwrite_return(err);
527

528
out_page_locked:
529
	if (ret != VM_FAULT_LOCKED)
S
Steven Whitehouse 已提交
530
		unlock_page(page);
531
out_trans_end:
532 533 534 535 536 537 538
	gfs2_trans_end(sdp);
out_trans_fail:
	gfs2_inplace_release(ip);
out_quota_unlock:
	gfs2_quota_unlock(ip);
out_unlock:
	gfs2_glock_dq(&gh);
539
out_uninit:
540
	gfs2_holder_uninit(&gh);
541
	if (ret == VM_FAULT_LOCKED) {
S
Steven Whitehouse 已提交
542
		set_page_dirty(page);
543
		wait_for_stable_page(page);
S
Steven Whitehouse 已提交
544
	}
545
	sb_end_pagefault(inode->i_sb);
546
	return ret;
547 548
}

549 550 551 552 553 554 555 556
static vm_fault_t gfs2_fault(struct vm_fault *vmf)
{
	struct inode *inode = file_inode(vmf->vma->vm_file);
	struct gfs2_inode *ip = GFS2_I(inode);
	struct gfs2_holder gh;
	vm_fault_t ret;
	int err;

557
	gfs2_holder_init(ip->i_gl, LM_ST_SHARED, 0, &gh);
558 559 560 561 562 563 564 565 566 567 568 569
	err = gfs2_glock_nq(&gh);
	if (err) {
		ret = block_page_mkwrite_return(err);
		goto out_uninit;
	}
	ret = filemap_fault(vmf);
	gfs2_glock_dq(&gh);
out_uninit:
	gfs2_holder_uninit(&gh);
	return ret;
}

570
static const struct vm_operations_struct gfs2_vm_ops = {
571
	.fault = gfs2_fault,
572
	.map_pages = filemap_map_pages,
573 574 575
	.page_mkwrite = gfs2_page_mkwrite,
};

D
David Teigland 已提交
576
/**
577
 * gfs2_mmap
D
David Teigland 已提交
578 579 580
 * @file: The file to map
 * @vma: The VMA which described the mapping
 *
581 582 583 584 585
 * There is no need to get a lock here unless we should be updating
 * atime. We ignore any locking errors since the only consequence is
 * a missed atime update (which will just be deferred until later).
 *
 * Returns: 0
D
David Teigland 已提交
586 587 588 589
 */

static int gfs2_mmap(struct file *file, struct vm_area_struct *vma)
{
590
	struct gfs2_inode *ip = GFS2_I(file->f_mapping->host);
D
David Teigland 已提交
591

592 593
	if (!(file->f_flags & O_NOATIME) &&
	    !IS_NOATIME(&ip->i_inode)) {
594 595
		struct gfs2_holder i_gh;
		int error;
D
David Teigland 已提交
596

597 598
		error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY,
					   &i_gh);
599 600
		if (error)
			return error;
601 602 603
		/* grab lock to update inode */
		gfs2_glock_dq_uninit(&i_gh);
		file_accessed(file);
604
	}
605
	vma->vm_ops = &gfs2_vm_ops;
D
David Teigland 已提交
606

607
	return 0;
D
David Teigland 已提交
608 609 610
}

/**
611 612 613
 * gfs2_open_common - This is common to open and atomic_open
 * @inode: The inode being opened
 * @file: The file being opened
D
David Teigland 已提交
614
 *
615 616 617 618 619 620
 * This maybe called under a glock or not depending upon how it has
 * been called. We must always be called under a glock for regular
 * files, however. For other file types, it does not matter whether
 * we hold the glock or not.
 *
 * Returns: Error code or 0 for success
D
David Teigland 已提交
621 622
 */

623
int gfs2_open_common(struct inode *inode, struct file *file)
D
David Teigland 已提交
624 625
{
	struct gfs2_file *fp;
626 627 628 629 630 631 632
	int ret;

	if (S_ISREG(inode->i_mode)) {
		ret = generic_file_open(inode, file);
		if (ret)
			return ret;
	}
D
David Teigland 已提交
633

634
	fp = kzalloc(sizeof(struct gfs2_file), GFP_NOFS);
D
David Teigland 已提交
635 636 637
	if (!fp)
		return -ENOMEM;

638
	mutex_init(&fp->f_fl_mutex);
D
David Teigland 已提交
639

640
	gfs2_assert_warn(GFS2_SB(inode), !file->private_data);
641
	file->private_data = fp;
642 643 644 645 646
	if (file->f_mode & FMODE_WRITE) {
		ret = gfs2_qa_get(GFS2_I(inode));
		if (ret)
			goto fail;
	}
647
	return 0;
648 649 650 651 652

fail:
	kfree(file->private_data);
	file->private_data = NULL;
	return ret;
653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674
}

/**
 * gfs2_open - open a file
 * @inode: the inode to open
 * @file: the struct file for this opening
 *
 * After atomic_open, this function is only used for opening files
 * which are already cached. We must still get the glock for regular
 * files to ensure that we have the file size uptodate for the large
 * file check which is in the common code. That is only an issue for
 * regular files though.
 *
 * Returns: errno
 */

static int gfs2_open(struct inode *inode, struct file *file)
{
	struct gfs2_inode *ip = GFS2_I(inode);
	struct gfs2_holder i_gh;
	int error;
	bool need_unlock = false;
D
David Teigland 已提交
675

676
	if (S_ISREG(ip->i_inode.i_mode)) {
D
David Teigland 已提交
677 678 679
		error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY,
					   &i_gh);
		if (error)
680 681 682
			return error;
		need_unlock = true;
	}
D
David Teigland 已提交
683

684
	error = gfs2_open_common(inode, file);
D
David Teigland 已提交
685

686
	if (need_unlock)
D
David Teigland 已提交
687 688 689 690 691 692
		gfs2_glock_dq_uninit(&i_gh);

	return error;
}

/**
693
 * gfs2_release - called to close a struct file
D
David Teigland 已提交
694 695 696 697 698 699
 * @inode: the inode the struct file belongs to
 * @file: the struct file being closed
 *
 * Returns: errno
 */

700
static int gfs2_release(struct inode *inode, struct file *file)
D
David Teigland 已提交
701
{
702
	struct gfs2_inode *ip = GFS2_I(inode);
D
David Teigland 已提交
703

B
Bob Peterson 已提交
704
	kfree(file->private_data);
705
	file->private_data = NULL;
D
David Teigland 已提交
706

707 708 709
	if (file->f_mode & FMODE_WRITE) {
		if (gfs2_rs_active(&ip->i_res))
			gfs2_rs_delete(ip, &inode->i_writecount);
710
		gfs2_qa_put(ip);
711
	}
D
David Teigland 已提交
712 713 714 715 716
	return 0;
}

/**
 * gfs2_fsync - sync the dirty data for a file (across the cluster)
717 718 719
 * @file: the file that points to the dentry
 * @start: the start position in the file to sync
 * @end: the end position in the file to sync
S
Steven Whitehouse 已提交
720
 * @datasync: set if we can ignore timestamp changes
D
David Teigland 已提交
721
 *
722 723 724 725 726 727 728 729 730 731
 * We split the data flushing here so that we don't wait for the data
 * until after we've also sent the metadata to disk. Note that for
 * data=ordered, we will write & wait for the data at the log flush
 * stage anyway, so this is unlikely to make much of a difference
 * except in the data=writeback case.
 *
 * If the fdatawrite fails due to any reason except -EIO, we will
 * continue the remainder of the fsync, although we'll still report
 * the error at the end. This is to match filemap_write_and_wait_range()
 * behaviour.
732
 *
D
David Teigland 已提交
733 734 735
 * Returns: errno
 */

736 737
static int gfs2_fsync(struct file *file, loff_t start, loff_t end,
		      int datasync)
D
David Teigland 已提交
738
{
739 740
	struct address_space *mapping = file->f_mapping;
	struct inode *inode = mapping->host;
741
	int sync_state = inode->i_state & I_DIRTY;
S
Steven Whitehouse 已提交
742
	struct gfs2_inode *ip = GFS2_I(inode);
743
	int ret = 0, ret1 = 0;
D
David Teigland 已提交
744

745 746 747 748 749
	if (mapping->nrpages) {
		ret1 = filemap_fdatawrite_range(mapping, start, end);
		if (ret1 == -EIO)
			return ret1;
	}
750

751 752
	if (!gfs2_is_jdata(ip))
		sync_state &= ~I_DIRTY_PAGES;
S
Steven Whitehouse 已提交
753
	if (datasync)
754
		sync_state &= ~I_DIRTY_SYNC;
D
David Teigland 已提交
755

S
Steven Whitehouse 已提交
756 757
	if (sync_state) {
		ret = sync_inode_metadata(inode, 1);
758
		if (ret)
S
Steven Whitehouse 已提交
759
			return ret;
760
		if (gfs2_is_jdata(ip))
761 762 763
			ret = file_write_and_wait(file);
		if (ret)
			return ret;
764
		gfs2_ail_flush(ip->i_gl, 1);
765 766
	}

767
	if (mapping->nrpages)
768
		ret = file_fdatawait_range(file, start, end);
769 770

	return ret ? ret : ret1;
D
David Teigland 已提交
771 772
}

773 774 775 776 777
static inline bool should_fault_in_pages(ssize_t ret, struct iov_iter *i,
					 size_t *prev_count,
					 size_t *window_size)
{
	size_t count = iov_iter_count(i);
778
	char __user *p;
779 780 781 782 783 784 785 786 787 788 789 790
	int pages = 1;

	if (likely(!count))
		return false;
	if (ret <= 0 && ret != -EFAULT)
		return false;
	if (!iter_is_iovec(i))
		return false;

	if (*prev_count != count || !*window_size) {
		int pages, nr_dirtied;

791
		pages = min_t(int, BIO_MAX_VECS, DIV_ROUND_UP(count, PAGE_SIZE));
792 793 794 795 796 797
		nr_dirtied = max(current->nr_dirtied_pause -
				 current->nr_dirtied, 1);
		pages = min(pages, nr_dirtied);
	}

	*prev_count = count;
798
	p = i->iov[0].iov_base + i->iov_offset;
799 800 801 802
	*window_size = (size_t)PAGE_SIZE * pages - offset_in_page(p);
	return true;
}

803 804
static ssize_t gfs2_file_direct_read(struct kiocb *iocb, struct iov_iter *to,
				     struct gfs2_holder *gh)
805 806 807
{
	struct file *file = iocb->ki_filp;
	struct gfs2_inode *ip = GFS2_I(file->f_mapping->host);
808 809
	size_t prev_count = 0, window_size = 0;
	size_t written = 0;
810 811
	ssize_t ret;

812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829
	/*
	 * In this function, we disable page faults when we're holding the
	 * inode glock while doing I/O.  If a page fault occurs, we indicate
	 * that the inode glock may be dropped, fault in the pages manually,
	 * and retry.
	 *
	 * Unlike generic_file_read_iter, for reads, iomap_dio_rw can trigger
	 * physical as well as manual page faults, and we need to disable both
	 * kinds.
	 *
	 * For direct I/O, gfs2 takes the inode glock in deferred mode.  This
	 * locking mode is compatible with other deferred holders, so multiple
	 * processes and nodes can do direct I/O to a file at the same time.
	 * There's no guarantee that reads or writes will be atomic.  Any
	 * coordination among readers and writers needs to happen externally.
	 */

	if (!iov_iter_count(to))
830 831
		return 0; /* skip atime */

832
	gfs2_holder_init(ip->i_gl, LM_ST_DEFERRED, 0, gh);
833
retry:
834
	ret = gfs2_glock_nq(gh);
835 836
	if (ret)
		goto out_uninit;
837 838 839 840 841 842 843 844 845 846 847 848
retry_under_glock:
	pagefault_disable();
	to->nofault = true;
	ret = iomap_dio_rw(iocb, to, &gfs2_iomap_ops, NULL,
			   IOMAP_DIO_PARTIAL, written);
	to->nofault = false;
	pagefault_enable();
	if (ret > 0)
		written = ret;

	if (should_fault_in_pages(ret, to, &prev_count, &window_size)) {
		size_t leftover;
849

850 851 852 853 854 855 856 857 858 859 860
		gfs2_holder_allow_demote(gh);
		leftover = fault_in_iov_iter_writeable(to, window_size);
		gfs2_holder_disallow_demote(gh);
		if (leftover != window_size) {
			if (!gfs2_holder_queued(gh))
				goto retry;
			goto retry_under_glock;
		}
	}
	if (gfs2_holder_queued(gh))
		gfs2_glock_dq(gh);
861
out_uninit:
862
	gfs2_holder_uninit(gh);
863 864 865
	if (ret < 0)
		return ret;
	return written;
866 867
}

868 869
static ssize_t gfs2_file_direct_write(struct kiocb *iocb, struct iov_iter *from,
				      struct gfs2_holder *gh)
870 871 872 873
{
	struct file *file = iocb->ki_filp;
	struct inode *inode = file->f_mapping->host;
	struct gfs2_inode *ip = GFS2_I(inode);
874 875
	size_t prev_count = 0, window_size = 0;
	size_t read = 0;
876 877
	ssize_t ret;

878 879 880 881 882 883 884 885 886 887
	/*
	 * In this function, we disable page faults when we're holding the
	 * inode glock while doing I/O.  If a page fault occurs, we indicate
	 * that the inode glock may be dropped, fault in the pages manually,
	 * and retry.
	 *
	 * For writes, iomap_dio_rw only triggers manual page faults, so we
	 * don't need to disable physical ones.
	 */

888 889 890 891 892 893 894 895
	/*
	 * Deferred lock, even if its a write, since we do no allocation on
	 * this path. All we need to change is the atime, and this lock mode
	 * ensures that other nodes have flushed their buffered read caches
	 * (i.e. their page cache entries for this inode). We do not,
	 * unfortunately, have the option of only flushing a range like the
	 * VFS does.
	 */
896
	gfs2_holder_init(ip->i_gl, LM_ST_DEFERRED, 0, gh);
897
retry:
898
	ret = gfs2_glock_nq(gh);
899 900
	if (ret)
		goto out_uninit;
901
retry_under_glock:
902
	/* Silently fall back to buffered I/O when writing beyond EOF */
903
	if (iocb->ki_pos + iov_iter_count(from) > i_size_read(&ip->i_inode))
904 905
		goto out;

906 907 908 909 910
	from->nofault = true;
	ret = iomap_dio_rw(iocb, from, &gfs2_iomap_ops, NULL,
			   IOMAP_DIO_PARTIAL, read);
	from->nofault = false;

911 912
	if (ret == -ENOTBLK)
		ret = 0;
913 914 915 916 917 918 919 920 921 922 923 924 925 926 927
	if (ret > 0)
		read = ret;

	if (should_fault_in_pages(ret, from, &prev_count, &window_size)) {
		size_t leftover;

		gfs2_holder_allow_demote(gh);
		leftover = fault_in_iov_iter_readable(from, window_size);
		gfs2_holder_disallow_demote(gh);
		if (leftover != window_size) {
			if (!gfs2_holder_queued(gh))
				goto retry;
			goto retry_under_glock;
		}
	}
928
out:
929 930
	if (gfs2_holder_queued(gh))
		gfs2_glock_dq(gh);
931
out_uninit:
932
	gfs2_holder_uninit(gh);
933 934 935
	if (ret < 0)
		return ret;
	return read;
936 937 938 939
}

static ssize_t gfs2_file_read_iter(struct kiocb *iocb, struct iov_iter *to)
{
940 941
	struct gfs2_inode *ip;
	struct gfs2_holder gh;
942
	size_t prev_count = 0, window_size = 0;
943
	size_t written = 0;
944 945
	ssize_t ret;

946 947 948 949 950 951 952
	/*
	 * In this function, we disable page faults when we're holding the
	 * inode glock while doing I/O.  If a page fault occurs, we indicate
	 * that the inode glock may be dropped, fault in the pages manually,
	 * and retry.
	 */

953
	if (iocb->ki_flags & IOCB_DIRECT) {
954
		ret = gfs2_file_direct_read(iocb, to, &gh);
955 956 957 958
		if (likely(ret != -ENOTBLK))
			return ret;
		iocb->ki_flags &= ~IOCB_DIRECT;
	}
959 960 961 962 963 964 965 966 967 968 969 970 971 972 973
	iocb->ki_flags |= IOCB_NOIO;
	ret = generic_file_read_iter(iocb, to);
	iocb->ki_flags &= ~IOCB_NOIO;
	if (ret >= 0) {
		if (!iov_iter_count(to))
			return ret;
		written = ret;
	} else {
		if (ret != -EAGAIN)
			return ret;
		if (iocb->ki_flags & IOCB_NOWAIT)
			return ret;
	}
	ip = GFS2_I(iocb->ki_filp->f_mapping->host);
	gfs2_holder_init(ip->i_gl, LM_ST_SHARED, 0, &gh);
974
retry:
975 976 977
	ret = gfs2_glock_nq(&gh);
	if (ret)
		goto out_uninit;
978 979
retry_under_glock:
	pagefault_disable();
980
	ret = generic_file_read_iter(iocb, to);
981
	pagefault_enable();
982 983
	if (ret > 0)
		written += ret;
984 985 986 987 988 989 990 991 992 993 994 995 996 997 998 999 1000 1001

	if (should_fault_in_pages(ret, to, &prev_count, &window_size)) {
		size_t leftover;

		gfs2_holder_allow_demote(&gh);
		leftover = fault_in_iov_iter_writeable(to, window_size);
		gfs2_holder_disallow_demote(&gh);
		if (leftover != window_size) {
			if (!gfs2_holder_queued(&gh)) {
				if (written)
					goto out_uninit;
				goto retry;
			}
			goto retry_under_glock;
		}
	}
	if (gfs2_holder_queued(&gh))
		gfs2_glock_dq(&gh);
1002 1003 1004
out_uninit:
	gfs2_holder_uninit(&gh);
	return written ? written : ret;
1005 1006
}

A
Andreas Gruenbacher 已提交
1007 1008 1009
static ssize_t gfs2_file_buffered_write(struct kiocb *iocb,
					struct iov_iter *from,
					struct gfs2_holder *gh)
1010 1011 1012
{
	struct file *file = iocb->ki_filp;
	struct inode *inode = file_inode(file);
1013 1014
	struct gfs2_inode *ip = GFS2_I(inode);
	struct gfs2_sbd *sdp = GFS2_SB(inode);
A
Andreas Gruenbacher 已提交
1015
	struct gfs2_holder *statfs_gh = NULL;
1016
	size_t prev_count = 0, window_size = 0;
1017
	size_t orig_count = iov_iter_count(from);
1018
	size_t read = 0;
1019 1020
	ssize_t ret;

1021 1022 1023 1024 1025 1026 1027
	/*
	 * In this function, we disable page faults when we're holding the
	 * inode glock while doing I/O.  If a page fault occurs, we indicate
	 * that the inode glock may be dropped, fault in the pages manually,
	 * and retry.
	 */

A
Andreas Gruenbacher 已提交
1028 1029 1030 1031 1032 1033 1034
	if (inode == sdp->sd_rindex) {
		statfs_gh = kmalloc(sizeof(*statfs_gh), GFP_NOFS);
		if (!statfs_gh)
			return -ENOMEM;
	}

	gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, gh);
1035
retry:
A
Andreas Gruenbacher 已提交
1036
	ret = gfs2_glock_nq(gh);
1037 1038
	if (ret)
		goto out_uninit;
1039
retry_under_glock:
1040 1041 1042 1043
	if (inode == sdp->sd_rindex) {
		struct gfs2_inode *m_ip = GFS2_I(sdp->sd_statfs_inode);

		ret = gfs2_glock_nq_init(m_ip->i_gl, LM_ST_EXCLUSIVE,
A
Andreas Gruenbacher 已提交
1044
					 GL_NOCACHE, statfs_gh);
1045 1046 1047 1048
		if (ret)
			goto out_unlock;
	}

1049
	current->backing_dev_info = inode_to_bdi(inode);
1050
	pagefault_disable();
1051
	ret = iomap_file_buffered_write(iocb, from, &gfs2_iomap_ops);
1052
	pagefault_enable();
1053
	current->backing_dev_info = NULL;
1054
	if (ret > 0) {
1055
		iocb->ki_pos += ret;
1056 1057
		read += ret;
	}
1058

A
Andreas Gruenbacher 已提交
1059 1060
	if (inode == sdp->sd_rindex)
		gfs2_glock_dq_uninit(statfs_gh);
1061

1062
	from->count = orig_count - read;
1063 1064 1065 1066 1067 1068 1069
	if (should_fault_in_pages(ret, from, &prev_count, &window_size)) {
		size_t leftover;

		gfs2_holder_allow_demote(gh);
		leftover = fault_in_iov_iter_readable(from, window_size);
		gfs2_holder_disallow_demote(gh);
		if (leftover != window_size) {
1070
			from->count = min(from->count, window_size - leftover);
1071 1072 1073 1074 1075 1076 1077 1078
			if (!gfs2_holder_queued(gh)) {
				if (read)
					goto out_uninit;
				goto retry;
			}
			goto retry_under_glock;
		}
	}
1079
out_unlock:
1080 1081
	if (gfs2_holder_queued(gh))
		gfs2_glock_dq(gh);
1082
out_uninit:
A
Andreas Gruenbacher 已提交
1083 1084 1085
	gfs2_holder_uninit(gh);
	if (statfs_gh)
		kfree(statfs_gh);
1086
	return read ? read : ret;
1087 1088
}

1089
/**
A
Al Viro 已提交
1090
 * gfs2_file_write_iter - Perform a write to a file
1091
 * @iocb: The io context
1092
 * @from: The data to write
1093 1094 1095 1096 1097 1098 1099 1100
 *
 * We have to do a lock/unlock here to refresh the inode size for
 * O_APPEND writes, otherwise we can land up writing at the wrong
 * offset. There is still a race, but provided the app is using its
 * own file locking, this will make O_APPEND work as expected.
 *
 */

A
Al Viro 已提交
1101
static ssize_t gfs2_file_write_iter(struct kiocb *iocb, struct iov_iter *from)
1102 1103
{
	struct file *file = iocb->ki_filp;
1104 1105
	struct inode *inode = file_inode(file);
	struct gfs2_inode *ip = GFS2_I(inode);
1106
	struct gfs2_holder gh;
1107
	ssize_t ret;
1108

A
Al Viro 已提交
1109
	gfs2_size_hint(file, iocb->ki_pos, iov_iter_count(from));
1110

1111
	if (iocb->ki_flags & IOCB_APPEND) {
1112 1113
		ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, 0, &gh);
		if (ret)
1114
			return ret;
1115 1116 1117
		gfs2_glock_dq_uninit(&gh);
	}

1118 1119 1120
	inode_lock(inode);
	ret = generic_write_checks(iocb, from);
	if (ret <= 0)
1121
		goto out_unlock;
1122 1123 1124

	ret = file_remove_privs(file);
	if (ret)
1125
		goto out_unlock;
1126 1127 1128

	ret = file_update_time(file);
	if (ret)
1129
		goto out_unlock;
1130

1131 1132
	if (iocb->ki_flags & IOCB_DIRECT) {
		struct address_space *mapping = file->f_mapping;
1133
		ssize_t buffered, ret2;
1134

1135
		ret = gfs2_file_direct_write(iocb, from, &gh);
1136
		if (ret < 0 || !iov_iter_count(from))
1137
			goto out_unlock;
1138

1139
		iocb->ki_flags |= IOCB_DSYNC;
A
Andreas Gruenbacher 已提交
1140
		buffered = gfs2_file_buffered_write(iocb, from, &gh);
1141 1142 1143
		if (unlikely(buffered <= 0)) {
			if (!ret)
				ret = buffered;
1144
			goto out_unlock;
1145
		}
1146 1147 1148 1149

		/*
		 * We need to ensure that the page cache pages are written to
		 * disk and invalidated to preserve the expected O_DIRECT
1150 1151 1152
		 * semantics.  If the writeback or invalidate fails, only report
		 * the direct I/O range as we don't know if the buffered pages
		 * made it to disk.
1153
		 */
1154 1155 1156 1157 1158 1159
		ret2 = generic_write_sync(iocb, buffered);
		invalidate_mapping_pages(mapping,
				(iocb->ki_pos - buffered) >> PAGE_SHIFT,
				(iocb->ki_pos - 1) >> PAGE_SHIFT);
		if (!ret || ret2 > 0)
			ret += ret2;
1160
	} else {
A
Andreas Gruenbacher 已提交
1161
		ret = gfs2_file_buffered_write(iocb, from, &gh);
1162
		if (likely(ret > 0))
1163
			ret = generic_write_sync(iocb, ret);
1164
	}
1165

1166
out_unlock:
1167
	inode_unlock(inode);
1168
	return ret;
1169 1170
}

1171 1172 1173
static int fallocate_chunk(struct inode *inode, loff_t offset, loff_t len,
			   int mode)
{
1174
	struct super_block *sb = inode->i_sb;
1175
	struct gfs2_inode *ip = GFS2_I(inode);
1176
	loff_t end = offset + len;
1177 1178 1179 1180 1181
	struct buffer_head *dibh;
	int error;

	error = gfs2_meta_inode_buffer(ip, &dibh);
	if (unlikely(error))
1182
		return error;
1183

1184
	gfs2_trans_add_meta(ip->i_gl, dibh);
1185 1186

	if (gfs2_is_stuffed(ip)) {
1187
		error = gfs2_unstuff_dinode(ip);
1188 1189 1190 1191
		if (unlikely(error))
			goto out;
	}

1192
	while (offset < end) {
1193 1194
		struct iomap iomap = { };

1195
		error = gfs2_iomap_alloc(inode, offset, end - offset, &iomap);
1196
		if (error)
1197
			goto out;
1198
		offset = iomap.offset + iomap.length;
1199
		if (!(iomap.flags & IOMAP_F_NEW))
1200
			continue;
1201 1202 1203 1204 1205
		error = sb_issue_zeroout(sb, iomap.addr >> inode->i_blkbits,
					 iomap.length >> inode->i_blkbits,
					 GFP_NOFS);
		if (error) {
			fs_err(GFS2_SB(inode), "Failed to zero data buffers\n");
1206
			goto out;
1207
		}
1208 1209
	}
out:
1210
	brelse(dibh);
1211 1212
	return error;
}
1213

1214 1215 1216 1217 1218 1219 1220 1221 1222 1223 1224 1225 1226 1227
/**
 * calc_max_reserv() - Reverse of write_calc_reserv. Given a number of
 *                     blocks, determine how many bytes can be written.
 * @ip:          The inode in question.
 * @len:         Max cap of bytes. What we return in *len must be <= this.
 * @data_blocks: Compute and return the number of data blocks needed
 * @ind_blocks:  Compute and return the number of indirect blocks needed
 * @max_blocks:  The total blocks available to work with.
 *
 * Returns: void, but @len, @data_blocks and @ind_blocks are filled in.
 */
static void calc_max_reserv(struct gfs2_inode *ip, loff_t *len,
			    unsigned int *data_blocks, unsigned int *ind_blocks,
			    unsigned int max_blocks)
1228
{
1229
	loff_t max = *len;
1230 1231 1232 1233 1234 1235 1236
	const struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode);
	unsigned int tmp, max_data = max_blocks - 3 * (sdp->sd_max_height - 1);

	for (tmp = max_data; tmp > sdp->sd_diptrs;) {
		tmp = DIV_ROUND_UP(tmp, sdp->sd_inptrs);
		max_data -= tmp;
	}
1237

1238 1239 1240 1241 1242 1243 1244 1245 1246
	*data_blocks = max_data;
	*ind_blocks = max_blocks - max_data;
	*len = ((loff_t)max_data - 3) << sdp->sd_sb.sb_bsize_shift;
	if (*len > max) {
		*len = max;
		gfs2_write_calc_reserv(ip, max, data_blocks, ind_blocks);
	}
}

1247
static long __gfs2_fallocate(struct file *file, int mode, loff_t offset, loff_t len)
1248
{
A
Al Viro 已提交
1249
	struct inode *inode = file_inode(file);
1250 1251
	struct gfs2_sbd *sdp = GFS2_SB(inode);
	struct gfs2_inode *ip = GFS2_I(inode);
1252
	struct gfs2_alloc_parms ap = { .aflags = 0, };
1253
	unsigned int data_blocks = 0, ind_blocks = 0, rblocks;
1254
	loff_t bytes, max_bytes, max_blks;
1255
	int error;
1256 1257
	const loff_t pos = offset;
	const loff_t count = len;
1258
	loff_t bsize_mask = ~((loff_t)sdp->sd_sb.sb_bsize - 1);
1259
	loff_t next = (offset + len - 1) >> sdp->sd_sb.sb_bsize_shift;
1260
	loff_t max_chunk_size = UINT_MAX & bsize_mask;
1261

1262 1263
	next = (next + 1) << sdp->sd_sb.sb_bsize_shift;

1264
	offset &= bsize_mask;
1265 1266 1267 1268 1269

	len = next - offset;
	bytes = sdp->sd_max_rg_data * sdp->sd_sb.sb_bsize / 2;
	if (!bytes)
		bytes = UINT_MAX;
1270 1271 1272
	bytes &= bsize_mask;
	if (bytes == 0)
		bytes = sdp->sd_sb.sb_bsize;
1273

1274
	gfs2_size_hint(file, offset, len);
B
Bob Peterson 已提交
1275

1276 1277 1278
	gfs2_write_calc_reserv(ip, PAGE_SIZE, &data_blocks, &ind_blocks);
	ap.min_target = data_blocks + ind_blocks;

1279 1280 1281
	while (len > 0) {
		if (len < bytes)
			bytes = len;
1282 1283 1284 1285 1286
		if (!gfs2_write_alloc_required(ip, offset, bytes)) {
			len -= bytes;
			offset += bytes;
			continue;
		}
1287 1288 1289 1290 1291 1292 1293 1294 1295 1296 1297

		/* We need to determine how many bytes we can actually
		 * fallocate without exceeding quota or going over the
		 * end of the fs. We start off optimistically by assuming
		 * we can write max_bytes */
		max_bytes = (len > max_chunk_size) ? max_chunk_size : len;

		/* Since max_bytes is most likely a theoretical max, we
		 * calculate a more realistic 'bytes' to serve as a good
		 * starting point for the number of bytes we may be able
		 * to write */
1298
		gfs2_write_calc_reserv(ip, bytes, &data_blocks, &ind_blocks);
1299
		ap.target = data_blocks + ind_blocks;
1300 1301

		error = gfs2_quota_lock_check(ip, &ap);
1302
		if (error)
1303
			return error;
1304 1305
		/* ap.allowed tells us how many blocks quota will allow
		 * us to write. Check if this reduces max_blks */
1306 1307
		max_blks = UINT_MAX;
		if (ap.allowed)
1308
			max_blks = ap.allowed;
1309

1310
		error = gfs2_inplace_reserve(ip, &ap);
1311
		if (error)
1312
			goto out_qunlock;
1313 1314

		/* check if the selected rgrp limits our max_blks further */
1315 1316
		if (ip->i_res.rs_reserved < max_blks)
			max_blks = ip->i_res.rs_reserved;
1317 1318 1319 1320 1321 1322

		/* Almost done. Calculate bytes that can be written using
		 * max_blks. We also recompute max_bytes, data_blocks and
		 * ind_blocks */
		calc_max_reserv(ip, &max_bytes, &data_blocks,
				&ind_blocks, max_blks);
1323 1324

		rblocks = RES_DINODE + ind_blocks + RES_STATFS + RES_QUOTA +
1325
			  RES_RG_HDR + gfs2_rg_blocks(ip, data_blocks + ind_blocks);
1326 1327 1328 1329
		if (gfs2_is_jdata(ip))
			rblocks += data_blocks ? data_blocks : 1;

		error = gfs2_trans_begin(sdp, rblocks,
1330
					 PAGE_SIZE >> inode->i_blkbits);
1331 1332 1333 1334 1335 1336 1337 1338 1339 1340 1341 1342 1343 1344
		if (error)
			goto out_trans_fail;

		error = fallocate_chunk(inode, offset, max_bytes, mode);
		gfs2_trans_end(sdp);

		if (error)
			goto out_trans_fail;

		len -= max_bytes;
		offset += max_bytes;
		gfs2_inplace_release(ip);
		gfs2_quota_unlock(ip);
	}
1345

1346
	if (!(mode & FALLOC_FL_KEEP_SIZE) && (pos + count) > inode->i_size)
1347
		i_size_write(inode, pos + count);
1348 1349
	file_update_time(file);
	mark_inode_dirty(inode);
1350

1351 1352 1353 1354
	if ((file->f_flags & O_DSYNC) || IS_SYNC(file->f_mapping->host))
		return vfs_fsync_range(file, pos, pos + count - 1,
			       (file->f_flags & __O_SYNC) ? 0 : 1);
	return 0;
1355 1356 1357 1358 1359

out_trans_fail:
	gfs2_inplace_release(ip);
out_qunlock:
	gfs2_quota_unlock(ip);
1360 1361 1362 1363 1364 1365
	return error;
}

static long gfs2_fallocate(struct file *file, int mode, loff_t offset, loff_t len)
{
	struct inode *inode = file_inode(file);
1366
	struct gfs2_sbd *sdp = GFS2_SB(inode);
1367 1368 1369 1370
	struct gfs2_inode *ip = GFS2_I(inode);
	struct gfs2_holder gh;
	int ret;

1371
	if (mode & ~(FALLOC_FL_PUNCH_HOLE | FALLOC_FL_KEEP_SIZE))
1372 1373 1374
		return -EOPNOTSUPP;
	/* fallocate is needed by gfs2_grow to reserve space in the rindex */
	if (gfs2_is_jdata(ip) && inode != sdp->sd_rindex)
1375 1376
		return -EOPNOTSUPP;

A
Al Viro 已提交
1377
	inode_lock(inode);
1378 1379 1380 1381 1382 1383 1384 1385 1386 1387 1388 1389 1390 1391 1392 1393 1394

	gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh);
	ret = gfs2_glock_nq(&gh);
	if (ret)
		goto out_uninit;

	if (!(mode & FALLOC_FL_KEEP_SIZE) &&
	    (offset + len) > inode->i_size) {
		ret = inode_newsize_ok(inode, offset + len);
		if (ret)
			goto out_unlock;
	}

	ret = get_write_access(inode);
	if (ret)
		goto out_unlock;

1395 1396 1397 1398 1399 1400 1401
	if (mode & FALLOC_FL_PUNCH_HOLE) {
		ret = __gfs2_punch_hole(file, offset, len);
	} else {
		ret = __gfs2_fallocate(file, mode, offset, len);
		if (ret)
			gfs2_rs_deltree(&ip->i_res);
	}
1402

1403
	put_write_access(inode);
1404
out_unlock:
1405
	gfs2_glock_dq(&gh);
1406
out_uninit:
1407
	gfs2_holder_uninit(&gh);
A
Al Viro 已提交
1408
	inode_unlock(inode);
1409
	return ret;
1410 1411
}

1412 1413 1414 1415
static ssize_t gfs2_file_splice_write(struct pipe_inode_info *pipe,
				      struct file *out, loff_t *ppos,
				      size_t len, unsigned int flags)
{
1416
	ssize_t ret;
1417 1418 1419

	gfs2_size_hint(out, *ppos, len);

1420 1421
	ret = iter_file_splice_write(pipe, out, ppos, len, flags);
	return ret;
1422 1423
}

1424 1425
#ifdef CONFIG_GFS2_FS_LOCKING_DLM

D
David Teigland 已提交
1426 1427 1428 1429 1430 1431 1432 1433 1434 1435 1436
/**
 * gfs2_lock - acquire/release a posix lock on a file
 * @file: the file pointer
 * @cmd: either modify or retrieve lock state, possibly wait
 * @fl: type and range of lock
 *
 * Returns: errno
 */

static int gfs2_lock(struct file *file, int cmd, struct file_lock *fl)
{
1437 1438
	struct gfs2_inode *ip = GFS2_I(file->f_mapping->host);
	struct gfs2_sbd *sdp = GFS2_SB(file->f_mapping->host);
1439
	struct lm_lockstruct *ls = &sdp->sd_lockstruct;
D
David Teigland 已提交
1440 1441 1442

	if (!(fl->fl_flags & FL_POSIX))
		return -ENOLCK;
M
Marc Eshel 已提交
1443 1444 1445 1446 1447
	if (cmd == F_CANCELLK) {
		/* Hack: */
		cmd = F_SETLK;
		fl->fl_type = F_UNLCK;
	}
1448
	if (unlikely(gfs2_withdrawn(sdp))) {
1449
		if (fl->fl_type == F_UNLCK)
1450
			locks_lock_file_wait(file, fl);
1451
		return -EIO;
1452
	}
D
David Teigland 已提交
1453
	if (IS_GETLK(cmd))
1454
		return dlm_posix_get(ls->ls_dlm, ip->i_no_addr, file, fl);
D
David Teigland 已提交
1455
	else if (fl->fl_type == F_UNLCK)
1456
		return dlm_posix_unlock(ls->ls_dlm, ip->i_no_addr, file, fl);
D
David Teigland 已提交
1457
	else
1458
		return dlm_posix_lock(ls->ls_dlm, ip->i_no_addr, file, cmd, fl);
D
David Teigland 已提交
1459 1460 1461 1462
}

static int do_flock(struct file *file, int cmd, struct file_lock *fl)
{
1463
	struct gfs2_file *fp = file->private_data;
D
David Teigland 已提交
1464
	struct gfs2_holder *fl_gh = &fp->f_fl_gh;
A
Al Viro 已提交
1465
	struct gfs2_inode *ip = GFS2_I(file_inode(file));
D
David Teigland 已提交
1466 1467
	struct gfs2_glock *gl;
	unsigned int state;
B
Bob Peterson 已提交
1468
	u16 flags;
D
David Teigland 已提交
1469
	int error = 0;
1470
	int sleeptime;
D
David Teigland 已提交
1471 1472

	state = (fl->fl_type == F_WRLCK) ? LM_ST_EXCLUSIVE : LM_ST_SHARED;
1473
	flags = (IS_SETLKW(cmd) ? 0 : LM_FLAG_TRY_1CB) | GL_EXACT;
D
David Teigland 已提交
1474

1475
	mutex_lock(&fp->f_fl_mutex);
D
David Teigland 已提交
1476

1477
	if (gfs2_holder_initialized(fl_gh)) {
1478
		struct file_lock request;
D
David Teigland 已提交
1479 1480
		if (fl_gh->gh_state == state)
			goto out;
1481 1482 1483 1484
		locks_init_lock(&request);
		request.fl_type = F_UNLCK;
		request.fl_flags = FL_FLOCK;
		locks_lock_file_wait(file, &request);
1485
		gfs2_glock_dq(fl_gh);
1486
		gfs2_holder_reinit(state, flags, fl_gh);
D
David Teigland 已提交
1487
	} else {
1488 1489
		error = gfs2_glock_get(GFS2_SB(&ip->i_inode), ip->i_no_addr,
				       &gfs2_flock_glops, CREATE, &gl);
D
David Teigland 已提交
1490 1491
		if (error)
			goto out;
1492 1493
		gfs2_holder_init(gl, state, flags, fl_gh);
		gfs2_glock_put(gl);
D
David Teigland 已提交
1494
	}
1495 1496 1497 1498 1499 1500 1501 1502
	for (sleeptime = 1; sleeptime <= 4; sleeptime <<= 1) {
		error = gfs2_glock_nq(fl_gh);
		if (error != GLR_TRYFAILED)
			break;
		fl_gh->gh_flags = LM_FLAG_TRY | GL_EXACT;
		fl_gh->gh_error = 0;
		msleep(sleeptime);
	}
D
David Teigland 已提交
1503 1504 1505 1506 1507
	if (error) {
		gfs2_holder_uninit(fl_gh);
		if (error == GLR_TRYFAILED)
			error = -EAGAIN;
	} else {
1508
		error = locks_lock_file_wait(file, fl);
1509
		gfs2_assert_warn(GFS2_SB(&ip->i_inode), !error);
D
David Teigland 已提交
1510 1511
	}

1512
out:
1513
	mutex_unlock(&fp->f_fl_mutex);
D
David Teigland 已提交
1514 1515 1516 1517 1518
	return error;
}

static void do_unflock(struct file *file, struct file_lock *fl)
{
1519
	struct gfs2_file *fp = file->private_data;
D
David Teigland 已提交
1520 1521
	struct gfs2_holder *fl_gh = &fp->f_fl_gh;

1522
	mutex_lock(&fp->f_fl_mutex);
1523
	locks_lock_file_wait(file, fl);
A
Andreas Gruenbacher 已提交
1524
	if (gfs2_holder_initialized(fl_gh)) {
1525
		gfs2_glock_dq(fl_gh);
1526 1527
		gfs2_holder_uninit(fl_gh);
	}
1528
	mutex_unlock(&fp->f_fl_mutex);
D
David Teigland 已提交
1529 1530 1531 1532 1533 1534 1535 1536 1537 1538 1539 1540 1541 1542 1543 1544 1545 1546 1547
}

/**
 * gfs2_flock - acquire/release a flock lock on a file
 * @file: the file pointer
 * @cmd: either modify or retrieve lock state, possibly wait
 * @fl: type and range of lock
 *
 * Returns: errno
 */

static int gfs2_flock(struct file *file, int cmd, struct file_lock *fl)
{
	if (!(fl->fl_flags & FL_FLOCK))
		return -ENOLCK;

	if (fl->fl_type == F_UNLCK) {
		do_unflock(file, fl);
		return 0;
1548
	} else {
D
David Teigland 已提交
1549
		return do_flock(file, cmd, fl);
1550
	}
D
David Teigland 已提交
1551 1552
}

1553
const struct file_operations gfs2_file_fops = {
1554
	.llseek		= gfs2_llseek,
1555
	.read_iter	= gfs2_file_read_iter,
A
Al Viro 已提交
1556
	.write_iter	= gfs2_file_write_iter,
1557
	.iopoll		= iocb_bio_iopoll,
1558
	.unlocked_ioctl	= gfs2_ioctl,
A
Arnd Bergmann 已提交
1559
	.compat_ioctl	= gfs2_compat_ioctl,
1560 1561
	.mmap		= gfs2_mmap,
	.open		= gfs2_open,
1562
	.release	= gfs2_release,
1563 1564 1565
	.fsync		= gfs2_fsync,
	.lock		= gfs2_lock,
	.flock		= gfs2_flock,
1566
	.splice_read	= generic_file_splice_read,
1567
	.splice_write	= gfs2_file_splice_write,
1568
	.setlease	= simple_nosetlease,
1569
	.fallocate	= gfs2_fallocate,
D
David Teigland 已提交
1570 1571
};

1572
const struct file_operations gfs2_dir_fops = {
A
Al Viro 已提交
1573
	.iterate_shared	= gfs2_readdir,
1574
	.unlocked_ioctl	= gfs2_ioctl,
A
Arnd Bergmann 已提交
1575
	.compat_ioctl	= gfs2_compat_ioctl,
1576
	.open		= gfs2_open,
1577
	.release	= gfs2_release,
1578 1579 1580
	.fsync		= gfs2_fsync,
	.lock		= gfs2_lock,
	.flock		= gfs2_flock,
1581
	.llseek		= default_llseek,
D
David Teigland 已提交
1582 1583
};

1584 1585
#endif /* CONFIG_GFS2_FS_LOCKING_DLM */

1586
const struct file_operations gfs2_file_fops_nolock = {
1587
	.llseek		= gfs2_llseek,
1588
	.read_iter	= gfs2_file_read_iter,
A
Al Viro 已提交
1589
	.write_iter	= gfs2_file_write_iter,
1590
	.iopoll		= iocb_bio_iopoll,
1591
	.unlocked_ioctl	= gfs2_ioctl,
A
Arnd Bergmann 已提交
1592
	.compat_ioctl	= gfs2_compat_ioctl,
1593 1594
	.mmap		= gfs2_mmap,
	.open		= gfs2_open,
1595
	.release	= gfs2_release,
1596
	.fsync		= gfs2_fsync,
1597
	.splice_read	= generic_file_splice_read,
1598
	.splice_write	= gfs2_file_splice_write,
1599
	.setlease	= generic_setlease,
1600
	.fallocate	= gfs2_fallocate,
1601 1602
};

1603
const struct file_operations gfs2_dir_fops_nolock = {
A
Al Viro 已提交
1604
	.iterate_shared	= gfs2_readdir,
1605
	.unlocked_ioctl	= gfs2_ioctl,
A
Arnd Bergmann 已提交
1606
	.compat_ioctl	= gfs2_compat_ioctl,
1607
	.open		= gfs2_open,
1608
	.release	= gfs2_release,
1609
	.fsync		= gfs2_fsync,
1610
	.llseek		= default_llseek,
1611 1612
};