dir.c 29.3 KB
Newer Older
L
Linus Torvalds 已提交
1 2 3 4 5 6 7 8 9 10
/*
 * dir.c - Operations for sysfs directories.
 */

#undef DEBUG

#include <linux/fs.h>
#include <linux/mount.h>
#include <linux/module.h>
#include <linux/kobject.h>
11
#include <linux/namei.h>
12
#include <linux/idr.h>
13
#include <linux/completion.h>
14
#include <asm/semaphore.h>
L
Linus Torvalds 已提交
15 16
#include "sysfs.h"

17
DEFINE_MUTEX(sysfs_mutex);
T
Tejun Heo 已提交
18
spinlock_t sysfs_assoc_lock = SPIN_LOCK_UNLOCKED;
L
Linus Torvalds 已提交
19

20 21 22
static spinlock_t sysfs_ino_lock = SPIN_LOCK_UNLOCKED;
static DEFINE_IDA(sysfs_ino_ida);

23 24 25 26 27 28 29 30
/**
 *	sysfs_link_sibling - link sysfs_dirent into sibling list
 *	@sd: sysfs_dirent of interest
 *
 *	Link @sd into its sibling list which starts from
 *	sd->s_parent->s_children.
 *
 *	Locking:
31
 *	mutex_lock(sysfs_mutex)
32
 */
33
void sysfs_link_sibling(struct sysfs_dirent *sd)
34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49
{
	struct sysfs_dirent *parent_sd = sd->s_parent;

	BUG_ON(sd->s_sibling);
	sd->s_sibling = parent_sd->s_children;
	parent_sd->s_children = sd;
}

/**
 *	sysfs_unlink_sibling - unlink sysfs_dirent from sibling list
 *	@sd: sysfs_dirent of interest
 *
 *	Unlink @sd from its sibling list which starts from
 *	sd->s_parent->s_children.
 *
 *	Locking:
50
 *	mutex_lock(sysfs_mutex)
51
 */
52
void sysfs_unlink_sibling(struct sysfs_dirent *sd)
53 54 55 56 57 58 59 60 61 62 63 64
{
	struct sysfs_dirent **pos;

	for (pos = &sd->s_parent->s_children; *pos; pos = &(*pos)->s_sibling) {
		if (*pos == sd) {
			*pos = sd->s_sibling;
			sd->s_sibling = NULL;
			break;
		}
	}
}

T
Tejun Heo 已提交
65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162
/**
 *	sysfs_get_dentry - get dentry for the given sysfs_dirent
 *	@sd: sysfs_dirent of interest
 *
 *	Get dentry for @sd.  Dentry is looked up if currently not
 *	present.  This function climbs sysfs_dirent tree till it
 *	reaches a sysfs_dirent with valid dentry attached and descends
 *	down from there looking up dentry for each step.
 *
 *	LOCKING:
 *	Kernel thread context (may sleep)
 *
 *	RETURNS:
 *	Pointer to found dentry on success, ERR_PTR() value on error.
 */
struct dentry *sysfs_get_dentry(struct sysfs_dirent *sd)
{
	struct sysfs_dirent *cur;
	struct dentry *parent_dentry, *dentry;
	int i, depth;

	/* Find the first parent which has valid s_dentry and get the
	 * dentry.
	 */
	mutex_lock(&sysfs_mutex);
 restart0:
	spin_lock(&sysfs_assoc_lock);
 restart1:
	spin_lock(&dcache_lock);

	dentry = NULL;
	depth = 0;
	cur = sd;
	while (!cur->s_dentry || !cur->s_dentry->d_inode) {
		if (cur->s_flags & SYSFS_FLAG_REMOVED) {
			dentry = ERR_PTR(-ENOENT);
			depth = 0;
			break;
		}
		cur = cur->s_parent;
		depth++;
	}
	if (!IS_ERR(dentry))
		dentry = dget_locked(cur->s_dentry);

	spin_unlock(&dcache_lock);
	spin_unlock(&sysfs_assoc_lock);

	/* from the found dentry, look up depth times */
	while (depth--) {
		/* find and get depth'th ancestor */
		for (cur = sd, i = 0; cur && i < depth; i++)
			cur = cur->s_parent;

		/* This can happen if tree structure was modified due
		 * to move/rename.  Restart.
		 */
		if (i != depth) {
			dput(dentry);
			goto restart0;
		}

		sysfs_get(cur);

		mutex_unlock(&sysfs_mutex);

		/* look it up */
		parent_dentry = dentry;
		dentry = lookup_one_len_kern(cur->s_name, parent_dentry,
					     strlen(cur->s_name));
		dput(parent_dentry);

		if (IS_ERR(dentry)) {
			sysfs_put(cur);
			return dentry;
		}

		mutex_lock(&sysfs_mutex);
		spin_lock(&sysfs_assoc_lock);

		/* This, again, can happen if tree structure has
		 * changed and we looked up the wrong thing.  Restart.
		 */
		if (cur->s_dentry != dentry) {
			dput(dentry);
			sysfs_put(cur);
			goto restart1;
		}

		spin_unlock(&sysfs_assoc_lock);

		sysfs_put(cur);
	}

	mutex_unlock(&sysfs_mutex);
	return dentry;
}

163 164 165 166 167 168 169 170 171 172 173 174
/**
 *	sysfs_get_active - get an active reference to sysfs_dirent
 *	@sd: sysfs_dirent to get an active reference to
 *
 *	Get an active reference of @sd.  This function is noop if @sd
 *	is NULL.
 *
 *	RETURNS:
 *	Pointer to @sd on success, NULL on failure.
 */
struct sysfs_dirent *sysfs_get_active(struct sysfs_dirent *sd)
{
175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191
	if (unlikely(!sd))
		return NULL;

	while (1) {
		int v, t;

		v = atomic_read(&sd->s_active);
		if (unlikely(v < 0))
			return NULL;

		t = atomic_cmpxchg(&sd->s_active, v, v + 1);
		if (likely(t == v))
			return sd;
		if (t < 0)
			return NULL;

		cpu_relax();
192 193 194 195 196 197 198 199 200 201 202 203
	}
}

/**
 *	sysfs_put_active - put an active reference to sysfs_dirent
 *	@sd: sysfs_dirent to put an active reference to
 *
 *	Put an active reference to @sd.  This function is noop if @sd
 *	is NULL.
 */
void sysfs_put_active(struct sysfs_dirent *sd)
{
204 205 206 207 208 209 210 211 212 213 214
	struct completion *cmpl;
	int v;

	if (unlikely(!sd))
		return;

	v = atomic_dec_return(&sd->s_active);
	if (likely(v != SD_DEACTIVATED_BIAS))
		return;

	/* atomic_dec_return() is a mb(), we'll always see the updated
215
	 * sd->s_sibling.
216
	 */
217
	cmpl = (void *)sd->s_sibling;
218
	complete(cmpl);
219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263
}

/**
 *	sysfs_get_active_two - get active references to sysfs_dirent and parent
 *	@sd: sysfs_dirent of interest
 *
 *	Get active reference to @sd and its parent.  Parent's active
 *	reference is grabbed first.  This function is noop if @sd is
 *	NULL.
 *
 *	RETURNS:
 *	Pointer to @sd on success, NULL on failure.
 */
struct sysfs_dirent *sysfs_get_active_two(struct sysfs_dirent *sd)
{
	if (sd) {
		if (sd->s_parent && unlikely(!sysfs_get_active(sd->s_parent)))
			return NULL;
		if (unlikely(!sysfs_get_active(sd))) {
			sysfs_put_active(sd->s_parent);
			return NULL;
		}
	}
	return sd;
}

/**
 *	sysfs_put_active_two - put active references to sysfs_dirent and parent
 *	@sd: sysfs_dirent of interest
 *
 *	Put active references to @sd and its parent.  This function is
 *	noop if @sd is NULL.
 */
void sysfs_put_active_two(struct sysfs_dirent *sd)
{
	if (sd) {
		sysfs_put_active(sd);
		sysfs_put_active(sd->s_parent);
	}
}

/**
 *	sysfs_deactivate - deactivate sysfs_dirent
 *	@sd: sysfs_dirent to deactivate
 *
264
 *	Deny new active references and drain existing ones.
265
 */
266
static void sysfs_deactivate(struct sysfs_dirent *sd)
267
{
268 269
	DECLARE_COMPLETION_ONSTACK(wait);
	int v;
270

271
	BUG_ON(sd->s_sibling || !(sd->s_flags & SYSFS_FLAG_REMOVED));
272
	sd->s_sibling = (void *)&wait;
273 274

	/* atomic_add_return() is a mb(), put_active() will always see
275
	 * the updated sd->s_sibling.
276
	 */
277 278 279 280 281
	v = atomic_add_return(SD_DEACTIVATED_BIAS, &sd->s_active);

	if (v != SD_DEACTIVATED_BIAS)
		wait_for_completion(&wait);

282
	sd->s_sibling = NULL;
283 284
}

T
Tejun Heo 已提交
285
static int sysfs_alloc_ino(ino_t *pino)
286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310
{
	int ino, rc;

 retry:
	spin_lock(&sysfs_ino_lock);
	rc = ida_get_new_above(&sysfs_ino_ida, 2, &ino);
	spin_unlock(&sysfs_ino_lock);

	if (rc == -EAGAIN) {
		if (ida_pre_get(&sysfs_ino_ida, GFP_KERNEL))
			goto retry;
		rc = -ENOMEM;
	}

	*pino = ino;
	return rc;
}

static void sysfs_free_ino(ino_t ino)
{
	spin_lock(&sysfs_ino_lock);
	ida_remove(&sysfs_ino_ida, ino);
	spin_unlock(&sysfs_ino_lock);
}

311 312
void release_sysfs_dirent(struct sysfs_dirent * sd)
{
T
Tejun Heo 已提交
313 314 315
	struct sysfs_dirent *parent_sd;

 repeat:
316 317 318
	/* Moving/renaming is always done while holding reference.
	 * sd->s_parent won't change beneath us.
	 */
T
Tejun Heo 已提交
319 320
	parent_sd = sd->s_parent;

321
	if (sysfs_type(sd) == SYSFS_KOBJ_LINK)
322
		sysfs_put(sd->s_elem.symlink.target_sd);
323
	if (sysfs_type(sd) & SYSFS_COPY_NAME)
T
Tejun Heo 已提交
324
		kfree(sd->s_name);
325
	kfree(sd->s_iattr);
326
	sysfs_free_ino(sd->s_ino);
327
	kmem_cache_free(sysfs_dir_cachep, sd);
T
Tejun Heo 已提交
328 329 330 331

	sd = parent_sd;
	if (sd && atomic_dec_and_test(&sd->s_count))
		goto repeat;
332 333
}

L
Linus Torvalds 已提交
334 335 336 337 338
static void sysfs_d_iput(struct dentry * dentry, struct inode * inode)
{
	struct sysfs_dirent * sd = dentry->d_fsdata;

	if (sd) {
T
Tejun Heo 已提交
339 340
		/* sd->s_dentry is protected with sysfs_assoc_lock.
		 * This allows sysfs_drop_dentry() to dereference it.
341
		 */
T
Tejun Heo 已提交
342
		spin_lock(&sysfs_assoc_lock);
343 344 345 346 347 348 349 350

		/* The dentry might have been deleted or another
		 * lookup could have happened updating sd->s_dentry to
		 * point the new dentry.  Ignore if it isn't pointing
		 * to this dentry.
		 */
		if (sd->s_dentry == dentry)
			sd->s_dentry = NULL;
T
Tejun Heo 已提交
351
		spin_unlock(&sysfs_assoc_lock);
L
Linus Torvalds 已提交
352 353 354 355 356 357 358 359 360
		sysfs_put(sd);
	}
	iput(inode);
}

static struct dentry_operations sysfs_dentry_ops = {
	.d_iput		= sysfs_d_iput,
};

361
struct sysfs_dirent *sysfs_new_dirent(const char *name, umode_t mode, int type)
L
Linus Torvalds 已提交
362
{
T
Tejun Heo 已提交
363 364 365 366 367 368 369 370
	char *dup_name = NULL;
	struct sysfs_dirent *sd = NULL;

	if (type & SYSFS_COPY_NAME) {
		name = dup_name = kstrdup(name, GFP_KERNEL);
		if (!name)
			goto err_out;
	}
L
Linus Torvalds 已提交
371

372
	sd = kmem_cache_zalloc(sysfs_dir_cachep, GFP_KERNEL);
L
Linus Torvalds 已提交
373
	if (!sd)
T
Tejun Heo 已提交
374
		goto err_out;
L
Linus Torvalds 已提交
375

T
Tejun Heo 已提交
376 377
	if (sysfs_alloc_ino(&sd->s_ino))
		goto err_out;
378

L
Linus Torvalds 已提交
379
	atomic_set(&sd->s_count, 1);
380
	atomic_set(&sd->s_active, 0);
381
	atomic_set(&sd->s_event, 1);
382

T
Tejun Heo 已提交
383
	sd->s_name = name;
384
	sd->s_mode = mode;
385
	sd->s_flags = type;
L
Linus Torvalds 已提交
386 387

	return sd;
T
Tejun Heo 已提交
388 389 390 391 392

 err_out:
	kfree(dup_name);
	kmem_cache_free(sysfs_dir_cachep, sd);
	return NULL;
L
Linus Torvalds 已提交
393 394
}

395 396 397 398 399 400 401 402 403 404 405
/**
 *	sysfs_attach_dentry - associate sysfs_dirent with dentry
 *	@sd: target sysfs_dirent
 *	@dentry: dentry to associate
 *
 *	Associate @sd with @dentry.  This is protected by
 *	sysfs_assoc_lock to avoid race with sysfs_d_iput().
 *
 *	LOCKING:
 *	mutex_lock(sysfs_mutex)
 */
406 407 408 409 410 411
static void sysfs_attach_dentry(struct sysfs_dirent *sd, struct dentry *dentry)
{
	dentry->d_op = &sysfs_dentry_ops;
	dentry->d_fsdata = sysfs_get(sd);

	/* protect sd->s_dentry against sysfs_d_iput */
T
Tejun Heo 已提交
412
	spin_lock(&sysfs_assoc_lock);
413
	sd->s_dentry = dentry;
T
Tejun Heo 已提交
414
	spin_unlock(&sysfs_assoc_lock);
415 416 417 418

	d_rehash(dentry);
}

419 420 421 422 423 424
static int sysfs_ilookup_test(struct inode *inode, void *arg)
{
	struct sysfs_dirent *sd = arg;
	return inode->i_ino == sd->s_ino;
}

425
/**
426 427 428
 *	sysfs_addrm_start - prepare for sysfs_dirent add/remove
 *	@acxt: pointer to sysfs_addrm_cxt to be used
 *	@parent_sd: parent sysfs_dirent
429
 *
430 431 432 433 434
 *	This function is called when the caller is about to add or
 *	remove sysfs_dirent under @parent_sd.  This function acquires
 *	sysfs_mutex, grabs inode for @parent_sd if available and lock
 *	i_mutex of it.  @acxt is used to keep and pass context to
 *	other addrm functions.
435 436
 *
 *	LOCKING:
437 438 439
 *	Kernel thread context (may sleep).  sysfs_mutex is locked on
 *	return.  i_mutex of parent inode is locked on return if
 *	available.
440
 */
441 442
void sysfs_addrm_start(struct sysfs_addrm_cxt *acxt,
		       struct sysfs_dirent *parent_sd)
443
{
444
	struct inode *inode;
445

446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534
	memset(acxt, 0, sizeof(*acxt));
	acxt->parent_sd = parent_sd;

	/* Lookup parent inode.  inode initialization and I_NEW
	 * clearing are protected by sysfs_mutex.  By grabbing it and
	 * looking up with _nowait variant, inode state can be
	 * determined reliably.
	 */
	mutex_lock(&sysfs_mutex);

	inode = ilookup5_nowait(sysfs_sb, parent_sd->s_ino, sysfs_ilookup_test,
				parent_sd);

	if (inode && !(inode->i_state & I_NEW)) {
		/* parent inode available */
		acxt->parent_inode = inode;

		/* sysfs_mutex is below i_mutex in lock hierarchy.
		 * First, trylock i_mutex.  If fails, unlock
		 * sysfs_mutex and lock them in order.
		 */
		if (!mutex_trylock(&inode->i_mutex)) {
			mutex_unlock(&sysfs_mutex);
			mutex_lock(&inode->i_mutex);
			mutex_lock(&sysfs_mutex);
		}
	} else
		iput(inode);
}

/**
 *	sysfs_add_one - add sysfs_dirent to parent
 *	@acxt: addrm context to use
 *	@sd: sysfs_dirent to be added
 *
 *	Get @acxt->parent_sd and set sd->s_parent to it and increment
 *	nlink of parent inode if @sd is a directory.  @sd is NOT
 *	linked into the children list of the parent.  The caller
 *	should invoke sysfs_link_sibling() after this function
 *	completes if @sd needs to be on the children list.
 *
 *	This function should be called between calls to
 *	sysfs_addrm_start() and sysfs_addrm_finish() and should be
 *	passed the same @acxt as passed to sysfs_addrm_start().
 *
 *	LOCKING:
 *	Determined by sysfs_addrm_start().
 */
void sysfs_add_one(struct sysfs_addrm_cxt *acxt, struct sysfs_dirent *sd)
{
	sd->s_parent = sysfs_get(acxt->parent_sd);

	if (sysfs_type(sd) == SYSFS_DIR && acxt->parent_inode)
		inc_nlink(acxt->parent_inode);

	acxt->cnt++;
}

/**
 *	sysfs_remove_one - remove sysfs_dirent from parent
 *	@acxt: addrm context to use
 *	@sd: sysfs_dirent to be added
 *
 *	Mark @sd removed and drop nlink of parent inode if @sd is a
 *	directory.  @sd is NOT unlinked from the children list of the
 *	parent.  The caller is repsonsible for removing @sd from the
 *	children list before calling this function.
 *
 *	This function should be called between calls to
 *	sysfs_addrm_start() and sysfs_addrm_finish() and should be
 *	passed the same @acxt as passed to sysfs_addrm_start().
 *
 *	LOCKING:
 *	Determined by sysfs_addrm_start().
 */
void sysfs_remove_one(struct sysfs_addrm_cxt *acxt, struct sysfs_dirent *sd)
{
	BUG_ON(sd->s_sibling || (sd->s_flags & SYSFS_FLAG_REMOVED));

	sd->s_flags |= SYSFS_FLAG_REMOVED;
	sd->s_sibling = acxt->removed;
	acxt->removed = sd;

	if (sysfs_type(sd) == SYSFS_DIR && acxt->parent_inode)
		drop_nlink(acxt->parent_inode);

	acxt->cnt++;
}

535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590
/**
 *	sysfs_drop_dentry - drop dentry for the specified sysfs_dirent
 *	@sd: target sysfs_dirent
 *
 *	Drop dentry for @sd.  @sd must have been unlinked from its
 *	parent on entry to this function such that it can't be looked
 *	up anymore.
 *
 *	@sd->s_dentry which is protected with sysfs_assoc_lock points
 *	to the currently associated dentry but we're not holding a
 *	reference to it and racing with dput().  Grab dcache_lock and
 *	verify dentry before dropping it.  If @sd->s_dentry is NULL or
 *	dput() beats us, no need to bother.
 */
static void sysfs_drop_dentry(struct sysfs_dirent *sd)
{
	struct dentry *dentry = NULL;
	struct inode *inode;

	/* We're not holding a reference to ->s_dentry dentry but the
	 * field will stay valid as long as sysfs_assoc_lock is held.
	 */
	spin_lock(&sysfs_assoc_lock);
	spin_lock(&dcache_lock);

	/* drop dentry if it's there and dput() didn't kill it yet */
	if (sd->s_dentry && sd->s_dentry->d_inode) {
		dentry = dget_locked(sd->s_dentry);
		spin_lock(&dentry->d_lock);
		__d_drop(dentry);
		spin_unlock(&dentry->d_lock);
	}

	spin_unlock(&dcache_lock);
	spin_unlock(&sysfs_assoc_lock);

	dput(dentry);
	/* XXX: unpin if directory, this will go away soon */
	if (sysfs_type(sd) == SYSFS_DIR)
		dput(dentry);

	/* adjust nlink and update timestamp */
	inode = ilookup(sysfs_sb, sd->s_ino);
	if (inode) {
		mutex_lock(&inode->i_mutex);

		inode->i_ctime = CURRENT_TIME;
		drop_nlink(inode);
		if (sysfs_type(sd) == SYSFS_DIR)
			drop_nlink(inode);

		mutex_unlock(&inode->i_mutex);
		iput(inode);
	}
}

591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629
/**
 *	sysfs_addrm_finish - finish up sysfs_dirent add/remove
 *	@acxt: addrm context to finish up
 *
 *	Finish up sysfs_dirent add/remove.  Resources acquired by
 *	sysfs_addrm_start() are released and removed sysfs_dirents are
 *	cleaned up.  Timestamps on the parent inode are updated.
 *
 *	LOCKING:
 *	All mutexes acquired by sysfs_addrm_start() are released.
 *
 *	RETURNS:
 *	Number of added/removed sysfs_dirents since sysfs_addrm_start().
 */
int sysfs_addrm_finish(struct sysfs_addrm_cxt *acxt)
{
	/* release resources acquired by sysfs_addrm_start() */
	mutex_unlock(&sysfs_mutex);
	if (acxt->parent_inode) {
		struct inode *inode = acxt->parent_inode;

		/* if added/removed, update timestamps on the parent */
		if (acxt->cnt)
			inode->i_ctime = inode->i_mtime = CURRENT_TIME;

		mutex_unlock(&inode->i_mutex);
		iput(inode);
	}

	/* kill removed sysfs_dirents */
	while (acxt->removed) {
		struct sysfs_dirent *sd = acxt->removed;

		acxt->removed = sd->s_sibling;
		sd->s_sibling = NULL;

		sysfs_drop_dentry(sd);
		sysfs_deactivate(sd);
		sysfs_put(sd);
T
Tejun Heo 已提交
630
	}
631 632

	return acxt->cnt;
633 634
}

635 636 637 638 639 640
/**
 *	sysfs_find_dirent - find sysfs_dirent with the given name
 *	@parent_sd: sysfs_dirent to search under
 *	@name: name to look for
 *
 *	Look for sysfs_dirent with name @name under @parent_sd.
641
 *
642
 *	LOCKING:
643
 *	mutex_lock(sysfs_mutex)
644
 *
645 646
 *	RETURNS:
 *	Pointer to sysfs_dirent if found, NULL if not.
647
 */
648 649
struct sysfs_dirent *sysfs_find_dirent(struct sysfs_dirent *parent_sd,
				       const unsigned char *name)
650
{
651 652 653 654 655 656 657
	struct sysfs_dirent *sd;

	for (sd = parent_sd->s_children; sd; sd = sd->s_sibling)
		if (sysfs_type(sd) && !strcmp(sd->s_name, name))
			return sd;
	return NULL;
}
658

659 660 661 662 663 664 665 666 667
/**
 *	sysfs_get_dirent - find and get sysfs_dirent with the given name
 *	@parent_sd: sysfs_dirent to search under
 *	@name: name to look for
 *
 *	Look for sysfs_dirent with name @name under @parent_sd and get
 *	it if found.
 *
 *	LOCKING:
668
 *	Kernel thread context (may sleep).  Grabs sysfs_mutex.
669 670 671 672 673 674 675 676 677
 *
 *	RETURNS:
 *	Pointer to sysfs_dirent if found, NULL if not.
 */
struct sysfs_dirent *sysfs_get_dirent(struct sysfs_dirent *parent_sd,
				      const unsigned char *name)
{
	struct sysfs_dirent *sd;

678
	mutex_lock(&sysfs_mutex);
679 680
	sd = sysfs_find_dirent(parent_sd, name);
	sysfs_get(sd);
681
	mutex_unlock(&sysfs_mutex);
682 683

	return sd;
684 685
}

686 687
static int create_dir(struct kobject *kobj, struct sysfs_dirent *parent_sd,
		      const char *name, struct sysfs_dirent **p_sd)
L
Linus Torvalds 已提交
688
{
689
	struct dentry *parent = parent_sd->s_dentry;
690
	struct sysfs_addrm_cxt acxt;
L
Linus Torvalds 已提交
691 692
	int error;
	umode_t mode = S_IFDIR| S_IRWXU | S_IRUGO | S_IXUGO;
693
	struct dentry *dentry;
694
	struct inode *inode;
695
	struct sysfs_dirent *sd;
L
Linus Torvalds 已提交
696

697
	sysfs_addrm_start(&acxt, parent_sd);
698

699
	/* allocate */
700 701 702
	dentry = lookup_one_len(name, parent, strlen(name));
	if (IS_ERR(dentry)) {
		error = PTR_ERR(dentry);
703
		goto out_finish;
704 705 706
	}

	error = -EEXIST;
707
	if (dentry->d_inode)
708 709
		goto out_dput;

710
	error = -ENOMEM;
711
	sd = sysfs_new_dirent(name, mode, SYSFS_DIR);
712
	if (!sd)
713
		goto out_drop;
714
	sd->s_elem.dir.kobj = kobj;
715

716
	inode = sysfs_get_inode(sd);
717
	if (!inode)
718 719
		goto out_sput;

720 721 722 723 724 725
	if (inode->i_state & I_NEW) {
		inode->i_op = &sysfs_dir_inode_operations;
		inode->i_fop = &sysfs_dir_operations;
		/* directory inodes start off with i_nlink == 2 (for ".") */
		inc_nlink(inode);
	}
726 727 728

	/* link in */
	error = -EEXIST;
729
	if (sysfs_find_dirent(parent_sd, name))
730 731
		goto out_iput;

732 733
	sysfs_add_one(&acxt, sd);
	sysfs_link_sibling(sd);
734
	sysfs_instantiate(dentry, inode);
735
	sysfs_attach_dentry(sd, dentry);
736

737
	*p_sd = sd;
738
	error = 0;
739
	goto out_finish;	/* pin directory dentry in core */
740

741 742
 out_iput:
	iput(inode);
743 744 745 746 747 748
 out_sput:
	sysfs_put(sd);
 out_drop:
	d_drop(dentry);
 out_dput:
	dput(dentry);
749 750
 out_finish:
	sysfs_addrm_finish(&acxt);
L
Linus Torvalds 已提交
751 752 753
	return error;
}

754 755
int sysfs_create_subdir(struct kobject *kobj, const char *name,
			struct sysfs_dirent **p_sd)
L
Linus Torvalds 已提交
756
{
757
	return create_dir(kobj, kobj->sd, name, p_sd);
L
Linus Torvalds 已提交
758 759 760 761 762
}

/**
 *	sysfs_create_dir - create a directory for an object.
 *	@kobj:		object we're creating directory for. 
763
 *	@shadow_parent:	parent object.
L
Linus Torvalds 已提交
764
 */
765 766
int sysfs_create_dir(struct kobject *kobj,
		     struct sysfs_dirent *shadow_parent_sd)
L
Linus Torvalds 已提交
767
{
768
	struct sysfs_dirent *parent_sd, *sd;
L
Linus Torvalds 已提交
769 770 771 772
	int error = 0;

	BUG_ON(!kobj);

773 774
	if (shadow_parent_sd)
		parent_sd = shadow_parent_sd;
775
	else if (kobj->parent)
776
		parent_sd = kobj->parent->sd;
L
Linus Torvalds 已提交
777
	else if (sysfs_mount && sysfs_mount->mnt_sb)
778
		parent_sd = sysfs_mount->mnt_sb->s_root->d_fsdata;
L
Linus Torvalds 已提交
779 780 781
	else
		return -EFAULT;

782
	error = create_dir(kobj, parent_sd, kobject_name(kobj), &sd);
L
Linus Torvalds 已提交
783
	if (!error)
784
		kobj->sd = sd;
L
Linus Torvalds 已提交
785 786 787 788 789 790 791 792
	return error;
}

static struct dentry * sysfs_lookup(struct inode *dir, struct dentry *dentry,
				struct nameidata *nd)
{
	struct sysfs_dirent * parent_sd = dentry->d_parent->d_fsdata;
	struct sysfs_dirent * sd;
793
	struct bin_attribute *bin_attr;
794 795
	struct inode *inode;
	int found = 0;
L
Linus Torvalds 已提交
796

797
	for (sd = parent_sd->s_children; sd; sd = sd->s_sibling) {
798
		if ((sysfs_type(sd) & SYSFS_NOT_PINNED) &&
799 800
		    !strcmp(sd->s_name, dentry->d_name.name)) {
			found = 1;
L
Linus Torvalds 已提交
801 802 803 804
			break;
		}
	}

805 806 807 808 809
	/* no such entry */
	if (!found)
		return NULL;

	/* attach dentry and inode */
810
	inode = sysfs_get_inode(sd);
811 812 813
	if (!inode)
		return ERR_PTR(-ENOMEM);

814 815
	mutex_lock(&sysfs_mutex);

816 817
	if (inode->i_state & I_NEW) {
		/* initialize inode according to type */
818 819
		switch (sysfs_type(sd)) {
		case SYSFS_KOBJ_ATTR:
820 821
			inode->i_size = PAGE_SIZE;
			inode->i_fop = &sysfs_file_operations;
822 823 824
			break;
		case SYSFS_KOBJ_BIN_ATTR:
			bin_attr = sd->s_elem.bin_attr.bin_attr;
825 826
			inode->i_size = bin_attr->size;
			inode->i_fop = &bin_fops;
827 828
			break;
		case SYSFS_KOBJ_LINK:
829
			inode->i_op = &sysfs_symlink_inode_operations;
830 831 832 833
			break;
		default:
			BUG();
		}
834
	}
835 836 837 838

	sysfs_instantiate(dentry, inode);
	sysfs_attach_dentry(sd, dentry);

839 840
	mutex_unlock(&sysfs_mutex);

841
	return NULL;
L
Linus Torvalds 已提交
842 843
}

844
const struct inode_operations sysfs_dir_inode_operations = {
L
Linus Torvalds 已提交
845
	.lookup		= sysfs_lookup,
846
	.setattr	= sysfs_setattr,
L
Linus Torvalds 已提交
847 848
};

849
static void remove_dir(struct sysfs_dirent *sd)
L
Linus Torvalds 已提交
850
{
851
	struct sysfs_addrm_cxt acxt;
L
Linus Torvalds 已提交
852

853 854 855 856
	sysfs_addrm_start(&acxt, sd->s_parent);
	sysfs_unlink_sibling(sd);
	sysfs_remove_one(&acxt, sd);
	sysfs_addrm_finish(&acxt);
L
Linus Torvalds 已提交
857 858
}

859
void sysfs_remove_subdir(struct sysfs_dirent *sd)
L
Linus Torvalds 已提交
860
{
861
	remove_dir(sd);
L
Linus Torvalds 已提交
862 863 864
}


865
static void __sysfs_remove_dir(struct sysfs_dirent *dir_sd)
L
Linus Torvalds 已提交
866
{
867
	struct sysfs_addrm_cxt acxt;
868
	struct sysfs_dirent **pos;
L
Linus Torvalds 已提交
869

870
	if (!dir_sd)
L
Linus Torvalds 已提交
871 872
		return;

873
	pr_debug("sysfs %s: removing dir\n", dir_sd->s_name);
874
	sysfs_addrm_start(&acxt, dir_sd);
875
	pos = &dir_sd->s_children;
876 877 878
	while (*pos) {
		struct sysfs_dirent *sd = *pos;

879
		if (sysfs_type(sd) && (sysfs_type(sd) & SYSFS_NOT_PINNED)) {
880
			*pos = sd->s_sibling;
881 882
			sd->s_sibling = NULL;
			sysfs_remove_one(&acxt, sd);
883 884
		} else
			pos = &(*pos)->s_sibling;
L
Linus Torvalds 已提交
885
	}
886
	sysfs_addrm_finish(&acxt);
887

888
	remove_dir(dir_sd);
889 890 891 892 893 894 895 896 897 898 899 900 901
}

/**
 *	sysfs_remove_dir - remove an object's directory.
 *	@kobj:	object.
 *
 *	The only thing special about this is that we remove any files in
 *	the directory before we remove the directory, and we've inlined
 *	what used to be sysfs_rmdir() below, instead of calling separately.
 */

void sysfs_remove_dir(struct kobject * kobj)
{
902
	struct sysfs_dirent *sd = kobj->sd;
903

T
Tejun Heo 已提交
904
	spin_lock(&sysfs_assoc_lock);
905
	kobj->sd = NULL;
T
Tejun Heo 已提交
906
	spin_unlock(&sysfs_assoc_lock);
907

908
	__sysfs_remove_dir(sd);
L
Linus Torvalds 已提交
909 910
}

911
int sysfs_rename_dir(struct kobject *kobj, struct sysfs_dirent *new_parent_sd,
912
		     const char *new_name)
L
Linus Torvalds 已提交
913
{
914 915
	struct sysfs_dirent *sd = kobj->sd;
	struct dentry *new_parent = new_parent_sd->s_dentry;
T
Tejun Heo 已提交
916 917
	struct dentry *new_dentry;
	char *dup_name;
918
	int error;
L
Linus Torvalds 已提交
919

920
	if (!new_parent_sd)
921
		return -EFAULT;
L
Linus Torvalds 已提交
922

923
	mutex_lock(&new_parent->d_inode->i_mutex);
L
Linus Torvalds 已提交
924

925
	new_dentry = lookup_one_len(new_name, new_parent, strlen(new_name));
926 927 928
	if (IS_ERR(new_dentry)) {
		error = PTR_ERR(new_dentry);
		goto out_unlock;
L
Linus Torvalds 已提交
929
	}
930 931 932 933 934 935

	/* By allowing two different directories with the same
	 * d_parent we allow this routine to move between different
	 * shadows of the same directory
	 */
	error = -EINVAL;
936
	if (sd->s_parent->s_dentry->d_inode != new_parent->d_inode ||
937
	    new_dentry->d_parent->d_inode != new_parent->d_inode ||
938
	    new_dentry == sd->s_dentry)
939 940 941 942 943 944
		goto out_dput;

	error = -EEXIST;
	if (new_dentry->d_inode)
		goto out_dput;

T
Tejun Heo 已提交
945 946 947 948 949 950
	/* rename kobject and sysfs_dirent */
	error = -ENOMEM;
	new_name = dup_name = kstrdup(new_name, GFP_KERNEL);
	if (!new_name)
		goto out_drop;

951 952
	error = kobject_set_name(kobj, "%s", new_name);
	if (error)
T
Tejun Heo 已提交
953
		goto out_free;
954

T
Tejun Heo 已提交
955 956 957 958
	kfree(sd->s_name);
	sd->s_name = new_name;

	/* move under the new parent */
959
	d_add(new_dentry, NULL);
960
	d_move(sd->s_dentry, new_dentry);
961

962 963
	mutex_lock(&sysfs_mutex);

964
	sysfs_unlink_sibling(sd);
965
	sysfs_get(new_parent_sd);
966
	sysfs_put(sd->s_parent);
967
	sd->s_parent = new_parent_sd;
968
	sysfs_link_sibling(sd);
969

970 971
	mutex_unlock(&sysfs_mutex);

972 973 974
	error = 0;
	goto out_unlock;

T
Tejun Heo 已提交
975 976
 out_free:
	kfree(dup_name);
977 978 979 980 981
 out_drop:
	d_drop(new_dentry);
 out_dput:
	dput(new_dentry);
 out_unlock:
982
	mutex_unlock(&new_parent->d_inode->i_mutex);
L
Linus Torvalds 已提交
983 984 985
	return error;
}

986 987 988 989 990 991 992
int sysfs_move_dir(struct kobject *kobj, struct kobject *new_parent)
{
	struct dentry *old_parent_dentry, *new_parent_dentry, *new_dentry;
	struct sysfs_dirent *new_parent_sd, *sd;
	int error;

	old_parent_dentry = kobj->parent ?
993
		kobj->parent->sd->s_dentry : sysfs_mount->mnt_sb->s_root;
994
	new_parent_dentry = new_parent ?
995
		new_parent->sd->s_dentry : sysfs_mount->mnt_sb->s_root;
996

M
Mark Lord 已提交
997 998
	if (old_parent_dentry->d_inode == new_parent_dentry->d_inode)
		return 0;	/* nothing to move */
999 1000 1001 1002 1003 1004 1005 1006
again:
	mutex_lock(&old_parent_dentry->d_inode->i_mutex);
	if (!mutex_trylock(&new_parent_dentry->d_inode->i_mutex)) {
		mutex_unlock(&old_parent_dentry->d_inode->i_mutex);
		goto again;
	}

	new_parent_sd = new_parent_dentry->d_fsdata;
1007
	sd = kobj->sd;
1008 1009 1010 1011 1012 1013 1014 1015 1016

	new_dentry = lookup_one_len(kobj->name, new_parent_dentry,
				    strlen(kobj->name));
	if (IS_ERR(new_dentry)) {
		error = PTR_ERR(new_dentry);
		goto out;
	} else
		error = 0;
	d_add(new_dentry, NULL);
1017
	d_move(sd->s_dentry, new_dentry);
1018 1019 1020
	dput(new_dentry);

	/* Remove from old parent's list and insert into new parent's list. */
1021 1022
	mutex_lock(&sysfs_mutex);

1023
	sysfs_unlink_sibling(sd);
1024 1025 1026
	sysfs_get(new_parent_sd);
	sysfs_put(sd->s_parent);
	sd->s_parent = new_parent_sd;
1027
	sysfs_link_sibling(sd);
1028

1029
	mutex_unlock(&sysfs_mutex);
1030 1031 1032 1033 1034 1035 1036
out:
	mutex_unlock(&new_parent_dentry->d_inode->i_mutex);
	mutex_unlock(&old_parent_dentry->d_inode->i_mutex);

	return error;
}

L
Linus Torvalds 已提交
1037 1038
static int sysfs_dir_open(struct inode *inode, struct file *file)
{
1039
	struct dentry * dentry = file->f_path.dentry;
L
Linus Torvalds 已提交
1040
	struct sysfs_dirent * parent_sd = dentry->d_fsdata;
1041
	struct sysfs_dirent * sd;
L
Linus Torvalds 已提交
1042

1043
	sd = sysfs_new_dirent("_DIR_", 0, 0);
1044 1045
	if (sd) {
		mutex_lock(&sysfs_mutex);
1046 1047
		sd->s_parent = sysfs_get(parent_sd);
		sysfs_link_sibling(sd);
1048 1049
		mutex_unlock(&sysfs_mutex);
	}
L
Linus Torvalds 已提交
1050

1051 1052
	file->private_data = sd;
	return sd ? 0 : -ENOMEM;
L
Linus Torvalds 已提交
1053 1054 1055 1056 1057 1058
}

static int sysfs_dir_close(struct inode *inode, struct file *file)
{
	struct sysfs_dirent * cursor = file->private_data;

1059
	mutex_lock(&sysfs_mutex);
1060
	sysfs_unlink_sibling(cursor);
1061
	mutex_unlock(&sysfs_mutex);
L
Linus Torvalds 已提交
1062 1063 1064 1065 1066 1067 1068 1069 1070 1071 1072 1073 1074 1075

	release_sysfs_dirent(cursor);

	return 0;
}

/* Relationship between s_mode and the DT_xxx types */
static inline unsigned char dt_type(struct sysfs_dirent *sd)
{
	return (sd->s_mode >> 12) & 15;
}

static int sysfs_readdir(struct file * filp, void * dirent, filldir_t filldir)
{
1076
	struct dentry *dentry = filp->f_path.dentry;
L
Linus Torvalds 已提交
1077 1078
	struct sysfs_dirent * parent_sd = dentry->d_fsdata;
	struct sysfs_dirent *cursor = filp->private_data;
1079
	struct sysfs_dirent **pos;
L
Linus Torvalds 已提交
1080 1081 1082 1083 1084
	ino_t ino;
	int i = filp->f_pos;

	switch (i) {
		case 0:
1085
			ino = parent_sd->s_ino;
L
Linus Torvalds 已提交
1086 1087 1088 1089 1090 1091
			if (filldir(dirent, ".", 1, i, ino, DT_DIR) < 0)
				break;
			filp->f_pos++;
			i++;
			/* fallthrough */
		case 1:
T
Tejun Heo 已提交
1092 1093 1094 1095
			if (parent_sd->s_parent)
				ino = parent_sd->s_parent->s_ino;
			else
				ino = parent_sd->s_ino;
L
Linus Torvalds 已提交
1096 1097 1098 1099 1100 1101
			if (filldir(dirent, "..", 2, i, ino, DT_DIR) < 0)
				break;
			filp->f_pos++;
			i++;
			/* fallthrough */
		default:
1102 1103
			mutex_lock(&sysfs_mutex);

1104 1105 1106 1107 1108 1109 1110
			pos = &parent_sd->s_children;
			while (*pos != cursor)
				pos = &(*pos)->s_sibling;

			/* unlink cursor */
			*pos = cursor->s_sibling;

A
Akinobu Mita 已提交
1111
			if (filp->f_pos == 2)
1112
				pos = &parent_sd->s_children;
A
Akinobu Mita 已提交
1113

1114 1115
			for ( ; *pos; pos = &(*pos)->s_sibling) {
				struct sysfs_dirent *next = *pos;
L
Linus Torvalds 已提交
1116 1117 1118
				const char * name;
				int len;

1119
				if (!sysfs_type(next))
L
Linus Torvalds 已提交
1120 1121
					continue;

T
Tejun Heo 已提交
1122
				name = next->s_name;
L
Linus Torvalds 已提交
1123
				len = strlen(name);
1124
				ino = next->s_ino;
L
Linus Torvalds 已提交
1125 1126 1127

				if (filldir(dirent, name, len, filp->f_pos, ino,
						 dt_type(next)) < 0)
1128
					break;
L
Linus Torvalds 已提交
1129 1130 1131

				filp->f_pos++;
			}
1132 1133 1134 1135

			/* put cursor back in */
			cursor->s_sibling = *pos;
			*pos = cursor;
1136 1137

			mutex_unlock(&sysfs_mutex);
L
Linus Torvalds 已提交
1138 1139 1140 1141 1142 1143
	}
	return 0;
}

static loff_t sysfs_dir_lseek(struct file * file, loff_t offset, int origin)
{
1144
	struct dentry * dentry = file->f_path.dentry;
L
Linus Torvalds 已提交
1145 1146 1147 1148 1149 1150 1151 1152 1153 1154 1155

	switch (origin) {
		case 1:
			offset += file->f_pos;
		case 0:
			if (offset >= 0)
				break;
		default:
			return -EINVAL;
	}
	if (offset != file->f_pos) {
1156 1157
		mutex_lock(&sysfs_mutex);

L
Linus Torvalds 已提交
1158 1159 1160 1161
		file->f_pos = offset;
		if (file->f_pos >= 2) {
			struct sysfs_dirent *sd = dentry->d_fsdata;
			struct sysfs_dirent *cursor = file->private_data;
1162
			struct sysfs_dirent **pos;
L
Linus Torvalds 已提交
1163 1164
			loff_t n = file->f_pos - 2;

1165 1166 1167 1168 1169
			sysfs_unlink_sibling(cursor);

			pos = &sd->s_children;
			while (n && *pos) {
				struct sysfs_dirent *next = *pos;
1170
				if (sysfs_type(next))
L
Linus Torvalds 已提交
1171
					n--;
1172
				pos = &(*pos)->s_sibling;
L
Linus Torvalds 已提交
1173
			}
1174 1175 1176

			cursor->s_sibling = *pos;
			*pos = cursor;
L
Linus Torvalds 已提交
1177
		}
1178 1179

		mutex_unlock(&sysfs_mutex);
L
Linus Torvalds 已提交
1180
	}
1181

L
Linus Torvalds 已提交
1182 1183 1184
	return offset;
}

1185 1186 1187 1188 1189 1190 1191 1192 1193 1194 1195 1196

/**
 *	sysfs_make_shadowed_dir - Setup so a directory can be shadowed
 *	@kobj:	object we're creating shadow of.
 */

int sysfs_make_shadowed_dir(struct kobject *kobj,
	void * (*follow_link)(struct dentry *, struct nameidata *))
{
	struct inode *inode;
	struct inode_operations *i_op;

1197
	inode = kobj->sd->s_dentry->d_inode;
1198 1199 1200 1201 1202 1203 1204 1205 1206 1207 1208 1209 1210 1211 1212 1213 1214 1215 1216 1217 1218 1219 1220 1221 1222 1223
	if (inode->i_op != &sysfs_dir_inode_operations)
		return -EINVAL;

	i_op = kmalloc(sizeof(*i_op), GFP_KERNEL);
	if (!i_op)
		return -ENOMEM;

	memcpy(i_op, &sysfs_dir_inode_operations, sizeof(*i_op));
	i_op->follow_link = follow_link;

	/* Locking of inode->i_op?
	 * Since setting i_op is a single word write and they
	 * are atomic we should be ok here.
	 */
	inode->i_op = i_op;
	return 0;
}

/**
 *	sysfs_create_shadow_dir - create a shadow directory for an object.
 *	@kobj:	object we're creating directory for.
 *
 *	sysfs_make_shadowed_dir must already have been called on this
 *	directory.
 */

1224
struct sysfs_dirent *sysfs_create_shadow_dir(struct kobject *kobj)
1225
{
1226
	struct dentry *dir = kobj->sd->s_dentry;
T
Tejun Heo 已提交
1227 1228 1229 1230
	struct inode *inode = dir->d_inode;
	struct dentry *parent = dir->d_parent;
	struct sysfs_dirent *parent_sd = parent->d_fsdata;
	struct dentry *shadow;
1231
	struct sysfs_dirent *sd;
1232
	struct sysfs_addrm_cxt acxt;
1233

1234
	sd = ERR_PTR(-EINVAL);
1235 1236 1237 1238 1239 1240 1241
	if (!sysfs_is_shadowed_inode(inode))
		goto out;

	shadow = d_alloc(parent, &dir->d_name);
	if (!shadow)
		goto nomem;

1242
	sd = sysfs_new_dirent("_SHADOW_", inode->i_mode, SYSFS_DIR);
1243 1244
	if (!sd)
		goto nomem;
1245
	sd->s_elem.dir.kobj = kobj;
1246

1247 1248 1249 1250 1251 1252 1253
	sysfs_addrm_start(&acxt, parent_sd);

	/* add but don't link into children list */
	sysfs_add_one(&acxt, sd);

	/* attach and instantiate dentry */
	sysfs_attach_dentry(sd, shadow);
1254
	d_instantiate(shadow, igrab(inode));
1255 1256 1257
	inc_nlink(inode);	/* tj: synchronization? */

	sysfs_addrm_finish(&acxt);
1258 1259 1260 1261

	dget(shadow);		/* Extra count - pin the dentry in core */

out:
1262
	return sd;
1263 1264
nomem:
	dput(shadow);
1265
	sd = ERR_PTR(-ENOMEM);
1266 1267 1268 1269 1270
	goto out;
}

/**
 *	sysfs_remove_shadow_dir - remove an object's directory.
1271
 *	@shadow_sd: sysfs_dirent of shadow directory
1272 1273 1274 1275 1276 1277
 *
 *	The only thing special about this is that we remove any files in
 *	the directory before we remove the directory, and we've inlined
 *	what used to be sysfs_rmdir() below, instead of calling separately.
 */

1278
void sysfs_remove_shadow_dir(struct sysfs_dirent *shadow_sd)
1279
{
1280
	__sysfs_remove_dir(shadow_sd);
1281 1282
}

1283
const struct file_operations sysfs_dir_operations = {
L
Linus Torvalds 已提交
1284 1285 1286 1287 1288 1289
	.open		= sysfs_dir_open,
	.release	= sysfs_dir_close,
	.llseek		= sysfs_dir_lseek,
	.read		= generic_read_dir,
	.readdir	= sysfs_readdir,
};