fs/xfs/xfs_inode.c

0b61f8a4SDave Chinner// SPDX-License-Identifier: GPL-2.0
1da177e4SLinus Torvalds/*
3e57ecf6SOlaf Weber * Copyright (c) 2000-2006 Silicon Graphics, Inc.
7b718769SNathan Scott * All Rights Reserved.
1da177e4SLinus Torvalds */
f0e28280SJeff Layton#include <linux/iversion.h>
40ebd81dSRobert P. J. Day
1da177e4SLinus Torvalds#include "xfs.h"
a844f451SNathan Scott#include "xfs_fs.h"
70a9883cSDave Chinner#include "xfs_shared.h"
239880efSDave Chinner#include "xfs_format.h"
239880efSDave Chinner#include "xfs_log_format.h"
239880efSDave Chinner#include "xfs_trans_resv.h"
1da177e4SLinus Torvalds#include "xfs_mount.h"
3ab78df2SDarrick J. Wong#include "xfs_defer.h"
a4fbe6abSDave Chinner#include "xfs_inode.h"
c24b5dfaSDave Chinner#include "xfs_dir2.h"
c24b5dfaSDave Chinner#include "xfs_attr.h"
239880efSDave Chinner#include "xfs_trans_space.h"
239880efSDave Chinner#include "xfs_trans.h"
1da177e4SLinus Torvalds#include "xfs_buf_item.h"
a844f451SNathan Scott#include "xfs_inode_item.h"
a844f451SNathan Scott#include "xfs_ialloc.h"
a844f451SNathan Scott#include "xfs_bmap.h"
68988114SDave Chinner#include "xfs_bmap_util.h"
e9e899a2SDarrick J. Wong#include "xfs_errortag.h"
1da177e4SLinus Torvalds#include "xfs_error.h"
1da177e4SLinus Torvalds#include "xfs_quota.h"
2a82b8beSDavid Chinner#include "xfs_filestream.h"
0b1b213fSChristoph Hellwig#include "xfs_trace.h"
33479e05SDave Chinner#include "xfs_icache.h"
c24b5dfaSDave Chinner#include "xfs_symlink.h"
239880efSDave Chinner#include "xfs_trans_priv.h"
239880efSDave Chinner#include "xfs_log.h"
a4fbe6abSDave Chinner#include "xfs_bmap_btree.h"
aa8968f2SDarrick J. Wong#include "xfs_reflink.h"
9bbafc71SDave Chinner#include "xfs_ag.h"
1da177e4SLinus Torvalds
1da177e4SLinus Torvaldskmem_zone_t *xfs_inode_zone;
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
8f04c47aSChristoph Hellwig * Used in xfs_itruncate_extents().  This is the maximum number of extents
1da177e4SLinus Torvalds * freed from a file in a single transaction.
1da177e4SLinus Torvalds */
1da177e4SLinus Torvalds#define	XFS_ITRUNC_MAX_EXTENTS	2
1da177e4SLinus Torvalds
54d7b5c1SDave ChinnerSTATIC int xfs_iunlink(struct xfs_trans *, struct xfs_inode *);
f40aadb2SDave ChinnerSTATIC int xfs_iunlink_remove(struct xfs_trans *tp, struct xfs_perag *pag,
f40aadb2SDave Chinner	struct xfs_inode *);
ab297431SZhi Yong Wu
2a0ec1d9SDave Chinner/*
2a0ec1d9SDave Chinner * helper function to extract extent size hint from inode
2a0ec1d9SDave Chinner */
2a0ec1d9SDave Chinnerxfs_extlen_t
2a0ec1d9SDave Chinnerxfs_get_extsz_hint(
2a0ec1d9SDave Chinner	struct xfs_inode	*ip)
2a0ec1d9SDave Chinner{
bdb2ed2dSChristoph Hellwig	/*
bdb2ed2dSChristoph Hellwig	 * No point in aligning allocations if we need to COW to actually
bdb2ed2dSChristoph Hellwig	 * write to them.
bdb2ed2dSChristoph Hellwig	 */
bdb2ed2dSChristoph Hellwig	if (xfs_is_always_cow_inode(ip))
bdb2ed2dSChristoph Hellwig		return 0;
db07349dSChristoph Hellwig	if ((ip->i_diflags & XFS_DIFLAG_EXTSIZE) && ip->i_extsize)
031474c2SChristoph Hellwig		return ip->i_extsize;
2a0ec1d9SDave Chinner	if (XFS_IS_REALTIME_INODE(ip))
2a0ec1d9SDave Chinner		return ip->i_mount->m_sb.sb_rextsize;
2a0ec1d9SDave Chinner	return 0;
2a0ec1d9SDave Chinner}
2a0ec1d9SDave Chinner
fa96acadSDave Chinner/*
f7ca3522SDarrick J. Wong * Helper function to extract CoW extent size hint from inode.
f7ca3522SDarrick J. Wong * Between the extent size hint and the CoW extent size hint, we
e153aa79SDarrick J. Wong * return the greater of the two.  If the value is zero (automatic),
e153aa79SDarrick J. Wong * use the default size.
f7ca3522SDarrick J. Wong */
f7ca3522SDarrick J. Wongxfs_extlen_t
f7ca3522SDarrick J. Wongxfs_get_cowextsz_hint(
f7ca3522SDarrick J. Wong	struct xfs_inode	*ip)
f7ca3522SDarrick J. Wong{
f7ca3522SDarrick J. Wong	xfs_extlen_t		a, b;
f7ca3522SDarrick J. Wong
f7ca3522SDarrick J. Wong	a = 0;
3e09ab8fSChristoph Hellwig	if (ip->i_diflags2 & XFS_DIFLAG2_COWEXTSIZE)
b33ce57dSChristoph Hellwig		a = ip->i_cowextsize;
f7ca3522SDarrick J. Wong	b = xfs_get_extsz_hint(ip);
f7ca3522SDarrick J. Wong
e153aa79SDarrick J. Wong	a = max(a, b);
e153aa79SDarrick J. Wong	if (a == 0)
e153aa79SDarrick J. Wong		return XFS_DEFAULT_COWEXTSZ_HINT;
f7ca3522SDarrick J. Wong	return a;
f7ca3522SDarrick J. Wong}
f7ca3522SDarrick J. Wong
f7ca3522SDarrick J. Wong/*
efa70be1SChristoph Hellwig * These two are wrapper routines around the xfs_ilock() routine used to
efa70be1SChristoph Hellwig * centralize some grungy code.  They are used in places that wish to lock the
efa70be1SChristoph Hellwig * inode solely for reading the extents.  The reason these places can't just
efa70be1SChristoph Hellwig * call xfs_ilock(ip, XFS_ILOCK_SHARED) is that the inode lock also guards to
efa70be1SChristoph Hellwig * bringing in of the extents from disk for a file in b-tree format.  If the
efa70be1SChristoph Hellwig * inode is in b-tree format, then we need to lock the inode exclusively until
efa70be1SChristoph Hellwig * the extents are read in.  Locking it exclusively all the time would limit
efa70be1SChristoph Hellwig * our parallelism unnecessarily, though.  What we do instead is check to see
efa70be1SChristoph Hellwig * if the extents have been read in yet, and only lock the inode exclusively
efa70be1SChristoph Hellwig * if they have not.
fa96acadSDave Chinner *
efa70be1SChristoph Hellwig * The functions return a value which should be given to the corresponding
01f4f327SChristoph Hellwig * xfs_iunlock() call.
fa96acadSDave Chinner */
fa96acadSDave Chinneruint
309ecac8SChristoph Hellwigxfs_ilock_data_map_shared(
309ecac8SChristoph Hellwig	struct xfs_inode	*ip)
fa96acadSDave Chinner{
309ecac8SChristoph Hellwig	uint			lock_mode = XFS_ILOCK_SHARED;
fa96acadSDave Chinner
b2197a36SChristoph Hellwig	if (xfs_need_iread_extents(&ip->i_df))
fa96acadSDave Chinner		lock_mode = XFS_ILOCK_EXCL;
fa96acadSDave Chinner	xfs_ilock(ip, lock_mode);
fa96acadSDave Chinner	return lock_mode;
fa96acadSDave Chinner}
fa96acadSDave Chinner
efa70be1SChristoph Hellwiguint
efa70be1SChristoph Hellwigxfs_ilock_attr_map_shared(
efa70be1SChristoph Hellwig	struct xfs_inode	*ip)
fa96acadSDave Chinner{
efa70be1SChristoph Hellwig	uint			lock_mode = XFS_ILOCK_SHARED;
efa70be1SChristoph Hellwig
b2197a36SChristoph Hellwig	if (ip->i_afp && xfs_need_iread_extents(ip->i_afp))
efa70be1SChristoph Hellwig		lock_mode = XFS_ILOCK_EXCL;
efa70be1SChristoph Hellwig	xfs_ilock(ip, lock_mode);
efa70be1SChristoph Hellwig	return lock_mode;
fa96acadSDave Chinner}
fa96acadSDave Chinner
fa96acadSDave Chinner/*
65523218SChristoph Hellwig * In addition to i_rwsem in the VFS inode, the xfs inode contains 2
65523218SChristoph Hellwig * multi-reader locks: i_mmap_lock and the i_lock.  This routine allows
65523218SChristoph Hellwig * various combinations of the locks to be obtained.
fa96acadSDave Chinner *
653c60b6SDave Chinner * The 3 locks should always be ordered so that the IO lock is obtained first,
653c60b6SDave Chinner * the mmap lock second and the ilock last in order to prevent deadlock.
fa96acadSDave Chinner *
653c60b6SDave Chinner * Basic locking order:
653c60b6SDave Chinner *
65523218SChristoph Hellwig * i_rwsem -> i_mmap_lock -> page_lock -> i_ilock
653c60b6SDave Chinner *
c1e8d7c6SMichel Lespinasse * mmap_lock locking order:
653c60b6SDave Chinner *
c1e8d7c6SMichel Lespinasse * i_rwsem -> page lock -> mmap_lock
c1e8d7c6SMichel Lespinasse * mmap_lock -> i_mmap_lock -> page_lock
653c60b6SDave Chinner *
c1e8d7c6SMichel Lespinasse * The difference in mmap_lock locking order mean that we cannot hold the
653c60b6SDave Chinner * i_mmap_lock over syscall based read(2)/write(2) based IO. These IO paths can
c1e8d7c6SMichel Lespinasse * fault in pages during copy in/out (for buffered IO) or require the mmap_lock
653c60b6SDave Chinner * in get_user_pages() to map the user pages into the kernel address space for
65523218SChristoph Hellwig * direct IO. Similarly the i_rwsem cannot be taken inside a page fault because
c1e8d7c6SMichel Lespinasse * page faults already hold the mmap_lock.
653c60b6SDave Chinner *
653c60b6SDave Chinner * Hence to serialise fully against both syscall and mmap based IO, we need to
65523218SChristoph Hellwig * take both the i_rwsem and the i_mmap_lock. These locks should *only* be both
653c60b6SDave Chinner * taken in places where we need to invalidate the page cache in a race
653c60b6SDave Chinner * free manner (e.g. truncate, hole punch and other extent manipulation
653c60b6SDave Chinner * functions).
fa96acadSDave Chinner */
fa96acadSDave Chinnervoid
fa96acadSDave Chinnerxfs_ilock(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
fa96acadSDave Chinner	trace_xfs_ilock(ip, lock_flags, _RET_IP_);
fa96acadSDave Chinner
fa96acadSDave Chinner	/*
fa96acadSDave Chinner	 * You can't set both SHARED and EXCL for the same lock,
fa96acadSDave Chinner	 * and only XFS_IOLOCK_SHARED, XFS_IOLOCK_EXCL, XFS_ILOCK_SHARED,
fa96acadSDave Chinner	 * and XFS_ILOCK_EXCL are valid values to set in lock_flags.
fa96acadSDave Chinner	 */
fa96acadSDave Chinner	ASSERT((lock_flags & (XFS_IOLOCK_SHARED | XFS_IOLOCK_EXCL)) !=
fa96acadSDave Chinner	       (XFS_IOLOCK_SHARED | XFS_IOLOCK_EXCL));
653c60b6SDave Chinner	ASSERT((lock_flags & (XFS_MMAPLOCK_SHARED | XFS_MMAPLOCK_EXCL)) !=
653c60b6SDave Chinner	       (XFS_MMAPLOCK_SHARED | XFS_MMAPLOCK_EXCL));
fa96acadSDave Chinner	ASSERT((lock_flags & (XFS_ILOCK_SHARED | XFS_ILOCK_EXCL)) !=
fa96acadSDave Chinner	       (XFS_ILOCK_SHARED | XFS_ILOCK_EXCL));
0952c818SDave Chinner	ASSERT((lock_flags & ~(XFS_LOCK_MASK | XFS_LOCK_SUBCLASS_MASK)) == 0);
fa96acadSDave Chinner
65523218SChristoph Hellwig	if (lock_flags & XFS_IOLOCK_EXCL) {
65523218SChristoph Hellwig		down_write_nested(&VFS_I(ip)->i_rwsem,
65523218SChristoph Hellwig				  XFS_IOLOCK_DEP(lock_flags));
65523218SChristoph Hellwig	} else if (lock_flags & XFS_IOLOCK_SHARED) {
65523218SChristoph Hellwig		down_read_nested(&VFS_I(ip)->i_rwsem,
65523218SChristoph Hellwig				 XFS_IOLOCK_DEP(lock_flags));
65523218SChristoph Hellwig	}
fa96acadSDave Chinner
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL)
653c60b6SDave Chinner		mrupdate_nested(&ip->i_mmaplock, XFS_MMAPLOCK_DEP(lock_flags));
653c60b6SDave Chinner	else if (lock_flags & XFS_MMAPLOCK_SHARED)
653c60b6SDave Chinner		mraccess_nested(&ip->i_mmaplock, XFS_MMAPLOCK_DEP(lock_flags));
653c60b6SDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_ILOCK_EXCL)
fa96acadSDave Chinner		mrupdate_nested(&ip->i_lock, XFS_ILOCK_DEP(lock_flags));
fa96acadSDave Chinner	else if (lock_flags & XFS_ILOCK_SHARED)
fa96acadSDave Chinner		mraccess_nested(&ip->i_lock, XFS_ILOCK_DEP(lock_flags));
fa96acadSDave Chinner}
fa96acadSDave Chinner
fa96acadSDave Chinner/*
fa96acadSDave Chinner * This is just like xfs_ilock(), except that the caller
fa96acadSDave Chinner * is guaranteed not to sleep.  It returns 1 if it gets
fa96acadSDave Chinner * the requested locks and 0 otherwise.  If the IO lock is
fa96acadSDave Chinner * obtained but the inode lock cannot be, then the IO lock
fa96acadSDave Chinner * is dropped before returning.
fa96acadSDave Chinner *
fa96acadSDave Chinner * ip -- the inode being locked
fa96acadSDave Chinner * lock_flags -- this parameter indicates the inode's locks to be
fa96acadSDave Chinner *       to be locked.  See the comment for xfs_ilock() for a list
fa96acadSDave Chinner *	 of valid values.
fa96acadSDave Chinner */
fa96acadSDave Chinnerint
fa96acadSDave Chinnerxfs_ilock_nowait(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
fa96acadSDave Chinner	trace_xfs_ilock_nowait(ip, lock_flags, _RET_IP_);
fa96acadSDave Chinner
fa96acadSDave Chinner	/*
fa96acadSDave Chinner	 * You can't set both SHARED and EXCL for the same lock,
fa96acadSDave Chinner	 * and only XFS_IOLOCK_SHARED, XFS_IOLOCK_EXCL, XFS_ILOCK_SHARED,
fa96acadSDave Chinner	 * and XFS_ILOCK_EXCL are valid values to set in lock_flags.
fa96acadSDave Chinner	 */
fa96acadSDave Chinner	ASSERT((lock_flags & (XFS_IOLOCK_SHARED | XFS_IOLOCK_EXCL)) !=
fa96acadSDave Chinner	       (XFS_IOLOCK_SHARED | XFS_IOLOCK_EXCL));
653c60b6SDave Chinner	ASSERT((lock_flags & (XFS_MMAPLOCK_SHARED | XFS_MMAPLOCK_EXCL)) !=
653c60b6SDave Chinner	       (XFS_MMAPLOCK_SHARED | XFS_MMAPLOCK_EXCL));
fa96acadSDave Chinner	ASSERT((lock_flags & (XFS_ILOCK_SHARED | XFS_ILOCK_EXCL)) !=
fa96acadSDave Chinner	       (XFS_ILOCK_SHARED | XFS_ILOCK_EXCL));
0952c818SDave Chinner	ASSERT((lock_flags & ~(XFS_LOCK_MASK | XFS_LOCK_SUBCLASS_MASK)) == 0);
fa96acadSDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_IOLOCK_EXCL) {
65523218SChristoph Hellwig		if (!down_write_trylock(&VFS_I(ip)->i_rwsem))
fa96acadSDave Chinner			goto out;
fa96acadSDave Chinner	} else if (lock_flags & XFS_IOLOCK_SHARED) {
65523218SChristoph Hellwig		if (!down_read_trylock(&VFS_I(ip)->i_rwsem))
fa96acadSDave Chinner			goto out;
fa96acadSDave Chinner	}
653c60b6SDave Chinner
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL) {
653c60b6SDave Chinner		if (!mrtryupdate(&ip->i_mmaplock))
653c60b6SDave Chinner			goto out_undo_iolock;
653c60b6SDave Chinner	} else if (lock_flags & XFS_MMAPLOCK_SHARED) {
653c60b6SDave Chinner		if (!mrtryaccess(&ip->i_mmaplock))
653c60b6SDave Chinner			goto out_undo_iolock;
653c60b6SDave Chinner	}
653c60b6SDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_ILOCK_EXCL) {
fa96acadSDave Chinner		if (!mrtryupdate(&ip->i_lock))
653c60b6SDave Chinner			goto out_undo_mmaplock;
fa96acadSDave Chinner	} else if (lock_flags & XFS_ILOCK_SHARED) {
fa96acadSDave Chinner		if (!mrtryaccess(&ip->i_lock))
653c60b6SDave Chinner			goto out_undo_mmaplock;
fa96acadSDave Chinner	}
fa96acadSDave Chinner	return 1;
fa96acadSDave Chinner
653c60b6SDave Chinnerout_undo_mmaplock:
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL)
653c60b6SDave Chinner		mrunlock_excl(&ip->i_mmaplock);
653c60b6SDave Chinner	else if (lock_flags & XFS_MMAPLOCK_SHARED)
653c60b6SDave Chinner		mrunlock_shared(&ip->i_mmaplock);
fa96acadSDave Chinnerout_undo_iolock:
fa96acadSDave Chinner	if (lock_flags & XFS_IOLOCK_EXCL)
65523218SChristoph Hellwig		up_write(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner	else if (lock_flags & XFS_IOLOCK_SHARED)
65523218SChristoph Hellwig		up_read(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinnerout:
fa96acadSDave Chinner	return 0;
fa96acadSDave Chinner}
fa96acadSDave Chinner
fa96acadSDave Chinner/*
fa96acadSDave Chinner * xfs_iunlock() is used to drop the inode locks acquired with
fa96acadSDave Chinner * xfs_ilock() and xfs_ilock_nowait().  The caller must pass
fa96acadSDave Chinner * in the flags given to xfs_ilock() or xfs_ilock_nowait() so
fa96acadSDave Chinner * that we know which locks to drop.
fa96acadSDave Chinner *
fa96acadSDave Chinner * ip -- the inode being unlocked
fa96acadSDave Chinner * lock_flags -- this parameter indicates the inode's locks to be
fa96acadSDave Chinner *       to be unlocked.  See the comment for xfs_ilock() for a list
fa96acadSDave Chinner *	 of valid values for this parameter.
fa96acadSDave Chinner *
fa96acadSDave Chinner */
fa96acadSDave Chinnervoid
fa96acadSDave Chinnerxfs_iunlock(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
fa96acadSDave Chinner	/*
fa96acadSDave Chinner	 * You can't set both SHARED and EXCL for the same lock,
fa96acadSDave Chinner	 * and only XFS_IOLOCK_SHARED, XFS_IOLOCK_EXCL, XFS_ILOCK_SHARED,
fa96acadSDave Chinner	 * and XFS_ILOCK_EXCL are valid values to set in lock_flags.
fa96acadSDave Chinner	 */
fa96acadSDave Chinner	ASSERT((lock_flags & (XFS_IOLOCK_SHARED | XFS_IOLOCK_EXCL)) !=
fa96acadSDave Chinner	       (XFS_IOLOCK_SHARED | XFS_IOLOCK_EXCL));
653c60b6SDave Chinner	ASSERT((lock_flags & (XFS_MMAPLOCK_SHARED | XFS_MMAPLOCK_EXCL)) !=
653c60b6SDave Chinner	       (XFS_MMAPLOCK_SHARED | XFS_MMAPLOCK_EXCL));
fa96acadSDave Chinner	ASSERT((lock_flags & (XFS_ILOCK_SHARED | XFS_ILOCK_EXCL)) !=
fa96acadSDave Chinner	       (XFS_ILOCK_SHARED | XFS_ILOCK_EXCL));
0952c818SDave Chinner	ASSERT((lock_flags & ~(XFS_LOCK_MASK | XFS_LOCK_SUBCLASS_MASK)) == 0);
fa96acadSDave Chinner	ASSERT(lock_flags != 0);
fa96acadSDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_IOLOCK_EXCL)
65523218SChristoph Hellwig		up_write(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner	else if (lock_flags & XFS_IOLOCK_SHARED)
65523218SChristoph Hellwig		up_read(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL)
653c60b6SDave Chinner		mrunlock_excl(&ip->i_mmaplock);
653c60b6SDave Chinner	else if (lock_flags & XFS_MMAPLOCK_SHARED)
653c60b6SDave Chinner		mrunlock_shared(&ip->i_mmaplock);
653c60b6SDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_ILOCK_EXCL)
fa96acadSDave Chinner		mrunlock_excl(&ip->i_lock);
fa96acadSDave Chinner	else if (lock_flags & XFS_ILOCK_SHARED)
fa96acadSDave Chinner		mrunlock_shared(&ip->i_lock);
fa96acadSDave Chinner
fa96acadSDave Chinner	trace_xfs_iunlock(ip, lock_flags, _RET_IP_);
fa96acadSDave Chinner}
fa96acadSDave Chinner
fa96acadSDave Chinner/*
fa96acadSDave Chinner * give up write locks.  the i/o lock cannot be held nested
fa96acadSDave Chinner * if it is being demoted.
fa96acadSDave Chinner */
fa96acadSDave Chinnervoid
fa96acadSDave Chinnerxfs_ilock_demote(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
653c60b6SDave Chinner	ASSERT(lock_flags & (XFS_IOLOCK_EXCL|XFS_MMAPLOCK_EXCL|XFS_ILOCK_EXCL));
653c60b6SDave Chinner	ASSERT((lock_flags &
653c60b6SDave Chinner		~(XFS_IOLOCK_EXCL|XFS_MMAPLOCK_EXCL|XFS_ILOCK_EXCL)) == 0);
fa96acadSDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_ILOCK_EXCL)
fa96acadSDave Chinner		mrdemote(&ip->i_lock);
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL)
653c60b6SDave Chinner		mrdemote(&ip->i_mmaplock);
fa96acadSDave Chinner	if (lock_flags & XFS_IOLOCK_EXCL)
65523218SChristoph Hellwig		downgrade_write(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner
fa96acadSDave Chinner	trace_xfs_ilock_demote(ip, lock_flags, _RET_IP_);
fa96acadSDave Chinner}
fa96acadSDave Chinner
742ae1e3SDave Chinner#if defined(DEBUG) || defined(XFS_WARN)
fa96acadSDave Chinnerint
fa96acadSDave Chinnerxfs_isilocked(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
fa96acadSDave Chinner	if (lock_flags & (XFS_ILOCK_EXCL|XFS_ILOCK_SHARED)) {
fa96acadSDave Chinner		if (!(lock_flags & XFS_ILOCK_SHARED))
fa96acadSDave Chinner			return !!ip->i_lock.mr_writer;
fa96acadSDave Chinner		return rwsem_is_locked(&ip->i_lock.mr_lock);
fa96acadSDave Chinner	}
fa96acadSDave Chinner
653c60b6SDave Chinner	if (lock_flags & (XFS_MMAPLOCK_EXCL|XFS_MMAPLOCK_SHARED)) {
653c60b6SDave Chinner		if (!(lock_flags & XFS_MMAPLOCK_SHARED))
653c60b6SDave Chinner			return !!ip->i_mmaplock.mr_writer;
653c60b6SDave Chinner		return rwsem_is_locked(&ip->i_mmaplock.mr_lock);
653c60b6SDave Chinner	}
653c60b6SDave Chinner
fa96acadSDave Chinner	if (lock_flags & (XFS_IOLOCK_EXCL|XFS_IOLOCK_SHARED)) {
fa96acadSDave Chinner		if (!(lock_flags & XFS_IOLOCK_SHARED))
65523218SChristoph Hellwig			return !debug_locks ||
65523218SChristoph Hellwig				lockdep_is_held_type(&VFS_I(ip)->i_rwsem, 0);
65523218SChristoph Hellwig		return rwsem_is_locked(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner	}
fa96acadSDave Chinner
fa96acadSDave Chinner	ASSERT(0);
fa96acadSDave Chinner	return 0;
fa96acadSDave Chinner}
fa96acadSDave Chinner#endif
fa96acadSDave Chinner
b6a9947eSDave Chinner/*
b6a9947eSDave Chinner * xfs_lockdep_subclass_ok() is only used in an ASSERT, so is only called when
b6a9947eSDave Chinner * DEBUG or XFS_WARN is set. And MAX_LOCKDEP_SUBCLASSES is then only defined
b6a9947eSDave Chinner * when CONFIG_LOCKDEP is set. Hence the complex define below to avoid build
b6a9947eSDave Chinner * errors and warnings.
b6a9947eSDave Chinner */
b6a9947eSDave Chinner#if (defined(DEBUG) || defined(XFS_WARN)) && defined(CONFIG_LOCKDEP)
3403ccc0SDave Chinnerstatic bool
3403ccc0SDave Chinnerxfs_lockdep_subclass_ok(
3403ccc0SDave Chinner	int subclass)
3403ccc0SDave Chinner{
3403ccc0SDave Chinner	return subclass < MAX_LOCKDEP_SUBCLASSES;
3403ccc0SDave Chinner}
3403ccc0SDave Chinner#else
3403ccc0SDave Chinner#define xfs_lockdep_subclass_ok(subclass)	(true)
3403ccc0SDave Chinner#endif
3403ccc0SDave Chinner
c24b5dfaSDave Chinner/*
653c60b6SDave Chinner * Bump the subclass so xfs_lock_inodes() acquires each lock with a different
0952c818SDave Chinner * value. This can be called for any type of inode lock combination, including
0952c818SDave Chinner * parent locking. Care must be taken to ensure we don't overrun the subclass
0952c818SDave Chinner * storage fields in the class mask we build.
c24b5dfaSDave Chinner */
c24b5dfaSDave Chinnerstatic inline int
c24b5dfaSDave Chinnerxfs_lock_inumorder(int lock_mode, int subclass)
c24b5dfaSDave Chinner{
0952c818SDave Chinner	int	class = 0;
0952c818SDave Chinner
0952c818SDave Chinner	ASSERT(!(lock_mode & (XFS_ILOCK_PARENT | XFS_ILOCK_RTBITMAP |
0952c818SDave Chinner			      XFS_ILOCK_RTSUM)));
3403ccc0SDave Chinner	ASSERT(xfs_lockdep_subclass_ok(subclass));
0952c818SDave Chinner
653c60b6SDave Chinner	if (lock_mode & (XFS_IOLOCK_SHARED|XFS_IOLOCK_EXCL)) {
0952c818SDave Chinner		ASSERT(subclass <= XFS_IOLOCK_MAX_SUBCLASS);
0952c818SDave Chinner		class += subclass << XFS_IOLOCK_SHIFT;
653c60b6SDave Chinner	}
653c60b6SDave Chinner
653c60b6SDave Chinner	if (lock_mode & (XFS_MMAPLOCK_SHARED|XFS_MMAPLOCK_EXCL)) {
0952c818SDave Chinner		ASSERT(subclass <= XFS_MMAPLOCK_MAX_SUBCLASS);
0952c818SDave Chinner		class += subclass << XFS_MMAPLOCK_SHIFT;
653c60b6SDave Chinner	}
653c60b6SDave Chinner
0952c818SDave Chinner	if (lock_mode & (XFS_ILOCK_SHARED|XFS_ILOCK_EXCL)) {
0952c818SDave Chinner		ASSERT(subclass <= XFS_ILOCK_MAX_SUBCLASS);
0952c818SDave Chinner		class += subclass << XFS_ILOCK_SHIFT;
0952c818SDave Chinner	}
c24b5dfaSDave Chinner
0952c818SDave Chinner	return (lock_mode & ~XFS_LOCK_SUBCLASS_MASK) | class;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner/*
95afcf5cSDave Chinner * The following routine will lock n inodes in exclusive mode.  We assume the
95afcf5cSDave Chinner * caller calls us with the inodes in i_ino order.
c24b5dfaSDave Chinner *
95afcf5cSDave Chinner * We need to detect deadlock where an inode that we lock is in the AIL and we
95afcf5cSDave Chinner * start waiting for another inode that is locked by a thread in a long running
95afcf5cSDave Chinner * transaction (such as truncate). This can result in deadlock since the long
95afcf5cSDave Chinner * running trans might need to wait for the inode we just locked in order to
95afcf5cSDave Chinner * push the tail and free space in the log.
0952c818SDave Chinner *
0952c818SDave Chinner * xfs_lock_inodes() can only be used to lock one type of lock at a time -
0952c818SDave Chinner * the iolock, the mmaplock or the ilock, but not more than one at a time. If we
0952c818SDave Chinner * lock more than one at a time, lockdep will report false positives saying we
0952c818SDave Chinner * have violated locking orders.
c24b5dfaSDave Chinner */
0d5a75e9SEric Sandeenstatic void
c24b5dfaSDave Chinnerxfs_lock_inodes(
efe2330fSChristoph Hellwig	struct xfs_inode	**ips,
c24b5dfaSDave Chinner	int			inodes,
c24b5dfaSDave Chinner	uint			lock_mode)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	int			attempts = 0, i, j, try_lock;
efe2330fSChristoph Hellwig	struct xfs_log_item	*lp;
c24b5dfaSDave Chinner
0952c818SDave Chinner	/*
0952c818SDave Chinner	 * Currently supports between 2 and 5 inodes with exclusive locking.  We
0952c818SDave Chinner	 * support an arbitrary depth of locking here, but absolute limits on
b63da6c8SRandy Dunlap	 * inodes depend on the type of locking and the limits placed by
0952c818SDave Chinner	 * lockdep annotations in xfs_lock_inumorder.  These are all checked by
0952c818SDave Chinner	 * the asserts.
0952c818SDave Chinner	 */
95afcf5cSDave Chinner	ASSERT(ips && inodes >= 2 && inodes <= 5);
0952c818SDave Chinner	ASSERT(lock_mode & (XFS_IOLOCK_EXCL | XFS_MMAPLOCK_EXCL |
0952c818SDave Chinner			    XFS_ILOCK_EXCL));
0952c818SDave Chinner	ASSERT(!(lock_mode & (XFS_IOLOCK_SHARED | XFS_MMAPLOCK_SHARED |
0952c818SDave Chinner			      XFS_ILOCK_SHARED)));
0952c818SDave Chinner	ASSERT(!(lock_mode & XFS_MMAPLOCK_EXCL) ||
0952c818SDave Chinner		inodes <= XFS_MMAPLOCK_MAX_SUBCLASS + 1);
0952c818SDave Chinner	ASSERT(!(lock_mode & XFS_ILOCK_EXCL) ||
0952c818SDave Chinner		inodes <= XFS_ILOCK_MAX_SUBCLASS + 1);
0952c818SDave Chinner
0952c818SDave Chinner	if (lock_mode & XFS_IOLOCK_EXCL) {
0952c818SDave Chinner		ASSERT(!(lock_mode & (XFS_MMAPLOCK_EXCL | XFS_ILOCK_EXCL)));
0952c818SDave Chinner	} else if (lock_mode & XFS_MMAPLOCK_EXCL)
0952c818SDave Chinner		ASSERT(!(lock_mode & XFS_ILOCK_EXCL));
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	try_lock = 0;
c24b5dfaSDave Chinner	i = 0;
c24b5dfaSDave Chinneragain:
c24b5dfaSDave Chinner	for (; i < inodes; i++) {
c24b5dfaSDave Chinner		ASSERT(ips[i]);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		if (i && (ips[i] == ips[i - 1]))	/* Already locked */
c24b5dfaSDave Chinner			continue;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		/*
95afcf5cSDave Chinner		 * If try_lock is not set yet, make sure all locked inodes are
95afcf5cSDave Chinner		 * not in the AIL.  If any are, set try_lock to be used later.
c24b5dfaSDave Chinner		 */
c24b5dfaSDave Chinner		if (!try_lock) {
c24b5dfaSDave Chinner			for (j = (i - 1); j >= 0 && !try_lock; j--) {
b3b14aacSChristoph Hellwig				lp = &ips[j]->i_itemp->ili_item;
22525c17SDave Chinner				if (lp && test_bit(XFS_LI_IN_AIL, &lp->li_flags))
c24b5dfaSDave Chinner					try_lock++;
c24b5dfaSDave Chinner			}
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		/*
c24b5dfaSDave Chinner		 * If any of the previous locks we have locked is in the AIL,
c24b5dfaSDave Chinner		 * we must TRY to get the second and subsequent locks. If
c24b5dfaSDave Chinner		 * we can't get any, we must release all we have
c24b5dfaSDave Chinner		 * and try again.
c24b5dfaSDave Chinner		 */
95afcf5cSDave Chinner		if (!try_lock) {
95afcf5cSDave Chinner			xfs_ilock(ips[i], xfs_lock_inumorder(lock_mode, i));
95afcf5cSDave Chinner			continue;
95afcf5cSDave Chinner		}
c24b5dfaSDave Chinner
95afcf5cSDave Chinner		/* try_lock means we have an inode locked that is in the AIL. */
c24b5dfaSDave Chinner		ASSERT(i != 0);
95afcf5cSDave Chinner		if (xfs_ilock_nowait(ips[i], xfs_lock_inumorder(lock_mode, i)))
95afcf5cSDave Chinner			continue;
95afcf5cSDave Chinner
95afcf5cSDave Chinner		/*
95afcf5cSDave Chinner		 * Unlock all previous guys and try again.  xfs_iunlock will try
95afcf5cSDave Chinner		 * to push the tail if the inode is in the AIL.
95afcf5cSDave Chinner		 */
c24b5dfaSDave Chinner		attempts++;
c24b5dfaSDave Chinner		for (j = i - 1; j >= 0; j--) {
c24b5dfaSDave Chinner			/*
95afcf5cSDave Chinner			 * Check to see if we've already unlocked this one.  Not
95afcf5cSDave Chinner			 * the first one going back, and the inode ptr is the
95afcf5cSDave Chinner			 * same.
c24b5dfaSDave Chinner			 */
95afcf5cSDave Chinner			if (j != (i - 1) && ips[j] == ips[j + 1])
c24b5dfaSDave Chinner				continue;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner			xfs_iunlock(ips[j], lock_mode);
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		if ((attempts % 5) == 0) {
c24b5dfaSDave Chinner			delay(1); /* Don't just spin the CPU */
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner		i = 0;
c24b5dfaSDave Chinner		try_lock = 0;
c24b5dfaSDave Chinner		goto again;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner/*
653c60b6SDave Chinner * xfs_lock_two_inodes() can only be used to lock one type of lock at a time -
7c2d238aSDarrick J. Wong * the mmaplock or the ilock, but not more than one type at a time. If we lock
7c2d238aSDarrick J. Wong * more than one at a time, lockdep will report false positives saying we have
7c2d238aSDarrick J. Wong * violated locking orders.  The iolock must be double-locked separately since
7c2d238aSDarrick J. Wong * we use i_rwsem for that.  We now support taking one lock EXCL and the other
7c2d238aSDarrick J. Wong * SHARED.
c24b5dfaSDave Chinner */
c24b5dfaSDave Chinnervoid
c24b5dfaSDave Chinnerxfs_lock_two_inodes(
7c2d238aSDarrick J. Wong	struct xfs_inode	*ip0,
7c2d238aSDarrick J. Wong	uint			ip0_mode,
7c2d238aSDarrick J. Wong	struct xfs_inode	*ip1,
7c2d238aSDarrick J. Wong	uint			ip1_mode)
c24b5dfaSDave Chinner{
7c2d238aSDarrick J. Wong	struct xfs_inode	*temp;
7c2d238aSDarrick J. Wong	uint			mode_temp;
c24b5dfaSDave Chinner	int			attempts = 0;
efe2330fSChristoph Hellwig	struct xfs_log_item	*lp;
c24b5dfaSDave Chinner
7c2d238aSDarrick J. Wong	ASSERT(hweight32(ip0_mode) == 1);
7c2d238aSDarrick J. Wong	ASSERT(hweight32(ip1_mode) == 1);
7c2d238aSDarrick J. Wong	ASSERT(!(ip0_mode & (XFS_IOLOCK_SHARED|XFS_IOLOCK_EXCL)));
7c2d238aSDarrick J. Wong	ASSERT(!(ip1_mode & (XFS_IOLOCK_SHARED|XFS_IOLOCK_EXCL)));
7c2d238aSDarrick J. Wong	ASSERT(!(ip0_mode & (XFS_MMAPLOCK_SHARED|XFS_MMAPLOCK_EXCL)) ||
7c2d238aSDarrick J. Wong	       !(ip0_mode & (XFS_ILOCK_SHARED|XFS_ILOCK_EXCL)));
7c2d238aSDarrick J. Wong	ASSERT(!(ip1_mode & (XFS_MMAPLOCK_SHARED|XFS_MMAPLOCK_EXCL)) ||
7c2d238aSDarrick J. Wong	       !(ip1_mode & (XFS_ILOCK_SHARED|XFS_ILOCK_EXCL)));
7c2d238aSDarrick J. Wong	ASSERT(!(ip1_mode & (XFS_MMAPLOCK_SHARED|XFS_MMAPLOCK_EXCL)) ||
7c2d238aSDarrick J. Wong	       !(ip0_mode & (XFS_ILOCK_SHARED|XFS_ILOCK_EXCL)));
7c2d238aSDarrick J. Wong	ASSERT(!(ip0_mode & (XFS_MMAPLOCK_SHARED|XFS_MMAPLOCK_EXCL)) ||
7c2d238aSDarrick J. Wong	       !(ip1_mode & (XFS_ILOCK_SHARED|XFS_ILOCK_EXCL)));
653c60b6SDave Chinner
c24b5dfaSDave Chinner	ASSERT(ip0->i_ino != ip1->i_ino);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (ip0->i_ino > ip1->i_ino) {
c24b5dfaSDave Chinner		temp = ip0;
c24b5dfaSDave Chinner		ip0 = ip1;
c24b5dfaSDave Chinner		ip1 = temp;
7c2d238aSDarrick J. Wong		mode_temp = ip0_mode;
7c2d238aSDarrick J. Wong		ip0_mode = ip1_mode;
7c2d238aSDarrick J. Wong		ip1_mode = mode_temp;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner again:
7c2d238aSDarrick J. Wong	xfs_ilock(ip0, xfs_lock_inumorder(ip0_mode, 0));
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If the first lock we have locked is in the AIL, we must TRY to get
c24b5dfaSDave Chinner	 * the second lock. If we can't get it, we must release the first one
c24b5dfaSDave Chinner	 * and try again.
c24b5dfaSDave Chinner	 */
b3b14aacSChristoph Hellwig	lp = &ip0->i_itemp->ili_item;
22525c17SDave Chinner	if (lp && test_bit(XFS_LI_IN_AIL, &lp->li_flags)) {
7c2d238aSDarrick J. Wong		if (!xfs_ilock_nowait(ip1, xfs_lock_inumorder(ip1_mode, 1))) {
7c2d238aSDarrick J. Wong			xfs_iunlock(ip0, ip0_mode);
c24b5dfaSDave Chinner			if ((++attempts % 5) == 0)
c24b5dfaSDave Chinner				delay(1); /* Don't just spin the CPU */
c24b5dfaSDave Chinner			goto again;
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner	} else {
7c2d238aSDarrick J. Wong		xfs_ilock(ip1, xfs_lock_inumorder(ip1_mode, 1));
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
1da177e4SLinus Torvaldsuint
1da177e4SLinus Torvaldsxfs_ip2xflags(
58f88ca2SDave Chinner	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
4422501dSChristoph Hellwig	uint			flags = 0;
1da177e4SLinus Torvalds
4422501dSChristoph Hellwig	if (ip->i_diflags & XFS_DIFLAG_ANY) {
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_REALTIME)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_REALTIME;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_PREALLOC)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_PREALLOC;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_IMMUTABLE)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_IMMUTABLE;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_APPEND)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_APPEND;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_SYNC)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_SYNC;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_NOATIME)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_NOATIME;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_NODUMP)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_NODUMP;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_RTINHERIT)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_RTINHERIT;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_PROJINHERIT)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_PROJINHERIT;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_NOSYMLINKS)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_NOSYMLINKS;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_EXTSIZE)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_EXTSIZE;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_EXTSZINHERIT)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_EXTSZINHERIT;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_NODEFRAG)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_NODEFRAG;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_FILESTREAM)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_FILESTREAM;
4422501dSChristoph Hellwig	}
4422501dSChristoph Hellwig
4422501dSChristoph Hellwig	if (ip->i_diflags2 & XFS_DIFLAG2_ANY) {
4422501dSChristoph Hellwig		if (ip->i_diflags2 & XFS_DIFLAG2_DAX)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_DAX;
4422501dSChristoph Hellwig		if (ip->i_diflags2 & XFS_DIFLAG2_COWEXTSIZE)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_COWEXTSIZE;
4422501dSChristoph Hellwig	}
4422501dSChristoph Hellwig
4422501dSChristoph Hellwig	if (XFS_IFORK_Q(ip))
4422501dSChristoph Hellwig		flags |= FS_XFLAG_HASATTR;
4422501dSChristoph Hellwig	return flags;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
c24b5dfaSDave Chinner * Lookups up an inode from "name". If ci_name is not NULL, then a CI match
c24b5dfaSDave Chinner * is allowed, otherwise it has to be an exact match. If a CI match is found,
c24b5dfaSDave Chinner * ci_name->name will point to a the actual name (caller must free) or
c24b5dfaSDave Chinner * will be set to NULL if an exact match is found.
c24b5dfaSDave Chinner */
c24b5dfaSDave Chinnerint
c24b5dfaSDave Chinnerxfs_lookup(
c24b5dfaSDave Chinner	xfs_inode_t		*dp,
c24b5dfaSDave Chinner	struct xfs_name		*name,
c24b5dfaSDave Chinner	xfs_inode_t		**ipp,
c24b5dfaSDave Chinner	struct xfs_name		*ci_name)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	xfs_ino_t		inum;
c24b5dfaSDave Chinner	int			error;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	trace_xfs_lookup(dp, name);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (XFS_FORCED_SHUTDOWN(dp->i_mount))
2451337dSDave Chinner		return -EIO;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	error = xfs_dir_lookup(NULL, dp, name, &inum, ci_name);
c24b5dfaSDave Chinner	if (error)
dbad7c99SDave Chinner		goto out_unlock;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	error = xfs_iget(dp->i_mount, NULL, inum, 0, 0, ipp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto out_free_name;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinnerout_free_name:
c24b5dfaSDave Chinner	if (ci_name)
c24b5dfaSDave Chinner		kmem_free(ci_name->name);
dbad7c99SDave Chinnerout_unlock:
c24b5dfaSDave Chinner	*ipp = NULL;
c24b5dfaSDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
8a569d71SDarrick J. Wong/* Propagate di_flags from a parent inode to a child inode. */
8a569d71SDarrick J. Wongstatic void
8a569d71SDarrick J. Wongxfs_inode_inherit_flags(
8a569d71SDarrick J. Wong	struct xfs_inode	*ip,
8a569d71SDarrick J. Wong	const struct xfs_inode	*pip)
8a569d71SDarrick J. Wong{
8a569d71SDarrick J. Wong	unsigned int		di_flags = 0;
603f000bSDarrick J. Wong	xfs_failaddr_t		failaddr;
8a569d71SDarrick J. Wong	umode_t			mode = VFS_I(ip)->i_mode;
8a569d71SDarrick J. Wong
8a569d71SDarrick J. Wong	if (S_ISDIR(mode)) {
db07349dSChristoph Hellwig		if (pip->i_diflags & XFS_DIFLAG_RTINHERIT)
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_RTINHERIT;
db07349dSChristoph Hellwig		if (pip->i_diflags & XFS_DIFLAG_EXTSZINHERIT) {
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_EXTSZINHERIT;
031474c2SChristoph Hellwig			ip->i_extsize = pip->i_extsize;
8a569d71SDarrick J. Wong		}
db07349dSChristoph Hellwig		if (pip->i_diflags & XFS_DIFLAG_PROJINHERIT)
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_PROJINHERIT;
8a569d71SDarrick J. Wong	} else if (S_ISREG(mode)) {
db07349dSChristoph Hellwig		if ((pip->i_diflags & XFS_DIFLAG_RTINHERIT) &&
38c26bfdSDave Chinner		    xfs_has_realtime(ip->i_mount))
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_REALTIME;
db07349dSChristoph Hellwig		if (pip->i_diflags & XFS_DIFLAG_EXTSZINHERIT) {
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_EXTSIZE;
031474c2SChristoph Hellwig			ip->i_extsize = pip->i_extsize;
8a569d71SDarrick J. Wong		}
8a569d71SDarrick J. Wong	}
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_NOATIME) &&
8a569d71SDarrick J. Wong	    xfs_inherit_noatime)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_NOATIME;
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_NODUMP) &&
8a569d71SDarrick J. Wong	    xfs_inherit_nodump)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_NODUMP;
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_SYNC) &&
8a569d71SDarrick J. Wong	    xfs_inherit_sync)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_SYNC;
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_NOSYMLINKS) &&
8a569d71SDarrick J. Wong	    xfs_inherit_nosymlinks)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_NOSYMLINKS;
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_NODEFRAG) &&
8a569d71SDarrick J. Wong	    xfs_inherit_nodefrag)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_NODEFRAG;
db07349dSChristoph Hellwig	if (pip->i_diflags & XFS_DIFLAG_FILESTREAM)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_FILESTREAM;
8a569d71SDarrick J. Wong
db07349dSChristoph Hellwig	ip->i_diflags |= di_flags;
603f000bSDarrick J. Wong
603f000bSDarrick J. Wong	/*
603f000bSDarrick J. Wong	 * Inode verifiers on older kernels only check that the extent size
603f000bSDarrick J. Wong	 * hint is an integer multiple of the rt extent size on realtime files.
603f000bSDarrick J. Wong	 * They did not check the hint alignment on a directory with both
603f000bSDarrick J. Wong	 * rtinherit and extszinherit flags set.  If the misaligned hint is
603f000bSDarrick J. Wong	 * propagated from a directory into a new realtime file, new file
603f000bSDarrick J. Wong	 * allocations will fail due to math errors in the rt allocator and/or
603f000bSDarrick J. Wong	 * trip the verifiers.  Validate the hint settings in the new file so
603f000bSDarrick J. Wong	 * that we don't let broken hints propagate.
603f000bSDarrick J. Wong	 */
603f000bSDarrick J. Wong	failaddr = xfs_inode_validate_extsize(ip->i_mount, ip->i_extsize,
603f000bSDarrick J. Wong			VFS_I(ip)->i_mode, ip->i_diflags);
603f000bSDarrick J. Wong	if (failaddr) {
603f000bSDarrick J. Wong		ip->i_diflags &= ~(XFS_DIFLAG_EXTSIZE |
603f000bSDarrick J. Wong				   XFS_DIFLAG_EXTSZINHERIT);
603f000bSDarrick J. Wong		ip->i_extsize = 0;
603f000bSDarrick J. Wong	}
8a569d71SDarrick J. Wong}
8a569d71SDarrick J. Wong
8a569d71SDarrick J. Wong/* Propagate di_flags2 from a parent inode to a child inode. */
8a569d71SDarrick J. Wongstatic void
8a569d71SDarrick J. Wongxfs_inode_inherit_flags2(
8a569d71SDarrick J. Wong	struct xfs_inode	*ip,
8a569d71SDarrick J. Wong	const struct xfs_inode	*pip)
8a569d71SDarrick J. Wong{
603f000bSDarrick J. Wong	xfs_failaddr_t		failaddr;
603f000bSDarrick J. Wong
3e09ab8fSChristoph Hellwig	if (pip->i_diflags2 & XFS_DIFLAG2_COWEXTSIZE) {
3e09ab8fSChristoph Hellwig		ip->i_diflags2 |= XFS_DIFLAG2_COWEXTSIZE;
b33ce57dSChristoph Hellwig		ip->i_cowextsize = pip->i_cowextsize;
8a569d71SDarrick J. Wong	}
3e09ab8fSChristoph Hellwig	if (pip->i_diflags2 & XFS_DIFLAG2_DAX)
3e09ab8fSChristoph Hellwig		ip->i_diflags2 |= XFS_DIFLAG2_DAX;
603f000bSDarrick J. Wong
603f000bSDarrick J. Wong	/* Don't let invalid cowextsize hints propagate. */
603f000bSDarrick J. Wong	failaddr = xfs_inode_validate_cowextsize(ip->i_mount, ip->i_cowextsize,
603f000bSDarrick J. Wong			VFS_I(ip)->i_mode, ip->i_diflags, ip->i_diflags2);
603f000bSDarrick J. Wong	if (failaddr) {
603f000bSDarrick J. Wong		ip->i_diflags2 &= ~XFS_DIFLAG2_COWEXTSIZE;
603f000bSDarrick J. Wong		ip->i_cowextsize = 0;
603f000bSDarrick J. Wong	}
8a569d71SDarrick J. Wong}
8a569d71SDarrick J. Wong
c24b5dfaSDave Chinner/*
1abcf261SDave Chinner * Initialise a newly allocated inode and return the in-core inode to the
1abcf261SDave Chinner * caller locked exclusively.
1da177e4SLinus Torvalds */
b652afd9SDave Chinnerint
1abcf261SDave Chinnerxfs_init_new_inode(
f736d93dSChristoph Hellwig	struct user_namespace	*mnt_userns,
1abcf261SDave Chinner	struct xfs_trans	*tp,
1abcf261SDave Chinner	struct xfs_inode	*pip,
1abcf261SDave Chinner	xfs_ino_t		ino,
576b1d67SAl Viro	umode_t			mode,
31b084aeSNathan Scott	xfs_nlink_t		nlink,
66f36464SChristoph Hellwig	dev_t			rdev,
6743099cSArkadiusz Mi?kiewicz	prid_t			prid,
e6a688c3SDave Chinner	bool			init_xattrs,
1abcf261SDave Chinner	struct xfs_inode	**ipp)
1da177e4SLinus Torvalds{
01ea173eSChristoph Hellwig	struct inode		*dir = pip ? VFS_I(pip) : NULL;
93848a99SChristoph Hellwig	struct xfs_mount	*mp = tp->t_mountp;
1abcf261SDave Chinner	struct xfs_inode	*ip;
1abcf261SDave Chinner	unsigned int		flags;
1da177e4SLinus Torvalds	int			error;
95582b00SDeepa Dinamani	struct timespec64	tv;
3987848cSDave Chinner	struct inode		*inode;
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
8b26984dSDave Chinner	 * Protect against obviously corrupt allocation btree records. Later
8b26984dSDave Chinner	 * xfs_iget checks will catch re-allocation of other active in-memory
8b26984dSDave Chinner	 * and on-disk inodes. If we don't catch reallocating the parent inode
8b26984dSDave Chinner	 * here we will deadlock in xfs_iget() so we have to do these checks
8b26984dSDave Chinner	 * first.
8b26984dSDave Chinner	 */
8b26984dSDave Chinner	if ((pip && ino == pip->i_ino) || !xfs_verify_dir_ino(mp, ino)) {
8b26984dSDave Chinner		xfs_alert(mp, "Allocated a known in-use inode 0x%llx!", ino);
8b26984dSDave Chinner		return -EFSCORRUPTED;
8b26984dSDave Chinner	}
8b26984dSDave Chinner
8b26984dSDave Chinner	/*
1abcf261SDave Chinner	 * Get the in-core inode with the lock held exclusively to prevent
1abcf261SDave Chinner	 * others from looking at until we're done.
1da177e4SLinus Torvalds	 */
1abcf261SDave Chinner	error = xfs_iget(mp, tp, ino, XFS_IGET_CREATE, XFS_ILOCK_EXCL, &ip);
bf904248SDavid Chinner	if (error)
1da177e4SLinus Torvalds		return error;
1abcf261SDave Chinner
1da177e4SLinus Torvalds	ASSERT(ip != NULL);
3987848cSDave Chinner	inode = VFS_I(ip);
54d7b5c1SDave Chinner	set_nlink(inode, nlink);
66f36464SChristoph Hellwig	inode->i_rdev = rdev;
ceaf603cSChristoph Hellwig	ip->i_projid = prid;
1da177e4SLinus Torvalds
*0560f31aSDave Chinner	if (dir && !(dir->i_mode & S_ISGID) && xfs_has_grpid(mp)) {
db998553SChristian Brauner		inode_fsuid_set(inode, mnt_userns);
01ea173eSChristoph Hellwig		inode->i_gid = dir->i_gid;
01ea173eSChristoph Hellwig		inode->i_mode = mode;
3d8f2821SChristoph Hellwig	} else {
7d6beb71SLinus Torvalds		inode_init_owner(mnt_userns, inode, dir, mode);
1da177e4SLinus Torvalds	}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
1da177e4SLinus Torvalds	 * If the group ID of the new file does not match the effective group
1da177e4SLinus Torvalds	 * ID or one of the supplementary group IDs, the S_ISGID bit is cleared
1da177e4SLinus Torvalds	 * (and only if the irix_sgid_inherit compatibility variable is set).
1da177e4SLinus Torvalds	 */
54295159SChristoph Hellwig	if (irix_sgid_inherit &&
f736d93dSChristoph Hellwig	    (inode->i_mode & S_ISGID) &&
f736d93dSChristoph Hellwig	    !in_group_p(i_gid_into_mnt(mnt_userns, inode)))
c19b3b05SDave Chinner		inode->i_mode &= ~S_ISGID;
1da177e4SLinus Torvalds
13d2c10bSChristoph Hellwig	ip->i_disk_size = 0;
daf83964SChristoph Hellwig	ip->i_df.if_nextents = 0;
6e73a545SChristoph Hellwig	ASSERT(ip->i_nblocks == 0);
dff35fd4SChristoph Hellwig
c2050a45SDeepa Dinamani	tv = current_time(inode);
3987848cSDave Chinner	inode->i_mtime = tv;
3987848cSDave Chinner	inode->i_atime = tv;
3987848cSDave Chinner	inode->i_ctime = tv;
dff35fd4SChristoph Hellwig
031474c2SChristoph Hellwig	ip->i_extsize = 0;
db07349dSChristoph Hellwig	ip->i_diflags = 0;
93848a99SChristoph Hellwig
38c26bfdSDave Chinner	if (xfs_has_v3inodes(mp)) {
f0e28280SJeff Layton		inode_set_iversion(inode, 1);
b33ce57dSChristoph Hellwig		ip->i_cowextsize = 0;
e98d5e88SChristoph Hellwig		ip->i_crtime = tv;
93848a99SChristoph Hellwig	}
93848a99SChristoph Hellwig
1da177e4SLinus Torvalds	flags = XFS_ILOG_CORE;
1da177e4SLinus Torvalds	switch (mode & S_IFMT) {
1da177e4SLinus Torvalds	case S_IFIFO:
1da177e4SLinus Torvalds	case S_IFCHR:
1da177e4SLinus Torvalds	case S_IFBLK:
1da177e4SLinus Torvalds	case S_IFSOCK:
f7e67b20SChristoph Hellwig		ip->i_df.if_format = XFS_DINODE_FMT_DEV;
1da177e4SLinus Torvalds		flags |= XFS_ILOG_DEV;
1da177e4SLinus Torvalds		break;
1da177e4SLinus Torvalds	case S_IFREG:
1da177e4SLinus Torvalds	case S_IFDIR:
db07349dSChristoph Hellwig		if (pip && (pip->i_diflags & XFS_DIFLAG_ANY))
8a569d71SDarrick J. Wong			xfs_inode_inherit_flags(ip, pip);
3e09ab8fSChristoph Hellwig		if (pip && (pip->i_diflags2 & XFS_DIFLAG2_ANY))
8a569d71SDarrick J. Wong			xfs_inode_inherit_flags2(ip, pip);
53004ee7SGustavo A. R. Silva		fallthrough;
1da177e4SLinus Torvalds	case S_IFLNK:
f7e67b20SChristoph Hellwig		ip->i_df.if_format = XFS_DINODE_FMT_EXTENTS;
fcacbc3fSChristoph Hellwig		ip->i_df.if_bytes = 0;
6bdcf26aSChristoph Hellwig		ip->i_df.if_u1.if_root = NULL;
1da177e4SLinus Torvalds		break;
1da177e4SLinus Torvalds	default:
1da177e4SLinus Torvalds		ASSERT(0);
1da177e4SLinus Torvalds	}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
e6a688c3SDave Chinner	 * If we need to create attributes immediately after allocating the
e6a688c3SDave Chinner	 * inode, initialise an empty attribute fork right now. We use the
e6a688c3SDave Chinner	 * default fork offset for attributes here as we don't know exactly what
e6a688c3SDave Chinner	 * size or how many attributes we might be adding. We can do this
e6a688c3SDave Chinner	 * safely here because we know the data fork is completely empty and
e6a688c3SDave Chinner	 * this saves us from needing to run a separate transaction to set the
e6a688c3SDave Chinner	 * fork offset in the immediate future.
e6a688c3SDave Chinner	 */
38c26bfdSDave Chinner	if (init_xattrs && xfs_has_attr(mp)) {
7821ea30SChristoph Hellwig		ip->i_forkoff = xfs_default_attroffset(ip) >> 3;
e6a688c3SDave Chinner		ip->i_afp = xfs_ifork_alloc(XFS_DINODE_FMT_EXTENTS, 0);
e6a688c3SDave Chinner	}
e6a688c3SDave Chinner
e6a688c3SDave Chinner	/*
1da177e4SLinus Torvalds	 * Log the new values stuffed into the inode.
1da177e4SLinus Torvalds	 */
ddc3415aSChristoph Hellwig	xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL);
1da177e4SLinus Torvalds	xfs_trans_log_inode(tp, ip, flags);
1da177e4SLinus Torvalds
58c90473SDave Chinner	/* now that we have an i_mode we can setup the inode structure */
41be8bedSChristoph Hellwig	xfs_setup_inode(ip);
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	*ipp = ip;
1da177e4SLinus Torvalds	return 0;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
e546cb79SDave Chinner/*
54d7b5c1SDave Chinner * Decrement the link count on an inode & log the change.  If this causes the
54d7b5c1SDave Chinner * link count to go to zero, move the inode to AGI unlinked list so that it can
54d7b5c1SDave Chinner * be freed when the last active reference goes away via xfs_inactive().
e546cb79SDave Chinner */
0d5a75e9SEric Sandeenstatic int			/* error */
e546cb79SDave Chinnerxfs_droplink(
e546cb79SDave Chinner	xfs_trans_t *tp,
e546cb79SDave Chinner	xfs_inode_t *ip)
e546cb79SDave Chinner{
e546cb79SDave Chinner	xfs_trans_ichgtime(tp, ip, XFS_ICHGTIME_CHG);
e546cb79SDave Chinner
e546cb79SDave Chinner	drop_nlink(VFS_I(ip));
e546cb79SDave Chinner	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
e546cb79SDave Chinner
54d7b5c1SDave Chinner	if (VFS_I(ip)->i_nlink)
54d7b5c1SDave Chinner		return 0;
54d7b5c1SDave Chinner
54d7b5c1SDave Chinner	return xfs_iunlink(tp, ip);
e546cb79SDave Chinner}
e546cb79SDave Chinner
e546cb79SDave Chinner/*
e546cb79SDave Chinner * Increment the link count on an inode & log the change.
e546cb79SDave Chinner */
91083269SEric Sandeenstatic void
e546cb79SDave Chinnerxfs_bumplink(
e546cb79SDave Chinner	xfs_trans_t *tp,
e546cb79SDave Chinner	xfs_inode_t *ip)
e546cb79SDave Chinner{
e546cb79SDave Chinner	xfs_trans_ichgtime(tp, ip, XFS_ICHGTIME_CHG);
e546cb79SDave Chinner
e546cb79SDave Chinner	inc_nlink(VFS_I(ip));
e546cb79SDave Chinner	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
e546cb79SDave Chinner}
e546cb79SDave Chinner
c24b5dfaSDave Chinnerint
c24b5dfaSDave Chinnerxfs_create(
f736d93dSChristoph Hellwig	struct user_namespace	*mnt_userns,
c24b5dfaSDave Chinner	xfs_inode_t		*dp,
c24b5dfaSDave Chinner	struct xfs_name		*name,
c24b5dfaSDave Chinner	umode_t			mode,
66f36464SChristoph Hellwig	dev_t			rdev,
e6a688c3SDave Chinner	bool			init_xattrs,
c24b5dfaSDave Chinner	xfs_inode_t		**ipp)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	int			is_dir = S_ISDIR(mode);
c24b5dfaSDave Chinner	struct xfs_mount	*mp = dp->i_mount;
c24b5dfaSDave Chinner	struct xfs_inode	*ip = NULL;
c24b5dfaSDave Chinner	struct xfs_trans	*tp = NULL;
c24b5dfaSDave Chinner	int			error;
c24b5dfaSDave Chinner	bool                    unlock_dp_on_error = false;
c24b5dfaSDave Chinner	prid_t			prid;
c24b5dfaSDave Chinner	struct xfs_dquot	*udqp = NULL;
c24b5dfaSDave Chinner	struct xfs_dquot	*gdqp = NULL;
c24b5dfaSDave Chinner	struct xfs_dquot	*pdqp = NULL;
062647a8SBrian Foster	struct xfs_trans_res	*tres;
c24b5dfaSDave Chinner	uint			resblks;
b652afd9SDave Chinner	xfs_ino_t		ino;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	trace_xfs_create(dp, name);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (XFS_FORCED_SHUTDOWN(mp))
2451337dSDave Chinner		return -EIO;
c24b5dfaSDave Chinner
163467d3SZhi Yong Wu	prid = xfs_get_initial_prid(dp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Make sure that we have allocated dquot(s) on disk.
c24b5dfaSDave Chinner	 */
a65e58e7SChristian Brauner	error = xfs_qm_vop_dqalloc(dp, mapped_fsuid(mnt_userns),
a65e58e7SChristian Brauner			mapped_fsgid(mnt_userns), prid,
c24b5dfaSDave Chinner			XFS_QMOPT_QUOTALL | XFS_QMOPT_INHERIT,
c24b5dfaSDave Chinner			&udqp, &gdqp, &pdqp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		return error;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (is_dir) {
c24b5dfaSDave Chinner		resblks = XFS_MKDIR_SPACE_RES(mp, name->len);
062647a8SBrian Foster		tres = &M_RES(mp)->tr_mkdir;
c24b5dfaSDave Chinner	} else {
c24b5dfaSDave Chinner		resblks = XFS_CREATE_SPACE_RES(mp, name->len);
062647a8SBrian Foster		tres = &M_RES(mp)->tr_create;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Initially assume that the file does not exist and
c24b5dfaSDave Chinner	 * reserve the resources for that case.  If that is not
c24b5dfaSDave Chinner	 * the case we'll drop the one we have and get a more
c24b5dfaSDave Chinner	 * appropriate transaction later.
c24b5dfaSDave Chinner	 */
f2f7b9ffSDarrick J. Wong	error = xfs_trans_alloc_icreate(mp, tres, udqp, gdqp, pdqp, resblks,
f2f7b9ffSDarrick J. Wong			&tp);
2451337dSDave Chinner	if (error == -ENOSPC) {
c24b5dfaSDave Chinner		/* flush outstanding delalloc blocks and retry */
c24b5dfaSDave Chinner		xfs_flush_inodes(mp);
f2f7b9ffSDarrick J. Wong		error = xfs_trans_alloc_icreate(mp, tres, udqp, gdqp, pdqp,
f2f7b9ffSDarrick J. Wong				resblks, &tp);
c24b5dfaSDave Chinner	}
4906e215SChristoph Hellwig	if (error)
f2f7b9ffSDarrick J. Wong		goto out_release_dquots;
c24b5dfaSDave Chinner
65523218SChristoph Hellwig	xfs_ilock(dp, XFS_ILOCK_EXCL | XFS_ILOCK_PARENT);
c24b5dfaSDave Chinner	unlock_dp_on_error = true;
c24b5dfaSDave Chinner
f5d92749SChandan Babu R	error = xfs_iext_count_may_overflow(dp, XFS_DATA_FORK,
f5d92749SChandan Babu R			XFS_IEXT_DIR_MANIP_CNT(mp));
f5d92749SChandan Babu R	if (error)
f5d92749SChandan Babu R		goto out_trans_cancel;
f5d92749SChandan Babu R
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * A newly created regular or special file just has one directory
c24b5dfaSDave Chinner	 * entry pointing to them, but a directory also the "." entry
c24b5dfaSDave Chinner	 * pointing to itself.
c24b5dfaSDave Chinner	 */
b652afd9SDave Chinner	error = xfs_dialloc(&tp, dp->i_ino, mode, &ino);
b652afd9SDave Chinner	if (!error)
b652afd9SDave Chinner		error = xfs_init_new_inode(mnt_userns, tp, dp, ino, mode,
b652afd9SDave Chinner				is_dir ? 2 : 1, rdev, prid, init_xattrs, &ip);
d6077aa3SJan Kara	if (error)
c24b5dfaSDave Chinner		goto out_trans_cancel;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Now we join the directory inode to the transaction.  We do not do it
b652afd9SDave Chinner	 * earlier because xfs_dialloc might commit the previous transaction
c24b5dfaSDave Chinner	 * (and release all the locks).  An error from here on will result in
c24b5dfaSDave Chinner	 * the transaction cancel unlocking dp so don't do it explicitly in the
c24b5dfaSDave Chinner	 * error path.
c24b5dfaSDave Chinner	 */
65523218SChristoph Hellwig	xfs_trans_ijoin(tp, dp, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner	unlock_dp_on_error = false;
c24b5dfaSDave Chinner
381eee69SBrian Foster	error = xfs_dir_createname(tp, dp, name, ip->i_ino,
63337b63SKaixu Xia					resblks - XFS_IALLOC_SPACE_RES(mp));
c24b5dfaSDave Chinner	if (error) {
2451337dSDave Chinner		ASSERT(error != -ENOSPC);
4906e215SChristoph Hellwig		goto out_trans_cancel;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner	xfs_trans_ichgtime(tp, dp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
c24b5dfaSDave Chinner	xfs_trans_log_inode(tp, dp, XFS_ILOG_CORE);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (is_dir) {
c24b5dfaSDave Chinner		error = xfs_dir_init(tp, ip, dp);
c24b5dfaSDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
c24b5dfaSDave Chinner
91083269SEric Sandeen		xfs_bumplink(tp, dp);
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If this is a synchronous mount, make sure that the
c24b5dfaSDave Chinner	 * create transaction goes to disk before returning to
c24b5dfaSDave Chinner	 * the user.
c24b5dfaSDave Chinner	 */
*0560f31aSDave Chinner	if (xfs_has_wsync(mp) || xfs_has_dirsync(mp))
c24b5dfaSDave Chinner		xfs_trans_set_sync(tp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Attach the dquot(s) to the inodes and modify them incore.
c24b5dfaSDave Chinner	 * These ids of the inode couldn't have changed since the new
c24b5dfaSDave Chinner	 * inode has been locked ever since it was created.
c24b5dfaSDave Chinner	 */
c24b5dfaSDave Chinner	xfs_qm_vop_create_dqattach(tp, ip, udqp, gdqp, pdqp);
c24b5dfaSDave Chinner
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto out_release_inode;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	xfs_qm_dqrele(udqp);
c24b5dfaSDave Chinner	xfs_qm_dqrele(gdqp);
c24b5dfaSDave Chinner	xfs_qm_dqrele(pdqp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	*ipp = ip;
c24b5dfaSDave Chinner	return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner out_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
c24b5dfaSDave Chinner out_release_inode:
c24b5dfaSDave Chinner	/*
58c90473SDave Chinner	 * Wait until after the current transaction is aborted to finish the
58c90473SDave Chinner	 * setup of the inode and release the inode.  This prevents recursive
58c90473SDave Chinner	 * transactions and deadlocks from xfs_inactive.
c24b5dfaSDave Chinner	 */
58c90473SDave Chinner	if (ip) {
58c90473SDave Chinner		xfs_finish_inode_setup(ip);
44a8736bSDarrick J. Wong		xfs_irele(ip);
58c90473SDave Chinner	}
f2f7b9ffSDarrick J. Wong out_release_dquots:
c24b5dfaSDave Chinner	xfs_qm_dqrele(udqp);
c24b5dfaSDave Chinner	xfs_qm_dqrele(gdqp);
c24b5dfaSDave Chinner	xfs_qm_dqrele(pdqp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (unlock_dp_on_error)
65523218SChristoph Hellwig		xfs_iunlock(dp, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinnerint
99b6436bSZhi Yong Wuxfs_create_tmpfile(
f736d93dSChristoph Hellwig	struct user_namespace	*mnt_userns,
99b6436bSZhi Yong Wu	struct xfs_inode	*dp,
330033d6SBrian Foster	umode_t			mode,
330033d6SBrian Foster	struct xfs_inode	**ipp)
99b6436bSZhi Yong Wu{
99b6436bSZhi Yong Wu	struct xfs_mount	*mp = dp->i_mount;
99b6436bSZhi Yong Wu	struct xfs_inode	*ip = NULL;
99b6436bSZhi Yong Wu	struct xfs_trans	*tp = NULL;
99b6436bSZhi Yong Wu	int			error;
99b6436bSZhi Yong Wu	prid_t                  prid;
99b6436bSZhi Yong Wu	struct xfs_dquot	*udqp = NULL;
99b6436bSZhi Yong Wu	struct xfs_dquot	*gdqp = NULL;
99b6436bSZhi Yong Wu	struct xfs_dquot	*pdqp = NULL;
99b6436bSZhi Yong Wu	struct xfs_trans_res	*tres;
99b6436bSZhi Yong Wu	uint			resblks;
b652afd9SDave Chinner	xfs_ino_t		ino;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	if (XFS_FORCED_SHUTDOWN(mp))
2451337dSDave Chinner		return -EIO;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	prid = xfs_get_initial_prid(dp);
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	/*
99b6436bSZhi Yong Wu	 * Make sure that we have allocated dquot(s) on disk.
99b6436bSZhi Yong Wu	 */
a65e58e7SChristian Brauner	error = xfs_qm_vop_dqalloc(dp, mapped_fsuid(mnt_userns),
a65e58e7SChristian Brauner			mapped_fsgid(mnt_userns), prid,
99b6436bSZhi Yong Wu			XFS_QMOPT_QUOTALL | XFS_QMOPT_INHERIT,
99b6436bSZhi Yong Wu			&udqp, &gdqp, &pdqp);
99b6436bSZhi Yong Wu	if (error)
99b6436bSZhi Yong Wu		return error;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	resblks = XFS_IALLOC_SPACE_RES(mp);
99b6436bSZhi Yong Wu	tres = &M_RES(mp)->tr_create_tmpfile;
253f4911SChristoph Hellwig
f2f7b9ffSDarrick J. Wong	error = xfs_trans_alloc_icreate(mp, tres, udqp, gdqp, pdqp, resblks,
f2f7b9ffSDarrick J. Wong			&tp);
4906e215SChristoph Hellwig	if (error)
f2f7b9ffSDarrick J. Wong		goto out_release_dquots;
99b6436bSZhi Yong Wu
b652afd9SDave Chinner	error = xfs_dialloc(&tp, dp->i_ino, mode, &ino);
b652afd9SDave Chinner	if (!error)
b652afd9SDave Chinner		error = xfs_init_new_inode(mnt_userns, tp, dp, ino, mode,
b652afd9SDave Chinner				0, 0, prid, false, &ip);
d6077aa3SJan Kara	if (error)
99b6436bSZhi Yong Wu		goto out_trans_cancel;
99b6436bSZhi Yong Wu
*0560f31aSDave Chinner	if (xfs_has_wsync(mp))
99b6436bSZhi Yong Wu		xfs_trans_set_sync(tp);
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	/*
99b6436bSZhi Yong Wu	 * Attach the dquot(s) to the inodes and modify them incore.
99b6436bSZhi Yong Wu	 * These ids of the inode couldn't have changed since the new
99b6436bSZhi Yong Wu	 * inode has been locked ever since it was created.
99b6436bSZhi Yong Wu	 */
99b6436bSZhi Yong Wu	xfs_qm_vop_create_dqattach(tp, ip, udqp, gdqp, pdqp);
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	error = xfs_iunlink(tp, ip);
99b6436bSZhi Yong Wu	if (error)
4906e215SChristoph Hellwig		goto out_trans_cancel;
99b6436bSZhi Yong Wu
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
99b6436bSZhi Yong Wu	if (error)
99b6436bSZhi Yong Wu		goto out_release_inode;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	xfs_qm_dqrele(udqp);
99b6436bSZhi Yong Wu	xfs_qm_dqrele(gdqp);
99b6436bSZhi Yong Wu	xfs_qm_dqrele(pdqp);
99b6436bSZhi Yong Wu
330033d6SBrian Foster	*ipp = ip;
99b6436bSZhi Yong Wu	return 0;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu out_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
99b6436bSZhi Yong Wu out_release_inode:
99b6436bSZhi Yong Wu	/*
58c90473SDave Chinner	 * Wait until after the current transaction is aborted to finish the
58c90473SDave Chinner	 * setup of the inode and release the inode.  This prevents recursive
58c90473SDave Chinner	 * transactions and deadlocks from xfs_inactive.
99b6436bSZhi Yong Wu	 */
58c90473SDave Chinner	if (ip) {
58c90473SDave Chinner		xfs_finish_inode_setup(ip);
44a8736bSDarrick J. Wong		xfs_irele(ip);
58c90473SDave Chinner	}
f2f7b9ffSDarrick J. Wong out_release_dquots:
99b6436bSZhi Yong Wu	xfs_qm_dqrele(udqp);
99b6436bSZhi Yong Wu	xfs_qm_dqrele(gdqp);
99b6436bSZhi Yong Wu	xfs_qm_dqrele(pdqp);
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	return error;
99b6436bSZhi Yong Wu}
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wuint
c24b5dfaSDave Chinnerxfs_link(
c24b5dfaSDave Chinner	xfs_inode_t		*tdp,
c24b5dfaSDave Chinner	xfs_inode_t		*sip,
c24b5dfaSDave Chinner	struct xfs_name		*target_name)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	xfs_mount_t		*mp = tdp->i_mount;
c24b5dfaSDave Chinner	xfs_trans_t		*tp;
c24b5dfaSDave Chinner	int			error;
c24b5dfaSDave Chinner	int			resblks;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	trace_xfs_link(tdp, target_name);
c24b5dfaSDave Chinner
c19b3b05SDave Chinner	ASSERT(!S_ISDIR(VFS_I(sip)->i_mode));
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (XFS_FORCED_SHUTDOWN(mp))
2451337dSDave Chinner		return -EIO;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(sip);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(tdp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	resblks = XFS_LINK_SPACE_RES(mp, target_name->len);
253f4911SChristoph Hellwig	error = xfs_trans_alloc(mp, &M_RES(mp)->tr_link, resblks, 0, 0, &tp);
2451337dSDave Chinner	if (error == -ENOSPC) {
c24b5dfaSDave Chinner		resblks = 0;
253f4911SChristoph Hellwig		error = xfs_trans_alloc(mp, &M_RES(mp)->tr_link, 0, 0, 0, &tp);
c24b5dfaSDave Chinner	}
4906e215SChristoph Hellwig	if (error)
253f4911SChristoph Hellwig		goto std_return;
c24b5dfaSDave Chinner
7c2d238aSDarrick J. Wong	xfs_lock_two_inodes(sip, XFS_ILOCK_EXCL, tdp, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	xfs_trans_ijoin(tp, sip, XFS_ILOCK_EXCL);
65523218SChristoph Hellwig	xfs_trans_ijoin(tp, tdp, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner
f5d92749SChandan Babu R	error = xfs_iext_count_may_overflow(tdp, XFS_DATA_FORK,
f5d92749SChandan Babu R			XFS_IEXT_DIR_MANIP_CNT(mp));
f5d92749SChandan Babu R	if (error)
f5d92749SChandan Babu R		goto error_return;
f5d92749SChandan Babu R
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If we are using project inheritance, we only allow hard link
c24b5dfaSDave Chinner	 * creation in our tree when the project IDs are the same; else
c24b5dfaSDave Chinner	 * the tree quota mechanism could be circumvented.
c24b5dfaSDave Chinner	 */
db07349dSChristoph Hellwig	if (unlikely((tdp->i_diflags & XFS_DIFLAG_PROJINHERIT) &&
ceaf603cSChristoph Hellwig		     tdp->i_projid != sip->i_projid)) {
2451337dSDave Chinner		error = -EXDEV;
c24b5dfaSDave Chinner		goto error_return;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
94f3cad5SEric Sandeen	if (!resblks) {
94f3cad5SEric Sandeen		error = xfs_dir_canenter(tp, tdp, target_name);
c24b5dfaSDave Chinner		if (error)
c24b5dfaSDave Chinner			goto error_return;
94f3cad5SEric Sandeen	}
c24b5dfaSDave Chinner
54d7b5c1SDave Chinner	/*
54d7b5c1SDave Chinner	 * Handle initial link state of O_TMPFILE inode
54d7b5c1SDave Chinner	 */
54d7b5c1SDave Chinner	if (VFS_I(sip)->i_nlink == 0) {
f40aadb2SDave Chinner		struct xfs_perag	*pag;
f40aadb2SDave Chinner
f40aadb2SDave Chinner		pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, sip->i_ino));
f40aadb2SDave Chinner		error = xfs_iunlink_remove(tp, pag, sip);
f40aadb2SDave Chinner		xfs_perag_put(pag);
ab297431SZhi Yong Wu		if (error)
4906e215SChristoph Hellwig			goto error_return;
ab297431SZhi Yong Wu	}
ab297431SZhi Yong Wu
c24b5dfaSDave Chinner	error = xfs_dir_createname(tp, tdp, target_name, sip->i_ino,
381eee69SBrian Foster				   resblks);
c24b5dfaSDave Chinner	if (error)
4906e215SChristoph Hellwig		goto error_return;
c24b5dfaSDave Chinner	xfs_trans_ichgtime(tp, tdp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
c24b5dfaSDave Chinner	xfs_trans_log_inode(tp, tdp, XFS_ILOG_CORE);
c24b5dfaSDave Chinner
91083269SEric Sandeen	xfs_bumplink(tp, sip);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If this is a synchronous mount, make sure that the
c24b5dfaSDave Chinner	 * link transaction goes to disk before returning to
c24b5dfaSDave Chinner	 * the user.
c24b5dfaSDave Chinner	 */
*0560f31aSDave Chinner	if (xfs_has_wsync(mp) || xfs_has_dirsync(mp))
c24b5dfaSDave Chinner		xfs_trans_set_sync(tp);
c24b5dfaSDave Chinner
70393313SChristoph Hellwig	return xfs_trans_commit(tp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner error_return:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
c24b5dfaSDave Chinner std_return:
c24b5dfaSDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
363e59baSDarrick J. Wong/* Clear the reflink flag and the cowblocks tag if possible. */
363e59baSDarrick J. Wongstatic void
363e59baSDarrick J. Wongxfs_itruncate_clear_reflink_flags(
363e59baSDarrick J. Wong	struct xfs_inode	*ip)
363e59baSDarrick J. Wong{
363e59baSDarrick J. Wong	struct xfs_ifork	*dfork;
363e59baSDarrick J. Wong	struct xfs_ifork	*cfork;
363e59baSDarrick J. Wong
363e59baSDarrick J. Wong	if (!xfs_is_reflink_inode(ip))
363e59baSDarrick J. Wong		return;
363e59baSDarrick J. Wong	dfork = XFS_IFORK_PTR(ip, XFS_DATA_FORK);
363e59baSDarrick J. Wong	cfork = XFS_IFORK_PTR(ip, XFS_COW_FORK);
363e59baSDarrick J. Wong	if (dfork->if_bytes == 0 && cfork->if_bytes == 0)
3e09ab8fSChristoph Hellwig		ip->i_diflags2 &= ~XFS_DIFLAG2_REFLINK;
363e59baSDarrick J. Wong	if (cfork->if_bytes == 0)
363e59baSDarrick J. Wong		xfs_inode_clear_cowblocks_tag(ip);
363e59baSDarrick J. Wong}
363e59baSDarrick J. Wong
1da177e4SLinus Torvalds/*
8f04c47aSChristoph Hellwig * Free up the underlying blocks past new_size.  The new size must be smaller
8f04c47aSChristoph Hellwig * than the current size.  This routine can be used both for the attribute and
8f04c47aSChristoph Hellwig * data fork, and does not modify the inode size, which is left to the caller.
1da177e4SLinus Torvalds *
f6485057SDavid Chinner * The transaction passed to this routine must have made a permanent log
f6485057SDavid Chinner * reservation of at least XFS_ITRUNCATE_LOG_RES.  This routine may commit the
f6485057SDavid Chinner * given transaction and start new ones, so make sure everything involved in
f6485057SDavid Chinner * the transaction is tidy before calling here.  Some transaction will be
f6485057SDavid Chinner * returned to the caller to be committed.  The incoming transaction must
f6485057SDavid Chinner * already include the inode, and both inode locks must be held exclusively.
f6485057SDavid Chinner * The inode must also be "held" within the transaction.  On return the inode
f6485057SDavid Chinner * will be "held" within the returned transaction.  This routine does NOT
f6485057SDavid Chinner * require any disk space to be reserved for it within the transaction.
1da177e4SLinus Torvalds *
f6485057SDavid Chinner * If we get an error, we must return with the inode locked and linked into the
f6485057SDavid Chinner * current transaction. This keeps things simple for the higher level code,
f6485057SDavid Chinner * because it always knows that the inode is locked and held in the transaction
f6485057SDavid Chinner * that returns to it whether errors occur or not.  We don't mark the inode
f6485057SDavid Chinner * dirty on error so that transactions can be easily aborted if possible.
1da177e4SLinus Torvalds */
1da177e4SLinus Torvaldsint
4e529339SBrian Fosterxfs_itruncate_extents_flags(
8f04c47aSChristoph Hellwig	struct xfs_trans	**tpp,
8f04c47aSChristoph Hellwig	struct xfs_inode	*ip,
8f04c47aSChristoph Hellwig	int			whichfork,
13b86fc3SBrian Foster	xfs_fsize_t		new_size,
4e529339SBrian Foster	int			flags)
1da177e4SLinus Torvalds{
8f04c47aSChristoph Hellwig	struct xfs_mount	*mp = ip->i_mount;
8f04c47aSChristoph Hellwig	struct xfs_trans	*tp = *tpp;
1da177e4SLinus Torvalds	xfs_fileoff_t		first_unmap_block;
8f04c47aSChristoph Hellwig	xfs_filblks_t		unmap_len;
8f04c47aSChristoph Hellwig	int			error = 0;
1da177e4SLinus Torvalds
0b56185bSChristoph Hellwig	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL));
0b56185bSChristoph Hellwig	ASSERT(!atomic_read(&VFS_I(ip)->i_count) ||
0b56185bSChristoph Hellwig	       xfs_isilocked(ip, XFS_IOLOCK_EXCL));
ce7ae151SChristoph Hellwig	ASSERT(new_size <= XFS_ISIZE(ip));
8f04c47aSChristoph Hellwig	ASSERT(tp->t_flags & XFS_TRANS_PERM_LOG_RES);
1da177e4SLinus Torvalds	ASSERT(ip->i_itemp != NULL);
898621d5SChristoph Hellwig	ASSERT(ip->i_itemp->ili_lock_flags == 0);
1da177e4SLinus Torvalds	ASSERT(!XFS_NOT_DQATTACHED(mp, ip));
1da177e4SLinus Torvalds
673e8e59SChristoph Hellwig	trace_xfs_itruncate_extents_start(ip, new_size);
673e8e59SChristoph Hellwig
4e529339SBrian Foster	flags |= xfs_bmapi_aflag(whichfork);
13b86fc3SBrian Foster
1da177e4SLinus Torvalds	/*
1da177e4SLinus Torvalds	 * Since it is possible for space to become allocated beyond
1da177e4SLinus Torvalds	 * the end of the file (in a crash where the space is allocated
1da177e4SLinus Torvalds	 * but the inode size is not yet updated), simply remove any
1da177e4SLinus Torvalds	 * blocks which show up between the new EOF and the maximum
4bbb04abSDarrick J. Wong	 * possible file size.
4bbb04abSDarrick J. Wong	 *
4bbb04abSDarrick J. Wong	 * We have to free all the blocks to the bmbt maximum offset, even if
4bbb04abSDarrick J. Wong	 * the page cache can't scale that far.
1da177e4SLinus Torvalds	 */
8f04c47aSChristoph Hellwig	first_unmap_block = XFS_B_TO_FSB(mp, (xfs_ufsize_t)new_size);
33005fd0SDarrick J. Wong	if (!xfs_verify_fileoff(mp, first_unmap_block)) {
4bbb04abSDarrick J. Wong		WARN_ON_ONCE(first_unmap_block > XFS_MAX_FILEOFF);
8f04c47aSChristoph Hellwig		return 0;
4bbb04abSDarrick J. Wong	}
8f04c47aSChristoph Hellwig
4bbb04abSDarrick J. Wong	unmap_len = XFS_MAX_FILEOFF - first_unmap_block + 1;
4bbb04abSDarrick J. Wong	while (unmap_len > 0) {
02dff7bfSBrian Foster		ASSERT(tp->t_firstblock == NULLFSBLOCK);
4bbb04abSDarrick J. Wong		error = __xfs_bunmapi(tp, ip, first_unmap_block, &unmap_len,
4bbb04abSDarrick J. Wong				flags, XFS_ITRUNC_MAX_EXTENTS);
8f04c47aSChristoph Hellwig		if (error)
d5a2e289SBrian Foster			goto out;
1da177e4SLinus Torvalds
6dd379c7SBrian Foster		/* free the just unmapped extents */
9e28a242SBrian Foster		error = xfs_defer_finish(&tp);
8f04c47aSChristoph Hellwig		if (error)
9b1f4e98SBrian Foster			goto out;
1da177e4SLinus Torvalds	}
8f04c47aSChristoph Hellwig
4919d42aSDarrick J. Wong	if (whichfork == XFS_DATA_FORK) {
aa8968f2SDarrick J. Wong		/* Remove all pending CoW reservations. */
4919d42aSDarrick J. Wong		error = xfs_reflink_cancel_cow_blocks(ip, &tp,
4bbb04abSDarrick J. Wong				first_unmap_block, XFS_MAX_FILEOFF, true);
aa8968f2SDarrick J. Wong		if (error)
aa8968f2SDarrick J. Wong			goto out;
aa8968f2SDarrick J. Wong
363e59baSDarrick J. Wong		xfs_itruncate_clear_reflink_flags(ip);
4919d42aSDarrick J. Wong	}
aa8968f2SDarrick J. Wong
673e8e59SChristoph Hellwig	/*
673e8e59SChristoph Hellwig	 * Always re-log the inode so that our permanent transaction can keep
673e8e59SChristoph Hellwig	 * on rolling it forward in the log.
673e8e59SChristoph Hellwig	 */
673e8e59SChristoph Hellwig	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
673e8e59SChristoph Hellwig
673e8e59SChristoph Hellwig	trace_xfs_itruncate_extents_end(ip, new_size);
673e8e59SChristoph Hellwig
8f04c47aSChristoph Hellwigout:
8f04c47aSChristoph Hellwig	*tpp = tp;
8f04c47aSChristoph Hellwig	return error;
8f04c47aSChristoph Hellwig}
8f04c47aSChristoph Hellwig
c24b5dfaSDave Chinnerint
c24b5dfaSDave Chinnerxfs_release(
c24b5dfaSDave Chinner	xfs_inode_t	*ip)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	xfs_mount_t	*mp = ip->i_mount;
7d88329eSDarrick J. Wong	int		error = 0;
c24b5dfaSDave Chinner
c19b3b05SDave Chinner	if (!S_ISREG(VFS_I(ip)->i_mode) || (VFS_I(ip)->i_mode == 0))
c24b5dfaSDave Chinner		return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/* If this is a read-only mount, don't do this (would generate I/O) */
c24b5dfaSDave Chinner	if (mp->m_flags & XFS_MOUNT_RDONLY)
c24b5dfaSDave Chinner		return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (!XFS_FORCED_SHUTDOWN(mp)) {
c24b5dfaSDave Chinner		int truncated;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		/*
c24b5dfaSDave Chinner		 * If we previously truncated this file and removed old data
c24b5dfaSDave Chinner		 * in the process, we want to initiate "early" writeout on
c24b5dfaSDave Chinner		 * the last close.  This is an attempt to combat the notorious
c24b5dfaSDave Chinner		 * NULL files problem which is particularly noticeable from a
c24b5dfaSDave Chinner		 * truncate down, buffered (re-)write (delalloc), followed by
c24b5dfaSDave Chinner		 * a crash.  What we are effectively doing here is
c24b5dfaSDave Chinner		 * significantly reducing the time window where we'd otherwise
c24b5dfaSDave Chinner		 * be exposed to that problem.
c24b5dfaSDave Chinner		 */
c24b5dfaSDave Chinner		truncated = xfs_iflags_test_and_clear(ip, XFS_ITRUNCATED);
c24b5dfaSDave Chinner		if (truncated) {
c24b5dfaSDave Chinner			xfs_iflags_clear(ip, XFS_IDIRTY_RELEASE);
eac152b4SDave Chinner			if (ip->i_delayed_blks > 0) {
2451337dSDave Chinner				error = filemap_flush(VFS_I(ip)->i_mapping);
c24b5dfaSDave Chinner				if (error)
c24b5dfaSDave Chinner					return error;
c24b5dfaSDave Chinner			}
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
54d7b5c1SDave Chinner	if (VFS_I(ip)->i_nlink == 0)
c24b5dfaSDave Chinner		return 0;
c24b5dfaSDave Chinner
7d88329eSDarrick J. Wong	/*
7d88329eSDarrick J. Wong	 * If we can't get the iolock just skip truncating the blocks past EOF
7d88329eSDarrick J. Wong	 * because we could deadlock with the mmap_lock otherwise. We'll get
7d88329eSDarrick J. Wong	 * another chance to drop them once the last reference to the inode is
7d88329eSDarrick J. Wong	 * dropped, so we'll never leak blocks permanently.
7d88329eSDarrick J. Wong	 */
7d88329eSDarrick J. Wong	if (!xfs_ilock_nowait(ip, XFS_IOLOCK_EXCL))
7d88329eSDarrick J. Wong		return 0;
c24b5dfaSDave Chinner
7d88329eSDarrick J. Wong	if (xfs_can_free_eofblocks(ip, false)) {
c24b5dfaSDave Chinner		/*
a36b9261SBrian Foster		 * Check if the inode is being opened, written and closed
a36b9261SBrian Foster		 * frequently and we have delayed allocation blocks outstanding
a36b9261SBrian Foster		 * (e.g. streaming writes from the NFS server), truncating the
a36b9261SBrian Foster		 * blocks past EOF will cause fragmentation to occur.
a36b9261SBrian Foster		 *
a36b9261SBrian Foster		 * In this case don't do the truncation, but we have to be
a36b9261SBrian Foster		 * careful how we detect this case. Blocks beyond EOF show up as
a36b9261SBrian Foster		 * i_delayed_blks even when the inode is clean, so we need to
a36b9261SBrian Foster		 * truncate them away first before checking for a dirty release.
a36b9261SBrian Foster		 * Hence on the first dirty close we will still remove the
a36b9261SBrian Foster		 * speculative allocation, but after that we will leave it in
a36b9261SBrian Foster		 * place.
a36b9261SBrian Foster		 */
a36b9261SBrian Foster		if (xfs_iflags_test(ip, XFS_IDIRTY_RELEASE))
7d88329eSDarrick J. Wong			goto out_unlock;
7d88329eSDarrick J. Wong
a36b9261SBrian Foster		error = xfs_free_eofblocks(ip);
a36b9261SBrian Foster		if (error)
7d88329eSDarrick J. Wong			goto out_unlock;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		/* delalloc blocks after truncation means it really is dirty */
c24b5dfaSDave Chinner		if (ip->i_delayed_blks)
c24b5dfaSDave Chinner			xfs_iflags_set(ip, XFS_IDIRTY_RELEASE);
c24b5dfaSDave Chinner	}
7d88329eSDarrick J. Wong
7d88329eSDarrick J. Wongout_unlock:
7d88329eSDarrick J. Wong	xfs_iunlock(ip, XFS_IOLOCK_EXCL);
7d88329eSDarrick J. Wong	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner/*
f7be2d7fSBrian Foster * xfs_inactive_truncate
f7be2d7fSBrian Foster *
f7be2d7fSBrian Foster * Called to perform a truncate when an inode becomes unlinked.
f7be2d7fSBrian Foster */
f7be2d7fSBrian FosterSTATIC int
f7be2d7fSBrian Fosterxfs_inactive_truncate(
f7be2d7fSBrian Foster	struct xfs_inode *ip)
f7be2d7fSBrian Foster{
f7be2d7fSBrian Foster	struct xfs_mount	*mp = ip->i_mount;
f7be2d7fSBrian Foster	struct xfs_trans	*tp;
f7be2d7fSBrian Foster	int			error;
f7be2d7fSBrian Foster
253f4911SChristoph Hellwig	error = xfs_trans_alloc(mp, &M_RES(mp)->tr_itruncate, 0, 0, 0, &tp);
f7be2d7fSBrian Foster	if (error) {
f7be2d7fSBrian Foster		ASSERT(XFS_FORCED_SHUTDOWN(mp));
f7be2d7fSBrian Foster		return error;
f7be2d7fSBrian Foster	}
f7be2d7fSBrian Foster	xfs_ilock(ip, XFS_ILOCK_EXCL);
f7be2d7fSBrian Foster	xfs_trans_ijoin(tp, ip, 0);
f7be2d7fSBrian Foster
f7be2d7fSBrian Foster	/*
f7be2d7fSBrian Foster	 * Log the inode size first to prevent stale data exposure in the event
f7be2d7fSBrian Foster	 * of a system crash before the truncate completes. See the related
69bca807SJan Kara	 * comment in xfs_vn_setattr_size() for details.
f7be2d7fSBrian Foster	 */
13d2c10bSChristoph Hellwig	ip->i_disk_size = 0;
f7be2d7fSBrian Foster	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
f7be2d7fSBrian Foster
f7be2d7fSBrian Foster	error = xfs_itruncate_extents(&tp, ip, XFS_DATA_FORK, 0);
f7be2d7fSBrian Foster	if (error)
f7be2d7fSBrian Foster		goto error_trans_cancel;
f7be2d7fSBrian Foster
daf83964SChristoph Hellwig	ASSERT(ip->i_df.if_nextents == 0);
f7be2d7fSBrian Foster
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
f7be2d7fSBrian Foster	if (error)
f7be2d7fSBrian Foster		goto error_unlock;
f7be2d7fSBrian Foster
f7be2d7fSBrian Foster	xfs_iunlock(ip, XFS_ILOCK_EXCL);
f7be2d7fSBrian Foster	return 0;
f7be2d7fSBrian Foster
f7be2d7fSBrian Fostererror_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
f7be2d7fSBrian Fostererror_unlock:
f7be2d7fSBrian Foster	xfs_iunlock(ip, XFS_ILOCK_EXCL);
f7be2d7fSBrian Foster	return error;
f7be2d7fSBrian Foster}
f7be2d7fSBrian Foster
f7be2d7fSBrian Foster/*
88877d2bSBrian Foster * xfs_inactive_ifree()
88877d2bSBrian Foster *
88877d2bSBrian Foster * Perform the inode free when an inode is unlinked.
88877d2bSBrian Foster */
88877d2bSBrian FosterSTATIC int
88877d2bSBrian Fosterxfs_inactive_ifree(
88877d2bSBrian Foster	struct xfs_inode *ip)
88877d2bSBrian Foster{
88877d2bSBrian Foster	struct xfs_mount	*mp = ip->i_mount;
88877d2bSBrian Foster	struct xfs_trans	*tp;
88877d2bSBrian Foster	int			error;
88877d2bSBrian Foster
9d43b180SBrian Foster	/*
76d771b4SChristoph Hellwig	 * We try to use a per-AG reservation for any block needed by the finobt
76d771b4SChristoph Hellwig	 * tree, but as the finobt feature predates the per-AG reservation
76d771b4SChristoph Hellwig	 * support a degraded file system might not have enough space for the
76d771b4SChristoph Hellwig	 * reservation at mount time.  In that case try to dip into the reserved
76d771b4SChristoph Hellwig	 * pool and pray.
9d43b180SBrian Foster	 *
9d43b180SBrian Foster	 * Send a warning if the reservation does happen to fail, as the inode
9d43b180SBrian Foster	 * now remains allocated and sits on the unlinked list until the fs is
9d43b180SBrian Foster	 * repaired.
9d43b180SBrian Foster	 */
e1f6ca11SDarrick J. Wong	if (unlikely(mp->m_finobt_nores)) {
253f4911SChristoph Hellwig		error = xfs_trans_alloc(mp, &M_RES(mp)->tr_ifree,
76d771b4SChristoph Hellwig				XFS_IFREE_SPACE_RES(mp), 0, XFS_TRANS_RESERVE,
76d771b4SChristoph Hellwig				&tp);
76d771b4SChristoph Hellwig	} else {
76d771b4SChristoph Hellwig		error = xfs_trans_alloc(mp, &M_RES(mp)->tr_ifree, 0, 0, 0, &tp);
76d771b4SChristoph Hellwig	}
88877d2bSBrian Foster	if (error) {
2451337dSDave Chinner		if (error == -ENOSPC) {
9d43b180SBrian Foster			xfs_warn_ratelimited(mp,
9d43b180SBrian Foster			"Failed to remove inode(s) from unlinked list. "
9d43b180SBrian Foster			"Please free space, unmount and run xfs_repair.");
9d43b180SBrian Foster		} else {
88877d2bSBrian Foster			ASSERT(XFS_FORCED_SHUTDOWN(mp));
9d43b180SBrian Foster		}
88877d2bSBrian Foster		return error;
88877d2bSBrian Foster	}
88877d2bSBrian Foster
96355d5aSDave Chinner	/*
96355d5aSDave Chinner	 * We do not hold the inode locked across the entire rolling transaction
96355d5aSDave Chinner	 * here. We only need to hold it for the first transaction that
96355d5aSDave Chinner	 * xfs_ifree() builds, which may mark the inode XFS_ISTALE if the
96355d5aSDave Chinner	 * underlying cluster buffer is freed. Relogging an XFS_ISTALE inode
96355d5aSDave Chinner	 * here breaks the relationship between cluster buffer invalidation and
96355d5aSDave Chinner	 * stale inode invalidation on cluster buffer item journal commit
96355d5aSDave Chinner	 * completion, and can result in leaving dirty stale inodes hanging
96355d5aSDave Chinner	 * around in memory.
96355d5aSDave Chinner	 *
96355d5aSDave Chinner	 * We have no need for serialising this inode operation against other
96355d5aSDave Chinner	 * operations - we freed the inode and hence reallocation is required
96355d5aSDave Chinner	 * and that will serialise on reallocating the space the deferops need
96355d5aSDave Chinner	 * to free. Hence we can unlock the inode on the first commit of
96355d5aSDave Chinner	 * the transaction rather than roll it right through the deferops. This
96355d5aSDave Chinner	 * avoids relogging the XFS_ISTALE inode.
96355d5aSDave Chinner	 *
96355d5aSDave Chinner	 * We check that xfs_ifree() hasn't grown an internal transaction roll
96355d5aSDave Chinner	 * by asserting that the inode is still locked when it returns.
96355d5aSDave Chinner	 */
88877d2bSBrian Foster	xfs_ilock(ip, XFS_ILOCK_EXCL);
96355d5aSDave Chinner	xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL);
88877d2bSBrian Foster
0e0417f3SBrian Foster	error = xfs_ifree(tp, ip);
96355d5aSDave Chinner	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL));
88877d2bSBrian Foster	if (error) {
88877d2bSBrian Foster		/*
88877d2bSBrian Foster		 * If we fail to free the inode, shut down.  The cancel
88877d2bSBrian Foster		 * might do that, we need to make sure.  Otherwise the
88877d2bSBrian Foster		 * inode might be lost for a long time or forever.
88877d2bSBrian Foster		 */
88877d2bSBrian Foster		if (!XFS_FORCED_SHUTDOWN(mp)) {
88877d2bSBrian Foster			xfs_notice(mp, "%s: xfs_ifree returned error %d",
88877d2bSBrian Foster				__func__, error);
88877d2bSBrian Foster			xfs_force_shutdown(mp, SHUTDOWN_META_IO_ERROR);
88877d2bSBrian Foster		}
4906e215SChristoph Hellwig		xfs_trans_cancel(tp);
88877d2bSBrian Foster		return error;
88877d2bSBrian Foster	}
88877d2bSBrian Foster
88877d2bSBrian Foster	/*
88877d2bSBrian Foster	 * Credit the quota account(s). The inode is gone.
88877d2bSBrian Foster	 */
88877d2bSBrian Foster	xfs_trans_mod_dquot_byino(tp, ip, XFS_TRANS_DQ_ICOUNT, -1);
88877d2bSBrian Foster
88877d2bSBrian Foster	/*
d4a97a04SBrian Foster	 * Just ignore errors at this point.  There is nothing we can do except
d4a97a04SBrian Foster	 * to try to keep going. Make sure it's not a silent error.
88877d2bSBrian Foster	 */
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
88877d2bSBrian Foster	if (error)
88877d2bSBrian Foster		xfs_notice(mp, "%s: xfs_trans_commit returned error %d",
88877d2bSBrian Foster			__func__, error);
88877d2bSBrian Foster
88877d2bSBrian Foster	return 0;
88877d2bSBrian Foster}
88877d2bSBrian Foster
88877d2bSBrian Foster/*
62af7d54SDarrick J. Wong * Returns true if we need to update the on-disk metadata before we can free
62af7d54SDarrick J. Wong * the memory used by this inode.  Updates include freeing post-eof
62af7d54SDarrick J. Wong * preallocations; freeing COW staging extents; and marking the inode free in
62af7d54SDarrick J. Wong * the inobt if it is on the unlinked list.
62af7d54SDarrick J. Wong */
62af7d54SDarrick J. Wongbool
62af7d54SDarrick J. Wongxfs_inode_needs_inactive(
62af7d54SDarrick J. Wong	struct xfs_inode	*ip)
62af7d54SDarrick J. Wong{
62af7d54SDarrick J. Wong	struct xfs_mount	*mp = ip->i_mount;
62af7d54SDarrick J. Wong	struct xfs_ifork	*cow_ifp = XFS_IFORK_PTR(ip, XFS_COW_FORK);
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/*
62af7d54SDarrick J. Wong	 * If the inode is already free, then there can be nothing
62af7d54SDarrick J. Wong	 * to clean up here.
62af7d54SDarrick J. Wong	 */
62af7d54SDarrick J. Wong	if (VFS_I(ip)->i_mode == 0)
62af7d54SDarrick J. Wong		return false;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* If this is a read-only mount, don't do this (would generate I/O) */
62af7d54SDarrick J. Wong	if (mp->m_flags & XFS_MOUNT_RDONLY)
62af7d54SDarrick J. Wong		return false;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* If the log isn't running, push inodes straight to reclaim. */
*0560f31aSDave Chinner	if (XFS_FORCED_SHUTDOWN(mp) || xfs_has_norecovery(mp))
62af7d54SDarrick J. Wong		return false;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* Metadata inodes require explicit resource cleanup. */
62af7d54SDarrick J. Wong	if (xfs_is_metadata_inode(ip))
62af7d54SDarrick J. Wong		return false;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* Want to clean out the cow blocks if there are any. */
62af7d54SDarrick J. Wong	if (cow_ifp && cow_ifp->if_bytes > 0)
62af7d54SDarrick J. Wong		return true;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* Unlinked files must be freed. */
62af7d54SDarrick J. Wong	if (VFS_I(ip)->i_nlink == 0)
62af7d54SDarrick J. Wong		return true;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/*
62af7d54SDarrick J. Wong	 * This file isn't being freed, so check if there are post-eof blocks
62af7d54SDarrick J. Wong	 * to free.  @force is true because we are evicting an inode from the
62af7d54SDarrick J. Wong	 * cache.  Post-eof blocks must be freed, lest we end up with broken
62af7d54SDarrick J. Wong	 * free space accounting.
62af7d54SDarrick J. Wong	 *
62af7d54SDarrick J. Wong	 * Note: don't bother with iolock here since lockdep complains about
62af7d54SDarrick J. Wong	 * acquiring it in reclaim context. We have the only reference to the
62af7d54SDarrick J. Wong	 * inode at this point anyways.
62af7d54SDarrick J. Wong	 */
62af7d54SDarrick J. Wong	return xfs_can_free_eofblocks(ip, true);
62af7d54SDarrick J. Wong}
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong/*
c24b5dfaSDave Chinner * xfs_inactive
c24b5dfaSDave Chinner *
c24b5dfaSDave Chinner * This is called when the vnode reference count for the vnode
c24b5dfaSDave Chinner * goes to zero.  If the file has been unlinked, then it must
c24b5dfaSDave Chinner * now be truncated.  Also, we clear all of the read-ahead state
c24b5dfaSDave Chinner * kept for the inode here since the file is now closed.
c24b5dfaSDave Chinner */
74564fb4SBrian Fostervoid
c24b5dfaSDave Chinnerxfs_inactive(
c24b5dfaSDave Chinner	xfs_inode_t	*ip)
c24b5dfaSDave Chinner{
3d3c8b52SJie Liu	struct xfs_mount	*mp;
c24b5dfaSDave Chinner	int			error;
c24b5dfaSDave Chinner	int			truncate = 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If the inode is already free, then there can be nothing
c24b5dfaSDave Chinner	 * to clean up here.
c24b5dfaSDave Chinner	 */
c19b3b05SDave Chinner	if (VFS_I(ip)->i_mode == 0) {
c24b5dfaSDave Chinner		ASSERT(ip->i_df.if_broot_bytes == 0);
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	mp = ip->i_mount;
17c12bcdSDarrick J. Wong	ASSERT(!xfs_iflags_test(ip, XFS_IRECOVERY));
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/* If this is a read-only mount, don't do this (would generate I/O) */
c24b5dfaSDave Chinner	if (mp->m_flags & XFS_MOUNT_RDONLY)
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner
383e32b0SDarrick J. Wong	/* Metadata inodes require explicit resource cleanup. */
383e32b0SDarrick J. Wong	if (xfs_is_metadata_inode(ip))
3ea06d73SDarrick J. Wong		goto out;
383e32b0SDarrick J. Wong
6231848cSDarrick J. Wong	/* Try to clean out the cow blocks if there are any. */
51d62690SChristoph Hellwig	if (xfs_inode_has_cow_data(ip))
6231848cSDarrick J. Wong		xfs_reflink_cancel_cow_range(ip, 0, NULLFILEOFF, true);
6231848cSDarrick J. Wong
54d7b5c1SDave Chinner	if (VFS_I(ip)->i_nlink != 0) {
c24b5dfaSDave Chinner		/*
c24b5dfaSDave Chinner		 * force is true because we are evicting an inode from the
c24b5dfaSDave Chinner		 * cache. Post-eof blocks must be freed, lest we end up with
c24b5dfaSDave Chinner		 * broken free space accounting.
3b4683c2SBrian Foster		 *
3b4683c2SBrian Foster		 * Note: don't bother with iolock here since lockdep complains
3b4683c2SBrian Foster		 * about acquiring it in reclaim context. We have the only
3b4683c2SBrian Foster		 * reference to the inode at this point anyways.
c24b5dfaSDave Chinner		 */
3b4683c2SBrian Foster		if (xfs_can_free_eofblocks(ip, true))
a36b9261SBrian Foster			xfs_free_eofblocks(ip);
74564fb4SBrian Foster
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c19b3b05SDave Chinner	if (S_ISREG(VFS_I(ip)->i_mode) &&
13d2c10bSChristoph Hellwig	    (ip->i_disk_size != 0 || XFS_ISIZE(ip) != 0 ||
daf83964SChristoph Hellwig	     ip->i_df.if_nextents > 0 || ip->i_delayed_blks > 0))
c24b5dfaSDave Chinner		truncate = 1;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(ip);
c24b5dfaSDave Chinner	if (error)
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner
c19b3b05SDave Chinner	if (S_ISLNK(VFS_I(ip)->i_mode))
36b21ddeSBrian Foster		error = xfs_inactive_symlink(ip);
f7be2d7fSBrian Foster	else if (truncate)
f7be2d7fSBrian Foster		error = xfs_inactive_truncate(ip);
36b21ddeSBrian Foster	if (error)
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If there are attributes associated with the file then blow them away
c24b5dfaSDave Chinner	 * now.  The code calls a routine that recursively deconstructs the
6dfe5a04SDave Chinner	 * attribute fork. If also blows away the in-core attribute fork.
c24b5dfaSDave Chinner	 */
6dfe5a04SDave Chinner	if (XFS_IFORK_Q(ip)) {
c24b5dfaSDave Chinner		error = xfs_attr_inactive(ip);
c24b5dfaSDave Chinner		if (error)
3ea06d73SDarrick J. Wong			goto out;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
6dfe5a04SDave Chinner	ASSERT(!ip->i_afp);
7821ea30SChristoph Hellwig	ASSERT(ip->i_forkoff == 0);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Free the inode.
c24b5dfaSDave Chinner	 */
3ea06d73SDarrick J. Wong	xfs_inactive_ifree(ip);
c24b5dfaSDave Chinner
3ea06d73SDarrick J. Wongout:
c24b5dfaSDave Chinner	/*
3ea06d73SDarrick J. Wong	 * We're done making metadata updates for this inode, so we can release
3ea06d73SDarrick J. Wong	 * the attached dquots.
c24b5dfaSDave Chinner	 */
c24b5dfaSDave Chinner	xfs_qm_dqdetach(ip);
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
1da177e4SLinus Torvalds/*
9b247179SDarrick J. Wong * In-Core Unlinked List Lookups
9b247179SDarrick J. Wong * =============================
9b247179SDarrick J. Wong *
9b247179SDarrick J. Wong * Every inode is supposed to be reachable from some other piece of metadata
9b247179SDarrick J. Wong * with the exception of the root directory.  Inodes with a connection to a
9b247179SDarrick J. Wong * file descriptor but not linked from anywhere in the on-disk directory tree
9b247179SDarrick J. Wong * are collectively known as unlinked inodes, though the filesystem itself
9b247179SDarrick J. Wong * maintains links to these inodes so that on-disk metadata are consistent.
9b247179SDarrick J. Wong *
9b247179SDarrick J. Wong * XFS implements a per-AG on-disk hash table of unlinked inodes.  The AGI
9b247179SDarrick J. Wong * header contains a number of buckets that point to an inode, and each inode
9b247179SDarrick J. Wong * record has a pointer to the next inode in the hash chain.  This
9b247179SDarrick J. Wong * singly-linked list causes scaling problems in the iunlink remove function
9b247179SDarrick J. Wong * because we must walk that list to find the inode that points to the inode
9b247179SDarrick J. Wong * being removed from the unlinked hash bucket list.
9b247179SDarrick J. Wong *
9b247179SDarrick J. Wong * What if we modelled the unlinked list as a collection of records capturing
9b247179SDarrick J. Wong * "X.next_unlinked = Y" relations?  If we indexed those records on Y, we'd
9b247179SDarrick J. Wong * have a fast way to look up unlinked list predecessors, which avoids the
9b247179SDarrick J. Wong * slow list walk.  That's exactly what we do here (in-core) with a per-AG
9b247179SDarrick J. Wong * rhashtable.
9b247179SDarrick J. Wong *
9b247179SDarrick J. Wong * Because this is a backref cache, we ignore operational failures since the
9b247179SDarrick J. Wong * iunlink code can fall back to the slow bucket walk.  The only errors that
9b247179SDarrick J. Wong * should bubble out are for obviously incorrect situations.
9b247179SDarrick J. Wong *
9b247179SDarrick J. Wong * All users of the backref cache MUST hold the AGI buffer lock to serialize
9b247179SDarrick J. Wong * access or have otherwise provided for concurrency control.
9b247179SDarrick J. Wong */
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/* Capture a "X.next_unlinked = Y" relationship. */
9b247179SDarrick J. Wongstruct xfs_iunlink {
9b247179SDarrick J. Wong	struct rhash_head	iu_rhash_head;
9b247179SDarrick J. Wong	xfs_agino_t		iu_agino;		/* X */
9b247179SDarrick J. Wong	xfs_agino_t		iu_next_unlinked;	/* Y */
9b247179SDarrick J. Wong};
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/* Unlinked list predecessor lookup hashtable construction */
9b247179SDarrick J. Wongstatic int
9b247179SDarrick J. Wongxfs_iunlink_obj_cmpfn(
9b247179SDarrick J. Wong	struct rhashtable_compare_arg	*arg,
9b247179SDarrick J. Wong	const void			*obj)
9b247179SDarrick J. Wong{
9b247179SDarrick J. Wong	const xfs_agino_t		*key = arg->key;
9b247179SDarrick J. Wong	const struct xfs_iunlink	*iu = obj;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	if (iu->iu_next_unlinked != *key)
9b247179SDarrick J. Wong		return 1;
9b247179SDarrick J. Wong	return 0;
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wongstatic const struct rhashtable_params xfs_iunlink_hash_params = {
9b247179SDarrick J. Wong	.min_size		= XFS_AGI_UNLINKED_BUCKETS,
9b247179SDarrick J. Wong	.key_len		= sizeof(xfs_agino_t),
9b247179SDarrick J. Wong	.key_offset		= offsetof(struct xfs_iunlink,
9b247179SDarrick J. Wong					   iu_next_unlinked),
9b247179SDarrick J. Wong	.head_offset		= offsetof(struct xfs_iunlink, iu_rhash_head),
9b247179SDarrick J. Wong	.automatic_shrinking	= true,
9b247179SDarrick J. Wong	.obj_cmpfn		= xfs_iunlink_obj_cmpfn,
9b247179SDarrick J. Wong};
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/*
9b247179SDarrick J. Wong * Return X, where X.next_unlinked == @agino.  Returns NULLAGINO if no such
9b247179SDarrick J. Wong * relation is found.
9b247179SDarrick J. Wong */
9b247179SDarrick J. Wongstatic xfs_agino_t
9b247179SDarrick J. Wongxfs_iunlink_lookup_backref(
9b247179SDarrick J. Wong	struct xfs_perag	*pag,
9b247179SDarrick J. Wong	xfs_agino_t		agino)
9b247179SDarrick J. Wong{
9b247179SDarrick J. Wong	struct xfs_iunlink	*iu;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	iu = rhashtable_lookup_fast(&pag->pagi_unlinked_hash, &agino,
9b247179SDarrick J. Wong			xfs_iunlink_hash_params);
9b247179SDarrick J. Wong	return iu ? iu->iu_agino : NULLAGINO;
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/*
9b247179SDarrick J. Wong * Take ownership of an iunlink cache entry and insert it into the hash table.
9b247179SDarrick J. Wong * If successful, the entry will be owned by the cache; if not, it is freed.
9b247179SDarrick J. Wong * Either way, the caller does not own @iu after this call.
9b247179SDarrick J. Wong */
9b247179SDarrick J. Wongstatic int
9b247179SDarrick J. Wongxfs_iunlink_insert_backref(
9b247179SDarrick J. Wong	struct xfs_perag	*pag,
9b247179SDarrick J. Wong	struct xfs_iunlink	*iu)
9b247179SDarrick J. Wong{
9b247179SDarrick J. Wong	int			error;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	error = rhashtable_insert_fast(&pag->pagi_unlinked_hash,
9b247179SDarrick J. Wong			&iu->iu_rhash_head, xfs_iunlink_hash_params);
9b247179SDarrick J. Wong	/*
9b247179SDarrick J. Wong	 * Fail loudly if there already was an entry because that's a sign of
9b247179SDarrick J. Wong	 * corruption of in-memory data.  Also fail loudly if we see an error
9b247179SDarrick J. Wong	 * code we didn't anticipate from the rhashtable code.  Currently we
9b247179SDarrick J. Wong	 * only anticipate ENOMEM.
9b247179SDarrick J. Wong	 */
9b247179SDarrick J. Wong	if (error) {
9b247179SDarrick J. Wong		WARN(error != -ENOMEM, "iunlink cache insert error %d", error);
9b247179SDarrick J. Wong		kmem_free(iu);
9b247179SDarrick J. Wong	}
9b247179SDarrick J. Wong	/*
9b247179SDarrick J. Wong	 * Absorb any runtime errors that aren't a result of corruption because
9b247179SDarrick J. Wong	 * this is a cache and we can always fall back to bucket list scanning.
9b247179SDarrick J. Wong	 */
9b247179SDarrick J. Wong	if (error != 0 && error != -EEXIST)
9b247179SDarrick J. Wong		error = 0;
9b247179SDarrick J. Wong	return error;
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/* Remember that @prev_agino.next_unlinked = @this_agino. */
9b247179SDarrick J. Wongstatic int
9b247179SDarrick J. Wongxfs_iunlink_add_backref(
9b247179SDarrick J. Wong	struct xfs_perag	*pag,
9b247179SDarrick J. Wong	xfs_agino_t		prev_agino,
9b247179SDarrick J. Wong	xfs_agino_t		this_agino)
9b247179SDarrick J. Wong{
9b247179SDarrick J. Wong	struct xfs_iunlink	*iu;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	if (XFS_TEST_ERROR(false, pag->pag_mount, XFS_ERRTAG_IUNLINK_FALLBACK))
9b247179SDarrick J. Wong		return 0;
9b247179SDarrick J. Wong
707e0ddaSTetsuo Handa	iu = kmem_zalloc(sizeof(*iu), KM_NOFS);
9b247179SDarrick J. Wong	iu->iu_agino = prev_agino;
9b247179SDarrick J. Wong	iu->iu_next_unlinked = this_agino;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	return xfs_iunlink_insert_backref(pag, iu);
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/*
9b247179SDarrick J. Wong * Replace X.next_unlinked = @agino with X.next_unlinked = @next_unlinked.
9b247179SDarrick J. Wong * If @next_unlinked is NULLAGINO, we drop the backref and exit.  If there
9b247179SDarrick J. Wong * wasn't any such entry then we don't bother.
9b247179SDarrick J. Wong */
9b247179SDarrick J. Wongstatic int
9b247179SDarrick J. Wongxfs_iunlink_change_backref(
9b247179SDarrick J. Wong	struct xfs_perag	*pag,
9b247179SDarrick J. Wong	xfs_agino_t		agino,
9b247179SDarrick J. Wong	xfs_agino_t		next_unlinked)
9b247179SDarrick J. Wong{
9b247179SDarrick J. Wong	struct xfs_iunlink	*iu;
9b247179SDarrick J. Wong	int			error;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	/* Look up the old entry; if there wasn't one then exit. */
9b247179SDarrick J. Wong	iu = rhashtable_lookup_fast(&pag->pagi_unlinked_hash, &agino,
9b247179SDarrick J. Wong			xfs_iunlink_hash_params);
9b247179SDarrick J. Wong	if (!iu)
9b247179SDarrick J. Wong		return 0;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	/*
9b247179SDarrick J. Wong	 * Remove the entry.  This shouldn't ever return an error, but if we
9b247179SDarrick J. Wong	 * couldn't remove the old entry we don't want to add it again to the
9b247179SDarrick J. Wong	 * hash table, and if the entry disappeared on us then someone's
9b247179SDarrick J. Wong	 * violated the locking rules and we need to fail loudly.  Either way
9b247179SDarrick J. Wong	 * we cannot remove the inode because internal state is or would have
9b247179SDarrick J. Wong	 * been corrupt.
9b247179SDarrick J. Wong	 */
9b247179SDarrick J. Wong	error = rhashtable_remove_fast(&pag->pagi_unlinked_hash,
9b247179SDarrick J. Wong			&iu->iu_rhash_head, xfs_iunlink_hash_params);
9b247179SDarrick J. Wong	if (error)
9b247179SDarrick J. Wong		return error;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	/* If there is no new next entry just free our item and return. */
9b247179SDarrick J. Wong	if (next_unlinked == NULLAGINO) {
9b247179SDarrick J. Wong		kmem_free(iu);
9b247179SDarrick J. Wong		return 0;
9b247179SDarrick J. Wong	}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	/* Update the entry and re-add it to the hash table. */
9b247179SDarrick J. Wong	iu->iu_next_unlinked = next_unlinked;
9b247179SDarrick J. Wong	return xfs_iunlink_insert_backref(pag, iu);
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/* Set up the in-core predecessor structures. */
9b247179SDarrick J. Wongint
9b247179SDarrick J. Wongxfs_iunlink_init(
9b247179SDarrick J. Wong	struct xfs_perag	*pag)
9b247179SDarrick J. Wong{
9b247179SDarrick J. Wong	return rhashtable_init(&pag->pagi_unlinked_hash,
9b247179SDarrick J. Wong			&xfs_iunlink_hash_params);
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/* Free the in-core predecessor structures. */
9b247179SDarrick J. Wongstatic void
9b247179SDarrick J. Wongxfs_iunlink_free_item(
9b247179SDarrick J. Wong	void			*ptr,
9b247179SDarrick J. Wong	void			*arg)
9b247179SDarrick J. Wong{
9b247179SDarrick J. Wong	struct xfs_iunlink	*iu = ptr;
9b247179SDarrick J. Wong	bool			*freed_anything = arg;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	*freed_anything = true;
9b247179SDarrick J. Wong	kmem_free(iu);
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wongvoid
9b247179SDarrick J. Wongxfs_iunlink_destroy(
9b247179SDarrick J. Wong	struct xfs_perag	*pag)
9b247179SDarrick J. Wong{
9b247179SDarrick J. Wong	bool			freed_anything = false;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	rhashtable_free_and_destroy(&pag->pagi_unlinked_hash,
9b247179SDarrick J. Wong			xfs_iunlink_free_item, &freed_anything);
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	ASSERT(freed_anything == false || XFS_FORCED_SHUTDOWN(pag->pag_mount));
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/*
9a4a5118SDarrick J. Wong * Point the AGI unlinked bucket at an inode and log the results.  The caller
9a4a5118SDarrick J. Wong * is responsible for validating the old value.
9a4a5118SDarrick J. Wong */
9a4a5118SDarrick J. WongSTATIC int
9a4a5118SDarrick J. Wongxfs_iunlink_update_bucket(
9a4a5118SDarrick J. Wong	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
9a4a5118SDarrick J. Wong	struct xfs_buf		*agibp,
9a4a5118SDarrick J. Wong	unsigned int		bucket_index,
9a4a5118SDarrick J. Wong	xfs_agino_t		new_agino)
9a4a5118SDarrick J. Wong{
370c782bSChristoph Hellwig	struct xfs_agi		*agi = agibp->b_addr;
9a4a5118SDarrick J. Wong	xfs_agino_t		old_value;
9a4a5118SDarrick J. Wong	int			offset;
9a4a5118SDarrick J. Wong
f40aadb2SDave Chinner	ASSERT(xfs_verify_agino_or_null(tp->t_mountp, pag->pag_agno, new_agino));
9a4a5118SDarrick J. Wong
9a4a5118SDarrick J. Wong	old_value = be32_to_cpu(agi->agi_unlinked[bucket_index]);
f40aadb2SDave Chinner	trace_xfs_iunlink_update_bucket(tp->t_mountp, pag->pag_agno, bucket_index,
9a4a5118SDarrick J. Wong			old_value, new_agino);
9a4a5118SDarrick J. Wong
9a4a5118SDarrick J. Wong	/*
9a4a5118SDarrick J. Wong	 * We should never find the head of the list already set to the value
9a4a5118SDarrick J. Wong	 * passed in because either we're adding or removing ourselves from the
9a4a5118SDarrick J. Wong	 * head of the list.
9a4a5118SDarrick J. Wong	 */
a5155b87SDarrick J. Wong	if (old_value == new_agino) {
8d57c216SDarrick J. Wong		xfs_buf_mark_corrupt(agibp);
9a4a5118SDarrick J. Wong		return -EFSCORRUPTED;
a5155b87SDarrick J. Wong	}
9a4a5118SDarrick J. Wong
9a4a5118SDarrick J. Wong	agi->agi_unlinked[bucket_index] = cpu_to_be32(new_agino);
9a4a5118SDarrick J. Wong	offset = offsetof(struct xfs_agi, agi_unlinked) +
9a4a5118SDarrick J. Wong			(sizeof(xfs_agino_t) * bucket_index);
9a4a5118SDarrick J. Wong	xfs_trans_log_buf(tp, agibp, offset, offset + sizeof(xfs_agino_t) - 1);
9a4a5118SDarrick J. Wong	return 0;
9a4a5118SDarrick J. Wong}
9a4a5118SDarrick J. Wong
f2fc16a3SDarrick J. Wong/* Set an on-disk inode's next_unlinked pointer. */
f2fc16a3SDarrick J. WongSTATIC void
f2fc16a3SDarrick J. Wongxfs_iunlink_update_dinode(
f2fc16a3SDarrick J. Wong	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
f2fc16a3SDarrick J. Wong	xfs_agino_t		agino,
f2fc16a3SDarrick J. Wong	struct xfs_buf		*ibp,
f2fc16a3SDarrick J. Wong	struct xfs_dinode	*dip,
f2fc16a3SDarrick J. Wong	struct xfs_imap		*imap,
f2fc16a3SDarrick J. Wong	xfs_agino_t		next_agino)
f2fc16a3SDarrick J. Wong{
f2fc16a3SDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
f2fc16a3SDarrick J. Wong	int			offset;
f2fc16a3SDarrick J. Wong
f40aadb2SDave Chinner	ASSERT(xfs_verify_agino_or_null(mp, pag->pag_agno, next_agino));
f2fc16a3SDarrick J. Wong
f40aadb2SDave Chinner	trace_xfs_iunlink_update_dinode(mp, pag->pag_agno, agino,
f2fc16a3SDarrick J. Wong			be32_to_cpu(dip->di_next_unlinked), next_agino);
f2fc16a3SDarrick J. Wong
f2fc16a3SDarrick J. Wong	dip->di_next_unlinked = cpu_to_be32(next_agino);
f2fc16a3SDarrick J. Wong	offset = imap->im_boffset +
f2fc16a3SDarrick J. Wong			offsetof(struct xfs_dinode, di_next_unlinked);
f2fc16a3SDarrick J. Wong
f2fc16a3SDarrick J. Wong	/* need to recalc the inode CRC if appropriate */
f2fc16a3SDarrick J. Wong	xfs_dinode_calc_crc(mp, dip);
f2fc16a3SDarrick J. Wong	xfs_trans_inode_buf(tp, ibp);
f2fc16a3SDarrick J. Wong	xfs_trans_log_buf(tp, ibp, offset, offset + sizeof(xfs_agino_t) - 1);
f2fc16a3SDarrick J. Wong}
f2fc16a3SDarrick J. Wong
f2fc16a3SDarrick J. Wong/* Set an in-core inode's unlinked pointer and return the old value. */
f2fc16a3SDarrick J. WongSTATIC int
f2fc16a3SDarrick J. Wongxfs_iunlink_update_inode(
f2fc16a3SDarrick J. Wong	struct xfs_trans	*tp,
f2fc16a3SDarrick J. Wong	struct xfs_inode	*ip,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
f2fc16a3SDarrick J. Wong	xfs_agino_t		next_agino,
f2fc16a3SDarrick J. Wong	xfs_agino_t		*old_next_agino)
f2fc16a3SDarrick J. Wong{
f2fc16a3SDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
f2fc16a3SDarrick J. Wong	struct xfs_dinode	*dip;
f2fc16a3SDarrick J. Wong	struct xfs_buf		*ibp;
f2fc16a3SDarrick J. Wong	xfs_agino_t		old_value;
f2fc16a3SDarrick J. Wong	int			error;
f2fc16a3SDarrick J. Wong
f40aadb2SDave Chinner	ASSERT(xfs_verify_agino_or_null(mp, pag->pag_agno, next_agino));
f2fc16a3SDarrick J. Wong
af9dcddeSChristoph Hellwig	error = xfs_imap_to_bp(mp, tp, &ip->i_imap, &ibp);
f2fc16a3SDarrick J. Wong	if (error)
f2fc16a3SDarrick J. Wong		return error;
af9dcddeSChristoph Hellwig	dip = xfs_buf_offset(ibp, ip->i_imap.im_boffset);
f2fc16a3SDarrick J. Wong
f2fc16a3SDarrick J. Wong	/* Make sure the old pointer isn't garbage. */
f2fc16a3SDarrick J. Wong	old_value = be32_to_cpu(dip->di_next_unlinked);
f40aadb2SDave Chinner	if (!xfs_verify_agino_or_null(mp, pag->pag_agno, old_value)) {
a5155b87SDarrick J. Wong		xfs_inode_verifier_error(ip, -EFSCORRUPTED, __func__, dip,
a5155b87SDarrick J. Wong				sizeof(*dip), __this_address);
f2fc16a3SDarrick J. Wong		error = -EFSCORRUPTED;
f2fc16a3SDarrick J. Wong		goto out;
f2fc16a3SDarrick J. Wong	}
f2fc16a3SDarrick J. Wong
f2fc16a3SDarrick J. Wong	/*
f2fc16a3SDarrick J. Wong	 * Since we're updating a linked list, we should never find that the
f2fc16a3SDarrick J. Wong	 * current pointer is the same as the new value, unless we're
f2fc16a3SDarrick J. Wong	 * terminating the list.
f2fc16a3SDarrick J. Wong	 */
f2fc16a3SDarrick J. Wong	*old_next_agino = old_value;
f2fc16a3SDarrick J. Wong	if (old_value == next_agino) {
a5155b87SDarrick J. Wong		if (next_agino != NULLAGINO) {
a5155b87SDarrick J. Wong			xfs_inode_verifier_error(ip, -EFSCORRUPTED, __func__,
a5155b87SDarrick J. Wong					dip, sizeof(*dip), __this_address);
f2fc16a3SDarrick J. Wong			error = -EFSCORRUPTED;
a5155b87SDarrick J. Wong		}
f2fc16a3SDarrick J. Wong		goto out;
f2fc16a3SDarrick J. Wong	}
f2fc16a3SDarrick J. Wong
f2fc16a3SDarrick J. Wong	/* Ok, update the new pointer. */
f40aadb2SDave Chinner	xfs_iunlink_update_dinode(tp, pag, XFS_INO_TO_AGINO(mp, ip->i_ino),
f2fc16a3SDarrick J. Wong			ibp, dip, &ip->i_imap, next_agino);
f2fc16a3SDarrick J. Wong	return 0;
f2fc16a3SDarrick J. Wongout:
f2fc16a3SDarrick J. Wong	xfs_trans_brelse(tp, ibp);
f2fc16a3SDarrick J. Wong	return error;
f2fc16a3SDarrick J. Wong}
f2fc16a3SDarrick J. Wong
9a4a5118SDarrick J. Wong/*
c4a6bf7fSDarrick J. Wong * This is called when the inode's link count has gone to 0 or we are creating
c4a6bf7fSDarrick J. Wong * a tmpfile via O_TMPFILE.  The inode @ip must have nlink == 0.
54d7b5c1SDave Chinner *
54d7b5c1SDave Chinner * We place the on-disk inode on a list in the AGI.  It will be pulled from this
54d7b5c1SDave Chinner * list when the inode is freed.
1da177e4SLinus Torvalds */
54d7b5c1SDave ChinnerSTATIC int
1da177e4SLinus Torvaldsxfs_iunlink(
54d7b5c1SDave Chinner	struct xfs_trans	*tp,
54d7b5c1SDave Chinner	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
5837f625SDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
f40aadb2SDave Chinner	struct xfs_perag	*pag;
5837f625SDarrick J. Wong	struct xfs_agi		*agi;
5837f625SDarrick J. Wong	struct xfs_buf		*agibp;
86bfd375SDarrick J. Wong	xfs_agino_t		next_agino;
5837f625SDarrick J. Wong	xfs_agino_t		agino = XFS_INO_TO_AGINO(mp, ip->i_ino);
5837f625SDarrick J. Wong	short			bucket_index = agino % XFS_AGI_UNLINKED_BUCKETS;
1da177e4SLinus Torvalds	int			error;
1da177e4SLinus Torvalds
c4a6bf7fSDarrick J. Wong	ASSERT(VFS_I(ip)->i_nlink == 0);
c19b3b05SDave Chinner	ASSERT(VFS_I(ip)->i_mode != 0);
4664c66cSDarrick J. Wong	trace_xfs_iunlink(ip);
1da177e4SLinus Torvalds
f40aadb2SDave Chinner	pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, ip->i_ino));
f40aadb2SDave Chinner
5837f625SDarrick J. Wong	/* Get the agi buffer first.  It ensures lock ordering on the list. */
f40aadb2SDave Chinner	error = xfs_read_agi(mp, tp, pag->pag_agno, &agibp);
859d7182SVlad Apostolov	if (error)
f40aadb2SDave Chinner		goto out;
370c782bSChristoph Hellwig	agi = agibp->b_addr;
5e1be0fbSChristoph Hellwig
1da177e4SLinus Torvalds	/*
86bfd375SDarrick J. Wong	 * Get the index into the agi hash table for the list this inode will
86bfd375SDarrick J. Wong	 * go on.  Make sure the pointer isn't garbage and that this inode
86bfd375SDarrick J. Wong	 * isn't already on the list.
1da177e4SLinus Torvalds	 */
86bfd375SDarrick J. Wong	next_agino = be32_to_cpu(agi->agi_unlinked[bucket_index]);
86bfd375SDarrick J. Wong	if (next_agino == agino ||
f40aadb2SDave Chinner	    !xfs_verify_agino_or_null(mp, pag->pag_agno, next_agino)) {
8d57c216SDarrick J. Wong		xfs_buf_mark_corrupt(agibp);
f40aadb2SDave Chinner		error = -EFSCORRUPTED;
f40aadb2SDave Chinner		goto out;
a5155b87SDarrick J. Wong	}
1da177e4SLinus Torvalds
86bfd375SDarrick J. Wong	if (next_agino != NULLAGINO) {
f2fc16a3SDarrick J. Wong		xfs_agino_t		old_agino;
f2fc16a3SDarrick J. Wong
1da177e4SLinus Torvalds		/*
f2fc16a3SDarrick J. Wong		 * There is already another inode in the bucket, so point this
f2fc16a3SDarrick J. Wong		 * inode to the current head of the list.
1da177e4SLinus Torvalds		 */
f40aadb2SDave Chinner		error = xfs_iunlink_update_inode(tp, ip, pag, next_agino,
f2fc16a3SDarrick J. Wong				&old_agino);
c319b58bSVlad Apostolov		if (error)
f40aadb2SDave Chinner			goto out;
f2fc16a3SDarrick J. Wong		ASSERT(old_agino == NULLAGINO);
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong		/*
9b247179SDarrick J. Wong		 * agino has been unlinked, add a backref from the next inode
9b247179SDarrick J. Wong		 * back to agino.
9b247179SDarrick J. Wong		 */
f40aadb2SDave Chinner		error = xfs_iunlink_add_backref(pag, agino, next_agino);
9b247179SDarrick J. Wong		if (error)
f40aadb2SDave Chinner			goto out;
1da177e4SLinus Torvalds	}
1da177e4SLinus Torvalds
9a4a5118SDarrick J. Wong	/* Point the head of the list to point to this inode. */
f40aadb2SDave Chinner	error = xfs_iunlink_update_bucket(tp, pag, agibp, bucket_index, agino);
f40aadb2SDave Chinnerout:
f40aadb2SDave Chinner	xfs_perag_put(pag);
f40aadb2SDave Chinner	return error;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
23ffa52cSDarrick J. Wong/* Return the imap, dinode pointer, and buffer for an inode. */
23ffa52cSDarrick J. WongSTATIC int
23ffa52cSDarrick J. Wongxfs_iunlink_map_ino(
23ffa52cSDarrick J. Wong	struct xfs_trans	*tp,
23ffa52cSDarrick J. Wong	xfs_agnumber_t		agno,
23ffa52cSDarrick J. Wong	xfs_agino_t		agino,
23ffa52cSDarrick J. Wong	struct xfs_imap		*imap,
23ffa52cSDarrick J. Wong	struct xfs_dinode	**dipp,
23ffa52cSDarrick J. Wong	struct xfs_buf		**bpp)
23ffa52cSDarrick J. Wong{
23ffa52cSDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
23ffa52cSDarrick J. Wong	int			error;
23ffa52cSDarrick J. Wong
23ffa52cSDarrick J. Wong	imap->im_blkno = 0;
23ffa52cSDarrick J. Wong	error = xfs_imap(mp, tp, XFS_AGINO_TO_INO(mp, agno, agino), imap, 0);
23ffa52cSDarrick J. Wong	if (error) {
23ffa52cSDarrick J. Wong		xfs_warn(mp, "%s: xfs_imap returned error %d.",
23ffa52cSDarrick J. Wong				__func__, error);
23ffa52cSDarrick J. Wong		return error;
23ffa52cSDarrick J. Wong	}
23ffa52cSDarrick J. Wong
af9dcddeSChristoph Hellwig	error = xfs_imap_to_bp(mp, tp, imap, bpp);
23ffa52cSDarrick J. Wong	if (error) {
23ffa52cSDarrick J. Wong		xfs_warn(mp, "%s: xfs_imap_to_bp returned error %d.",
23ffa52cSDarrick J. Wong				__func__, error);
23ffa52cSDarrick J. Wong		return error;
23ffa52cSDarrick J. Wong	}
23ffa52cSDarrick J. Wong
af9dcddeSChristoph Hellwig	*dipp = xfs_buf_offset(*bpp, imap->im_boffset);
23ffa52cSDarrick J. Wong	return 0;
23ffa52cSDarrick J. Wong}
23ffa52cSDarrick J. Wong
23ffa52cSDarrick J. Wong/*
23ffa52cSDarrick J. Wong * Walk the unlinked chain from @head_agino until we find the inode that
23ffa52cSDarrick J. Wong * points to @target_agino.  Return the inode number, map, dinode pointer,
23ffa52cSDarrick J. Wong * and inode cluster buffer of that inode as @agino, @imap, @dipp, and @bpp.
23ffa52cSDarrick J. Wong *
23ffa52cSDarrick J. Wong * @tp, @pag, @head_agino, and @target_agino are input parameters.
23ffa52cSDarrick J. Wong * @agino, @imap, @dipp, and @bpp are all output parameters.
23ffa52cSDarrick J. Wong *
23ffa52cSDarrick J. Wong * Do not call this function if @target_agino is the head of the list.
23ffa52cSDarrick J. Wong */
23ffa52cSDarrick J. WongSTATIC int
23ffa52cSDarrick J. Wongxfs_iunlink_map_prev(
23ffa52cSDarrick J. Wong	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
23ffa52cSDarrick J. Wong	xfs_agino_t		head_agino,
23ffa52cSDarrick J. Wong	xfs_agino_t		target_agino,
23ffa52cSDarrick J. Wong	xfs_agino_t		*agino,
23ffa52cSDarrick J. Wong	struct xfs_imap		*imap,
23ffa52cSDarrick J. Wong	struct xfs_dinode	**dipp,
f40aadb2SDave Chinner	struct xfs_buf		**bpp)
23ffa52cSDarrick J. Wong{
23ffa52cSDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
23ffa52cSDarrick J. Wong	xfs_agino_t		next_agino;
23ffa52cSDarrick J. Wong	int			error;
23ffa52cSDarrick J. Wong
23ffa52cSDarrick J. Wong	ASSERT(head_agino != target_agino);
23ffa52cSDarrick J. Wong	*bpp = NULL;
23ffa52cSDarrick J. Wong
9b247179SDarrick J. Wong	/* See if our backref cache can find it faster. */
9b247179SDarrick J. Wong	*agino = xfs_iunlink_lookup_backref(pag, target_agino);
9b247179SDarrick J. Wong	if (*agino != NULLAGINO) {
f40aadb2SDave Chinner		error = xfs_iunlink_map_ino(tp, pag->pag_agno, *agino, imap,
f40aadb2SDave Chinner				dipp, bpp);
9b247179SDarrick J. Wong		if (error)
9b247179SDarrick J. Wong			return error;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong		if (be32_to_cpu((*dipp)->di_next_unlinked) == target_agino)
9b247179SDarrick J. Wong			return 0;
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong		/*
9b247179SDarrick J. Wong		 * If we get here the cache contents were corrupt, so drop the
9b247179SDarrick J. Wong		 * buffer and fall back to walking the bucket list.
9b247179SDarrick J. Wong		 */
9b247179SDarrick J. Wong		xfs_trans_brelse(tp, *bpp);
9b247179SDarrick J. Wong		*bpp = NULL;
9b247179SDarrick J. Wong		WARN_ON_ONCE(1);
9b247179SDarrick J. Wong	}
9b247179SDarrick J. Wong
f40aadb2SDave Chinner	trace_xfs_iunlink_map_prev_fallback(mp, pag->pag_agno);
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	/* Otherwise, walk the entire bucket until we find it. */
23ffa52cSDarrick J. Wong	next_agino = head_agino;
23ffa52cSDarrick J. Wong	while (next_agino != target_agino) {
23ffa52cSDarrick J. Wong		xfs_agino_t	unlinked_agino;
23ffa52cSDarrick J. Wong
23ffa52cSDarrick J. Wong		if (*bpp)
23ffa52cSDarrick J. Wong			xfs_trans_brelse(tp, *bpp);
23ffa52cSDarrick J. Wong
23ffa52cSDarrick J. Wong		*agino = next_agino;
f40aadb2SDave Chinner		error = xfs_iunlink_map_ino(tp, pag->pag_agno, next_agino, imap,
f40aadb2SDave Chinner				dipp, bpp);
23ffa52cSDarrick J. Wong		if (error)
23ffa52cSDarrick J. Wong			return error;
23ffa52cSDarrick J. Wong
23ffa52cSDarrick J. Wong		unlinked_agino = be32_to_cpu((*dipp)->di_next_unlinked);
23ffa52cSDarrick J. Wong		/*
23ffa52cSDarrick J. Wong		 * Make sure this pointer is valid and isn't an obvious
23ffa52cSDarrick J. Wong		 * infinite loop.
23ffa52cSDarrick J. Wong		 */
f40aadb2SDave Chinner		if (!xfs_verify_agino(mp, pag->pag_agno, unlinked_agino) ||
23ffa52cSDarrick J. Wong		    next_agino == unlinked_agino) {
23ffa52cSDarrick J. Wong			XFS_CORRUPTION_ERROR(__func__,
23ffa52cSDarrick J. Wong					XFS_ERRLEVEL_LOW, mp,
23ffa52cSDarrick J. Wong					*dipp, sizeof(**dipp));
23ffa52cSDarrick J. Wong			error = -EFSCORRUPTED;
23ffa52cSDarrick J. Wong			return error;
23ffa52cSDarrick J. Wong		}
23ffa52cSDarrick J. Wong		next_agino = unlinked_agino;
23ffa52cSDarrick J. Wong	}
23ffa52cSDarrick J. Wong
23ffa52cSDarrick J. Wong	return 0;
23ffa52cSDarrick J. Wong}
23ffa52cSDarrick J. Wong
1da177e4SLinus Torvalds/*
1da177e4SLinus Torvalds * Pull the on-disk inode from the AGI unlinked list.
1da177e4SLinus Torvalds */
1da177e4SLinus TorvaldsSTATIC int
1da177e4SLinus Torvaldsxfs_iunlink_remove(
5837f625SDarrick J. Wong	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
5837f625SDarrick J. Wong	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
5837f625SDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
5837f625SDarrick J. Wong	struct xfs_agi		*agi;
5837f625SDarrick J. Wong	struct xfs_buf		*agibp;
5837f625SDarrick J. Wong	struct xfs_buf		*last_ibp;
5837f625SDarrick J. Wong	struct xfs_dinode	*last_dip = NULL;
5837f625SDarrick J. Wong	xfs_agino_t		agino = XFS_INO_TO_AGINO(mp, ip->i_ino);
1da177e4SLinus Torvalds	xfs_agino_t		next_agino;
b1d2a068SDarrick J. Wong	xfs_agino_t		head_agino;
5837f625SDarrick J. Wong	short			bucket_index = agino % XFS_AGI_UNLINKED_BUCKETS;
1da177e4SLinus Torvalds	int			error;
1da177e4SLinus Torvalds
4664c66cSDarrick J. Wong	trace_xfs_iunlink_remove(ip);
4664c66cSDarrick J. Wong
5837f625SDarrick J. Wong	/* Get the agi buffer first.  It ensures lock ordering on the list. */
f40aadb2SDave Chinner	error = xfs_read_agi(mp, tp, pag->pag_agno, &agibp);
5e1be0fbSChristoph Hellwig	if (error)
1da177e4SLinus Torvalds		return error;
370c782bSChristoph Hellwig	agi = agibp->b_addr;
5e1be0fbSChristoph Hellwig
1da177e4SLinus Torvalds	/*
86bfd375SDarrick J. Wong	 * Get the index into the agi hash table for the list this inode will
86bfd375SDarrick J. Wong	 * go on.  Make sure the head pointer isn't garbage.
1da177e4SLinus Torvalds	 */
b1d2a068SDarrick J. Wong	head_agino = be32_to_cpu(agi->agi_unlinked[bucket_index]);
f40aadb2SDave Chinner	if (!xfs_verify_agino(mp, pag->pag_agno, head_agino)) {
d2e73665SDarrick J. Wong		XFS_CORRUPTION_ERROR(__func__, XFS_ERRLEVEL_LOW, mp,
d2e73665SDarrick J. Wong				agi, sizeof(*agi));
d2e73665SDarrick J. Wong		return -EFSCORRUPTED;
d2e73665SDarrick J. Wong	}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
b1d2a068SDarrick J. Wong	 * Set our inode's next_unlinked pointer to NULL and then return
b1d2a068SDarrick J. Wong	 * the old pointer value so that we can update whatever was previous
b1d2a068SDarrick J. Wong	 * to us in the list to point to whatever was next in the list.
1da177e4SLinus Torvalds	 */
f40aadb2SDave Chinner	error = xfs_iunlink_update_inode(tp, ip, pag, NULLAGINO, &next_agino);
f2fc16a3SDarrick J. Wong	if (error)
1da177e4SLinus Torvalds		return error;
9a4a5118SDarrick J. Wong
9b247179SDarrick J. Wong	/*
9b247179SDarrick J. Wong	 * If there was a backref pointing from the next inode back to this
9b247179SDarrick J. Wong	 * one, remove it because we've removed this inode from the list.
9b247179SDarrick J. Wong	 *
9b247179SDarrick J. Wong	 * Later, if this inode was in the middle of the list we'll update
9b247179SDarrick J. Wong	 * this inode's backref to point from the next inode.
9b247179SDarrick J. Wong	 */
9b247179SDarrick J. Wong	if (next_agino != NULLAGINO) {
f40aadb2SDave Chinner		error = xfs_iunlink_change_backref(pag, next_agino, NULLAGINO);
9b247179SDarrick J. Wong		if (error)
92a00544SGao Xiang			return error;
9b247179SDarrick J. Wong	}
9b247179SDarrick J. Wong
92a00544SGao Xiang	if (head_agino != agino) {
f2fc16a3SDarrick J. Wong		struct xfs_imap	imap;
f2fc16a3SDarrick J. Wong		xfs_agino_t	prev_agino;
f2fc16a3SDarrick J. Wong
23ffa52cSDarrick J. Wong		/* We need to search the list for the inode being freed. */
f40aadb2SDave Chinner		error = xfs_iunlink_map_prev(tp, pag, head_agino, agino,
f40aadb2SDave Chinner				&prev_agino, &imap, &last_dip, &last_ibp);
23ffa52cSDarrick J. Wong		if (error)
92a00544SGao Xiang			return error;
475ee413SChristoph Hellwig
f2fc16a3SDarrick J. Wong		/* Point the previous inode on the list to the next inode. */
f40aadb2SDave Chinner		xfs_iunlink_update_dinode(tp, pag, prev_agino, last_ibp,
f2fc16a3SDarrick J. Wong				last_dip, &imap, next_agino);
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong		/*
9b247179SDarrick J. Wong		 * Now we deal with the backref for this inode.  If this inode
9b247179SDarrick J. Wong		 * pointed at a real inode, change the backref that pointed to
9b247179SDarrick J. Wong		 * us to point to our old next.  If this inode was the end of
9b247179SDarrick J. Wong		 * the list, delete the backref that pointed to us.  Note that
9b247179SDarrick J. Wong		 * change_backref takes care of deleting the backref if
9b247179SDarrick J. Wong		 * next_agino is NULLAGINO.
9b247179SDarrick J. Wong		 */
92a00544SGao Xiang		return xfs_iunlink_change_backref(agibp->b_pag, agino,
92a00544SGao Xiang				next_agino);
1da177e4SLinus Torvalds	}
9b247179SDarrick J. Wong
92a00544SGao Xiang	/* Point the head of the list to the next unlinked inode. */
f40aadb2SDave Chinner	return xfs_iunlink_update_bucket(tp, pag, agibp, bucket_index,
92a00544SGao Xiang			next_agino);
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
5b3eed75SDave Chinner/*
71e3e356SDave Chinner * Look up the inode number specified and if it is not already marked XFS_ISTALE
71e3e356SDave Chinner * mark it stale. We should only find clean inodes in this lookup that aren't
71e3e356SDave Chinner * already stale.
5806165aSDave Chinner */
71e3e356SDave Chinnerstatic void
71e3e356SDave Chinnerxfs_ifree_mark_inode_stale(
f40aadb2SDave Chinner	struct xfs_perag	*pag,
5806165aSDave Chinner	struct xfs_inode	*free_ip,
d9fdd0adSBrian Foster	xfs_ino_t		inum)
5806165aSDave Chinner{
f40aadb2SDave Chinner	struct xfs_mount	*mp = pag->pag_mount;
71e3e356SDave Chinner	struct xfs_inode_log_item *iip;
5806165aSDave Chinner	struct xfs_inode	*ip;
5806165aSDave Chinner
5806165aSDave Chinnerretry:
5806165aSDave Chinner	rcu_read_lock();
5806165aSDave Chinner	ip = radix_tree_lookup(&pag->pag_ici_root, XFS_INO_TO_AGINO(mp, inum));
5806165aSDave Chinner
5806165aSDave Chinner	/* Inode not in memory, nothing to do */
71e3e356SDave Chinner	if (!ip) {
71e3e356SDave Chinner		rcu_read_unlock();
71e3e356SDave Chinner		return;
71e3e356SDave Chinner	}
5806165aSDave Chinner
5806165aSDave Chinner	/*
5806165aSDave Chinner	 * because this is an RCU protected lookup, we could find a recently
5806165aSDave Chinner	 * freed or even reallocated inode during the lookup. We need to check
5806165aSDave Chinner	 * under the i_flags_lock for a valid inode here. Skip it if it is not
5806165aSDave Chinner	 * valid, the wrong inode or stale.
5806165aSDave Chinner	 */
5806165aSDave Chinner	spin_lock(&ip->i_flags_lock);
718ecc50SDave Chinner	if (ip->i_ino != inum || __xfs_iflags_test(ip, XFS_ISTALE))
718ecc50SDave Chinner		goto out_iflags_unlock;
5806165aSDave Chinner
5806165aSDave Chinner	/*
5806165aSDave Chinner	 * Don't try to lock/unlock the current inode, but we _cannot_ skip the
5806165aSDave Chinner	 * other inodes that we did not find in the list attached to the buffer
5806165aSDave Chinner	 * and are not already marked stale. If we can't lock it, back off and
5806165aSDave Chinner	 * retry.
5806165aSDave Chinner	 */
5806165aSDave Chinner	if (ip != free_ip) {
5806165aSDave Chinner		if (!xfs_ilock_nowait(ip, XFS_ILOCK_EXCL)) {
71e3e356SDave Chinner			spin_unlock(&ip->i_flags_lock);
5806165aSDave Chinner			rcu_read_unlock();
5806165aSDave Chinner			delay(1);
5806165aSDave Chinner			goto retry;
5806165aSDave Chinner		}
5806165aSDave Chinner	}
71e3e356SDave Chinner	ip->i_flags |= XFS_ISTALE;
5806165aSDave Chinner
71e3e356SDave Chinner	/*
718ecc50SDave Chinner	 * If the inode is flushing, it is already attached to the buffer.  All
71e3e356SDave Chinner	 * we needed to do here is mark the inode stale so buffer IO completion
71e3e356SDave Chinner	 * will remove it from the AIL.
71e3e356SDave Chinner	 */
71e3e356SDave Chinner	iip = ip->i_itemp;
718ecc50SDave Chinner	if (__xfs_iflags_test(ip, XFS_IFLUSHING)) {
71e3e356SDave Chinner		ASSERT(!list_empty(&iip->ili_item.li_bio_list));
71e3e356SDave Chinner		ASSERT(iip->ili_last_fields);
71e3e356SDave Chinner		goto out_iunlock;
71e3e356SDave Chinner	}
5806165aSDave Chinner
5806165aSDave Chinner	/*
48d55e2aSDave Chinner	 * Inodes not attached to the buffer can be released immediately.
48d55e2aSDave Chinner	 * Everything else has to go through xfs_iflush_abort() on journal
48d55e2aSDave Chinner	 * commit as the flock synchronises removal of the inode from the
48d55e2aSDave Chinner	 * cluster buffer against inode reclaim.
5806165aSDave Chinner	 */
718ecc50SDave Chinner	if (!iip || list_empty(&iip->ili_item.li_bio_list))
71e3e356SDave Chinner		goto out_iunlock;
718ecc50SDave Chinner
718ecc50SDave Chinner	__xfs_iflags_set(ip, XFS_IFLUSHING);
718ecc50SDave Chinner	spin_unlock(&ip->i_flags_lock);
718ecc50SDave Chinner	rcu_read_unlock();
5806165aSDave Chinner
71e3e356SDave Chinner	/* we have a dirty inode in memory that has not yet been flushed. */
71e3e356SDave Chinner	spin_lock(&iip->ili_lock);
71e3e356SDave Chinner	iip->ili_last_fields = iip->ili_fields;
71e3e356SDave Chinner	iip->ili_fields = 0;
71e3e356SDave Chinner	iip->ili_fsync_fields = 0;
71e3e356SDave Chinner	spin_unlock(&iip->ili_lock);
71e3e356SDave Chinner	ASSERT(iip->ili_last_fields);
71e3e356SDave Chinner
718ecc50SDave Chinner	if (ip != free_ip)
718ecc50SDave Chinner		xfs_iunlock(ip, XFS_ILOCK_EXCL);
718ecc50SDave Chinner	return;
718ecc50SDave Chinner
71e3e356SDave Chinnerout_iunlock:
71e3e356SDave Chinner	if (ip != free_ip)
71e3e356SDave Chinner		xfs_iunlock(ip, XFS_ILOCK_EXCL);
718ecc50SDave Chinnerout_iflags_unlock:
718ecc50SDave Chinner	spin_unlock(&ip->i_flags_lock);
718ecc50SDave Chinner	rcu_read_unlock();
5806165aSDave Chinner}
5806165aSDave Chinner
5806165aSDave Chinner/*
0b8182dbSZhi Yong Wu * A big issue when freeing the inode cluster is that we _cannot_ skip any
5b3eed75SDave Chinner * inodes that are in memory - they all must be marked stale and attached to
5b3eed75SDave Chinner * the cluster buffer.
5b3eed75SDave Chinner */
f40aadb2SDave Chinnerstatic int
1da177e4SLinus Torvaldsxfs_ifree_cluster(
71e3e356SDave Chinner	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
f40aadb2SDave Chinner	struct xfs_inode	*free_ip,
09b56604SBrian Foster	struct xfs_icluster	*xic)
1da177e4SLinus Torvalds{
71e3e356SDave Chinner	struct xfs_mount	*mp = free_ip->i_mount;
71e3e356SDave Chinner	struct xfs_ino_geometry	*igeo = M_IGEO(mp);
71e3e356SDave Chinner	struct xfs_buf		*bp;
71e3e356SDave Chinner	xfs_daddr_t		blkno;
71e3e356SDave Chinner	xfs_ino_t		inum = xic->first_ino;
1da177e4SLinus Torvalds	int			nbufs;
5b257b4aSDave Chinner	int			i, j;
3cdaa189SBrian Foster	int			ioffset;
ce92464cSDarrick J. Wong	int			error;
1da177e4SLinus Torvalds
ef325959SDarrick J. Wong	nbufs = igeo->ialloc_blks / igeo->blocks_per_cluster;
1da177e4SLinus Torvalds
ef325959SDarrick J. Wong	for (j = 0; j < nbufs; j++, inum += igeo->inodes_per_cluster) {
09b56604SBrian Foster		/*
09b56604SBrian Foster		 * The allocation bitmap tells us which inodes of the chunk were
09b56604SBrian Foster		 * physically allocated. Skip the cluster if an inode falls into
09b56604SBrian Foster		 * a sparse region.
09b56604SBrian Foster		 */
3cdaa189SBrian Foster		ioffset = inum - xic->first_ino;
3cdaa189SBrian Foster		if ((xic->alloc & XFS_INOBT_MASK(ioffset)) == 0) {
ef325959SDarrick J. Wong			ASSERT(ioffset % igeo->inodes_per_cluster == 0);
09b56604SBrian Foster			continue;
09b56604SBrian Foster		}
09b56604SBrian Foster
1da177e4SLinus Torvalds		blkno = XFS_AGB_TO_DADDR(mp, XFS_INO_TO_AGNO(mp, inum),
1da177e4SLinus Torvalds					 XFS_INO_TO_AGBNO(mp, inum));
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds		/*
5b257b4aSDave Chinner		 * We obtain and lock the backing buffer first in the process
718ecc50SDave Chinner		 * here to ensure dirty inodes attached to the buffer remain in
718ecc50SDave Chinner		 * the flushing state while we mark them stale.
718ecc50SDave Chinner		 *
5b257b4aSDave Chinner		 * If we scan the in-memory inodes first, then buffer IO can
5b257b4aSDave Chinner		 * complete before we get a lock on it, and hence we may fail
5b257b4aSDave Chinner		 * to mark all the active inodes on the buffer stale.
1da177e4SLinus Torvalds		 */
ce92464cSDarrick J. Wong		error = xfs_trans_get_buf(tp, mp->m_ddev_targp, blkno,
ef325959SDarrick J. Wong				mp->m_bsize * igeo->blocks_per_cluster,
ce92464cSDarrick J. Wong				XBF_UNMAPPED, &bp);
71e3e356SDave Chinner		if (error)
ce92464cSDarrick J. Wong			return error;
b0f539deSDave Chinner
b0f539deSDave Chinner		/*
b0f539deSDave Chinner		 * This buffer may not have been correctly initialised as we
b0f539deSDave Chinner		 * didn't read it from disk. That's not important because we are
b0f539deSDave Chinner		 * only using to mark the buffer as stale in the log, and to
b0f539deSDave Chinner		 * attach stale cached inodes on it. That means it will never be
b0f539deSDave Chinner		 * dispatched for IO. If it is, we want to know about it, and we
b0f539deSDave Chinner		 * want it to fail. We can acheive this by adding a write
b0f539deSDave Chinner		 * verifier to the buffer.
b0f539deSDave Chinner		 */
1813dd64SDave Chinner		bp->b_ops = &xfs_inode_buf_ops;
b0f539deSDave Chinner
5b257b4aSDave Chinner		/*
71e3e356SDave Chinner		 * Now we need to set all the cached clean inodes as XFS_ISTALE,
71e3e356SDave Chinner		 * too. This requires lookups, and will skip inodes that we've
71e3e356SDave Chinner		 * already marked XFS_ISTALE.
5b257b4aSDave Chinner		 */
71e3e356SDave Chinner		for (i = 0; i < igeo->inodes_per_cluster; i++)
f40aadb2SDave Chinner			xfs_ifree_mark_inode_stale(pag, free_ip, inum + i);
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds		xfs_trans_stale_inode_buf(tp, bp);
1da177e4SLinus Torvalds		xfs_trans_binval(tp, bp);
1da177e4SLinus Torvalds	}
2a30f36dSChandra Seetharaman	return 0;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
1da177e4SLinus Torvalds * This is called to return an inode to the inode free list.
1da177e4SLinus Torvalds * The inode should already be truncated to 0 length and have
1da177e4SLinus Torvalds * no pages associated with it.  This routine also assumes that
1da177e4SLinus Torvalds * the inode is already a part of the transaction.
1da177e4SLinus Torvalds *
1da177e4SLinus Torvalds * The on-disk copy of the inode will have been added to the list
1da177e4SLinus Torvalds * of unlinked inodes in the AGI. We need to remove the inode from
1da177e4SLinus Torvalds * that list atomically with respect to freeing it here.
1da177e4SLinus Torvalds */
1da177e4SLinus Torvaldsint
1da177e4SLinus Torvaldsxfs_ifree(
0e0417f3SBrian Foster	struct xfs_trans	*tp,
0e0417f3SBrian Foster	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
f40aadb2SDave Chinner	struct xfs_mount	*mp = ip->i_mount;
f40aadb2SDave Chinner	struct xfs_perag	*pag;
09b56604SBrian Foster	struct xfs_icluster	xic = { 0 };
1319ebefSDave Chinner	struct xfs_inode_log_item *iip = ip->i_itemp;
f40aadb2SDave Chinner	int			error;
1da177e4SLinus Torvalds
579aa9caSChristoph Hellwig	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL));
54d7b5c1SDave Chinner	ASSERT(VFS_I(ip)->i_nlink == 0);
daf83964SChristoph Hellwig	ASSERT(ip->i_df.if_nextents == 0);
13d2c10bSChristoph Hellwig	ASSERT(ip->i_disk_size == 0 || !S_ISREG(VFS_I(ip)->i_mode));
6e73a545SChristoph Hellwig	ASSERT(ip->i_nblocks == 0);
1da177e4SLinus Torvalds
f40aadb2SDave Chinner	pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, ip->i_ino));
f40aadb2SDave Chinner
1da177e4SLinus Torvalds	/*
1da177e4SLinus Torvalds	 * Pull the on-disk inode from the AGI unlinked list.
1da177e4SLinus Torvalds	 */
f40aadb2SDave Chinner	error = xfs_iunlink_remove(tp, pag, ip);
1baaed8fSDave Chinner	if (error)
f40aadb2SDave Chinner		goto out;
1da177e4SLinus Torvalds
f40aadb2SDave Chinner	error = xfs_difree(tp, pag, ip->i_ino, &xic);
1baaed8fSDave Chinner	if (error)
f40aadb2SDave Chinner		goto out;
1baaed8fSDave Chinner
b2c20045SChristoph Hellwig	/*
b2c20045SChristoph Hellwig	 * Free any local-format data sitting around before we reset the
b2c20045SChristoph Hellwig	 * data fork to extents format.  Note that the attr fork data has
b2c20045SChristoph Hellwig	 * already been freed by xfs_attr_inactive.
b2c20045SChristoph Hellwig	 */
f7e67b20SChristoph Hellwig	if (ip->i_df.if_format == XFS_DINODE_FMT_LOCAL) {
b2c20045SChristoph Hellwig		kmem_free(ip->i_df.if_u1.if_data);
b2c20045SChristoph Hellwig		ip->i_df.if_u1.if_data = NULL;
b2c20045SChristoph Hellwig		ip->i_df.if_bytes = 0;
b2c20045SChristoph Hellwig	}
98c4f78dSDarrick J. Wong
c19b3b05SDave Chinner	VFS_I(ip)->i_mode = 0;		/* mark incore inode as free */
db07349dSChristoph Hellwig	ip->i_diflags = 0;
f40aadb2SDave Chinner	ip->i_diflags2 = mp->m_ino_geo.new_diflags2;
7821ea30SChristoph Hellwig	ip->i_forkoff = 0;		/* mark the attr fork not in use */
f7e67b20SChristoph Hellwig	ip->i_df.if_format = XFS_DINODE_FMT_EXTENTS;
9b3beb02SChristoph Hellwig	if (xfs_iflags_test(ip, XFS_IPRESERVE_DM_FIELDS))
9b3beb02SChristoph Hellwig		xfs_iflags_clear(ip, XFS_IPRESERVE_DM_FIELDS);
dc1baa71SEric Sandeen
dc1baa71SEric Sandeen	/* Don't attempt to replay owner changes for a deleted inode */
1319ebefSDave Chinner	spin_lock(&iip->ili_lock);
1319ebefSDave Chinner	iip->ili_fields &= ~(XFS_ILOG_AOWNER | XFS_ILOG_DOWNER);
1319ebefSDave Chinner	spin_unlock(&iip->ili_lock);
dc1baa71SEric Sandeen
1da177e4SLinus Torvalds	/*
1da177e4SLinus Torvalds	 * Bump the generation count so no one will be confused
1da177e4SLinus Torvalds	 * by reincarnations of this inode.
1da177e4SLinus Torvalds	 */
9e9a2674SDave Chinner	VFS_I(ip)->i_generation++;
1da177e4SLinus Torvalds	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
1da177e4SLinus Torvalds
09b56604SBrian Foster	if (xic.deleted)
f40aadb2SDave Chinner		error = xfs_ifree_cluster(tp, pag, ip, &xic);
f40aadb2SDave Chinnerout:
f40aadb2SDave Chinner	xfs_perag_put(pag);
2a30f36dSChandra Seetharaman	return error;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
60ec6783SChristoph Hellwig * This is called to unpin an inode.  The caller must have the inode locked
60ec6783SChristoph Hellwig * in at least shared mode so that the buffer cannot be subsequently pinned
60ec6783SChristoph Hellwig * once someone is waiting for it to be unpinned.
1da177e4SLinus Torvalds */
60ec6783SChristoph Hellwigstatic void
f392e631SChristoph Hellwigxfs_iunpin(
60ec6783SChristoph Hellwig	struct xfs_inode	*ip)
a3f74ffbSDavid Chinner{
579aa9caSChristoph Hellwig	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL|XFS_ILOCK_SHARED));
a3f74ffbSDavid Chinner
4aaf15d1SDave Chinner	trace_xfs_inode_unpin_nowait(ip, _RET_IP_);
4aaf15d1SDave Chinner
a3f74ffbSDavid Chinner	/* Give the log a push to start the unpinning I/O */
5f9b4b0dSDave Chinner	xfs_log_force_seq(ip->i_mount, ip->i_itemp->ili_commit_seq, 0, NULL);
a14a348bSChristoph Hellwig
a3f74ffbSDavid Chinner}
a3f74ffbSDavid Chinner
f392e631SChristoph Hellwigstatic void
f392e631SChristoph Hellwig__xfs_iunpin_wait(
f392e631SChristoph Hellwig	struct xfs_inode	*ip)
f392e631SChristoph Hellwig{
f392e631SChristoph Hellwig	wait_queue_head_t *wq = bit_waitqueue(&ip->i_flags, __XFS_IPINNED_BIT);
f392e631SChristoph Hellwig	DEFINE_WAIT_BIT(wait, &ip->i_flags, __XFS_IPINNED_BIT);
f392e631SChristoph Hellwig
f392e631SChristoph Hellwig	xfs_iunpin(ip);
f392e631SChristoph Hellwig
f392e631SChristoph Hellwig	do {
21417136SIngo Molnar		prepare_to_wait(wq, &wait.wq_entry, TASK_UNINTERRUPTIBLE);
f392e631SChristoph Hellwig		if (xfs_ipincount(ip))
f392e631SChristoph Hellwig			io_schedule();
f392e631SChristoph Hellwig	} while (xfs_ipincount(ip));
21417136SIngo Molnar	finish_wait(wq, &wait.wq_entry);
f392e631SChristoph Hellwig}
f392e631SChristoph Hellwig
777df5afSDave Chinnervoid
1da177e4SLinus Torvaldsxfs_iunpin_wait(
60ec6783SChristoph Hellwig	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
f392e631SChristoph Hellwig	if (xfs_ipincount(ip))
f392e631SChristoph Hellwig		__xfs_iunpin_wait(ip);
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
27320369SDave Chinner/*
27320369SDave Chinner * Removing an inode from the namespace involves removing the directory entry
27320369SDave Chinner * and dropping the link count on the inode. Removing the directory entry can
27320369SDave Chinner * result in locking an AGF (directory blocks were freed) and removing a link
27320369SDave Chinner * count can result in placing the inode on an unlinked list which results in
27320369SDave Chinner * locking an AGI.
27320369SDave Chinner *
27320369SDave Chinner * The big problem here is that we have an ordering constraint on AGF and AGI
27320369SDave Chinner * locking - inode allocation locks the AGI, then can allocate a new extent for
27320369SDave Chinner * new inodes, locking the AGF after the AGI. Similarly, freeing the inode
27320369SDave Chinner * removes the inode from the unlinked list, requiring that we lock the AGI
27320369SDave Chinner * first, and then freeing the inode can result in an inode chunk being freed
27320369SDave Chinner * and hence freeing disk space requiring that we lock an AGF.
27320369SDave Chinner *
27320369SDave Chinner * Hence the ordering that is imposed by other parts of the code is AGI before
27320369SDave Chinner * AGF. This means we cannot remove the directory entry before we drop the inode
27320369SDave Chinner * reference count and put it on the unlinked list as this results in a lock
27320369SDave Chinner * order of AGF then AGI, and this can deadlock against inode allocation and
27320369SDave Chinner * freeing. Therefore we must drop the link counts before we remove the
27320369SDave Chinner * directory entry.
27320369SDave Chinner *
27320369SDave Chinner * This is still safe from a transactional point of view - it is not until we
310a75a3SDarrick J. Wong * get to xfs_defer_finish() that we have the possibility of multiple
27320369SDave Chinner * transactions in this operation. Hence as long as we remove the directory
27320369SDave Chinner * entry and drop the link count in the first transaction of the remove
27320369SDave Chinner * operation, there are no transactional constraints on the ordering here.
27320369SDave Chinner */
c24b5dfaSDave Chinnerint
c24b5dfaSDave Chinnerxfs_remove(
c24b5dfaSDave Chinner	xfs_inode_t             *dp,
c24b5dfaSDave Chinner	struct xfs_name		*name,
c24b5dfaSDave Chinner	xfs_inode_t		*ip)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	xfs_mount_t		*mp = dp->i_mount;
c24b5dfaSDave Chinner	xfs_trans_t             *tp = NULL;
c19b3b05SDave Chinner	int			is_dir = S_ISDIR(VFS_I(ip)->i_mode);
c24b5dfaSDave Chinner	int                     error = 0;
c24b5dfaSDave Chinner	uint			resblks;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	trace_xfs_remove(dp, name);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (XFS_FORCED_SHUTDOWN(mp))
2451337dSDave Chinner		return -EIO;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(dp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(ip);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * We try to get the real space reservation first,
c24b5dfaSDave Chinner	 * allowing for directory btree deletion(s) implying
c24b5dfaSDave Chinner	 * possible bmap insert(s).  If we can't get the space
c24b5dfaSDave Chinner	 * reservation then we use 0 instead, and avoid the bmap
c24b5dfaSDave Chinner	 * btree insert(s) in the directory code by, if the bmap
c24b5dfaSDave Chinner	 * insert tries to happen, instead trimming the LAST
c24b5dfaSDave Chinner	 * block from the directory.
c24b5dfaSDave Chinner	 */
c24b5dfaSDave Chinner	resblks = XFS_REMOVE_SPACE_RES(mp);
253f4911SChristoph Hellwig	error = xfs_trans_alloc(mp, &M_RES(mp)->tr_remove, resblks, 0, 0, &tp);
2451337dSDave Chinner	if (error == -ENOSPC) {
c24b5dfaSDave Chinner		resblks = 0;
253f4911SChristoph Hellwig		error = xfs_trans_alloc(mp, &M_RES(mp)->tr_remove, 0, 0, 0,
253f4911SChristoph Hellwig				&tp);
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner	if (error) {
2451337dSDave Chinner		ASSERT(error != -ENOSPC);
253f4911SChristoph Hellwig		goto std_return;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
7c2d238aSDarrick J. Wong	xfs_lock_two_inodes(dp, XFS_ILOCK_EXCL, ip, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner
65523218SChristoph Hellwig	xfs_trans_ijoin(tp, dp, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner	xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If we're removing a directory perform some additional validation.
c24b5dfaSDave Chinner	 */
c24b5dfaSDave Chinner	if (is_dir) {
54d7b5c1SDave Chinner		ASSERT(VFS_I(ip)->i_nlink >= 2);
54d7b5c1SDave Chinner		if (VFS_I(ip)->i_nlink != 2) {
2451337dSDave Chinner			error = -ENOTEMPTY;
c24b5dfaSDave Chinner			goto out_trans_cancel;
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner		if (!xfs_dir_isempty(ip)) {
2451337dSDave Chinner			error = -ENOTEMPTY;
c24b5dfaSDave Chinner			goto out_trans_cancel;
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner
27320369SDave Chinner		/* Drop the link from ip's "..".  */
c24b5dfaSDave Chinner		error = xfs_droplink(tp, dp);
c24b5dfaSDave Chinner		if (error)
27320369SDave Chinner			goto out_trans_cancel;
c24b5dfaSDave Chinner
27320369SDave Chinner		/* Drop the "." link from ip to self.  */
c24b5dfaSDave Chinner		error = xfs_droplink(tp, ip);
c24b5dfaSDave Chinner		if (error)
27320369SDave Chinner			goto out_trans_cancel;
5838d035SDarrick J. Wong
5838d035SDarrick J. Wong		/*
5838d035SDarrick J. Wong		 * Point the unlinked child directory's ".." entry to the root
5838d035SDarrick J. Wong		 * directory to eliminate back-references to inodes that may
5838d035SDarrick J. Wong		 * get freed before the child directory is closed.  If the fs
5838d035SDarrick J. Wong		 * gets shrunk, this can lead to dirent inode validation errors.
5838d035SDarrick J. Wong		 */
5838d035SDarrick J. Wong		if (dp->i_ino != tp->t_mountp->m_sb.sb_rootino) {
5838d035SDarrick J. Wong			error = xfs_dir_replace(tp, ip, &xfs_name_dotdot,
5838d035SDarrick J. Wong					tp->t_mountp->m_sb.sb_rootino, 0);
5838d035SDarrick J. Wong			if (error)
5838d035SDarrick J. Wong				return error;
5838d035SDarrick J. Wong		}
c24b5dfaSDave Chinner	} else {
c24b5dfaSDave Chinner		/*
c24b5dfaSDave Chinner		 * When removing a non-directory we need to log the parent
c24b5dfaSDave Chinner		 * inode here.  For a directory this is done implicitly
c24b5dfaSDave Chinner		 * by the xfs_droplink call for the ".." entry.
c24b5dfaSDave Chinner		 */
c24b5dfaSDave Chinner		xfs_trans_log_inode(tp, dp, XFS_ILOG_CORE);
c24b5dfaSDave Chinner	}
27320369SDave Chinner	xfs_trans_ichgtime(tp, dp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
c24b5dfaSDave Chinner
27320369SDave Chinner	/* Drop the link from dp to ip. */
c24b5dfaSDave Chinner	error = xfs_droplink(tp, ip);
c24b5dfaSDave Chinner	if (error)
27320369SDave Chinner		goto out_trans_cancel;
c24b5dfaSDave Chinner
381eee69SBrian Foster	error = xfs_dir_removename(tp, dp, name, ip->i_ino, resblks);
27320369SDave Chinner	if (error) {
2451337dSDave Chinner		ASSERT(error != -ENOENT);
c8eac49eSBrian Foster		goto out_trans_cancel;
27320369SDave Chinner	}
27320369SDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If this is a synchronous mount, make sure that the
c24b5dfaSDave Chinner	 * remove transaction goes to disk before returning to
c24b5dfaSDave Chinner	 * the user.
c24b5dfaSDave Chinner	 */
*0560f31aSDave Chinner	if (xfs_has_wsync(mp) || xfs_has_dirsync(mp))
c24b5dfaSDave Chinner		xfs_trans_set_sync(tp);
c24b5dfaSDave Chinner
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
2cd2ef6aSChristoph Hellwig	if (is_dir && xfs_inode_is_filestream(ip))
c24b5dfaSDave Chinner		xfs_filestream_deassociate(ip);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner out_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
c24b5dfaSDave Chinner std_return:
c24b5dfaSDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
f6bba201SDave Chinner/*
f6bba201SDave Chinner * Enter all inodes for a rename transaction into a sorted array.
f6bba201SDave Chinner */
95afcf5cSDave Chinner#define __XFS_SORT_INODES	5
f6bba201SDave ChinnerSTATIC void
f6bba201SDave Chinnerxfs_sort_for_rename(
95afcf5cSDave Chinner	struct xfs_inode	*dp1,	/* in: old (source) directory inode */
95afcf5cSDave Chinner	struct xfs_inode	*dp2,	/* in: new (target) directory inode */
95afcf5cSDave Chinner	struct xfs_inode	*ip1,	/* in: inode of old entry */
95afcf5cSDave Chinner	struct xfs_inode	*ip2,	/* in: inode of new entry */
95afcf5cSDave Chinner	struct xfs_inode	*wip,	/* in: whiteout inode */
95afcf5cSDave Chinner	struct xfs_inode	**i_tab,/* out: sorted array of inodes */
95afcf5cSDave Chinner	int			*num_inodes)  /* in/out: inodes in array */
f6bba201SDave Chinner{
f6bba201SDave Chinner	int			i, j;
f6bba201SDave Chinner
95afcf5cSDave Chinner	ASSERT(*num_inodes == __XFS_SORT_INODES);
95afcf5cSDave Chinner	memset(i_tab, 0, *num_inodes * sizeof(struct xfs_inode *));
95afcf5cSDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * i_tab contains a list of pointers to inodes.  We initialize
f6bba201SDave Chinner	 * the table here & we'll sort it.  We will then use it to
f6bba201SDave Chinner	 * order the acquisition of the inode locks.
f6bba201SDave Chinner	 *
f6bba201SDave Chinner	 * Note that the table may contain duplicates.  e.g., dp1 == dp2.
f6bba201SDave Chinner	 */
95afcf5cSDave Chinner	i = 0;
95afcf5cSDave Chinner	i_tab[i++] = dp1;
95afcf5cSDave Chinner	i_tab[i++] = dp2;
95afcf5cSDave Chinner	i_tab[i++] = ip1;
95afcf5cSDave Chinner	if (ip2)
95afcf5cSDave Chinner		i_tab[i++] = ip2;
95afcf5cSDave Chinner	if (wip)
95afcf5cSDave Chinner		i_tab[i++] = wip;
95afcf5cSDave Chinner	*num_inodes = i;
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Sort the elements via bubble sort.  (Remember, there are at
95afcf5cSDave Chinner	 * most 5 elements to sort, so this is adequate.)
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	for (i = 0; i < *num_inodes; i++) {
f6bba201SDave Chinner		for (j = 1; j < *num_inodes; j++) {
f6bba201SDave Chinner			if (i_tab[j]->i_ino < i_tab[j-1]->i_ino) {
95afcf5cSDave Chinner				struct xfs_inode *temp = i_tab[j];
f6bba201SDave Chinner				i_tab[j] = i_tab[j-1];
f6bba201SDave Chinner				i_tab[j-1] = temp;
f6bba201SDave Chinner			}
f6bba201SDave Chinner		}
f6bba201SDave Chinner	}
f6bba201SDave Chinner}
f6bba201SDave Chinner
310606b0SDave Chinnerstatic int
310606b0SDave Chinnerxfs_finish_rename(
c9cfdb38SBrian Foster	struct xfs_trans	*tp)
310606b0SDave Chinner{
310606b0SDave Chinner	/*
310606b0SDave Chinner	 * If this is a synchronous mount, make sure that the rename transaction
310606b0SDave Chinner	 * goes to disk before returning to the user.
310606b0SDave Chinner	 */
*0560f31aSDave Chinner	if (xfs_has_wsync(tp->t_mountp) || xfs_has_dirsync(tp->t_mountp))
310606b0SDave Chinner		xfs_trans_set_sync(tp);
310606b0SDave Chinner
70393313SChristoph Hellwig	return xfs_trans_commit(tp);
310606b0SDave Chinner}
310606b0SDave Chinner
f6bba201SDave Chinner/*
d31a1825SCarlos Maiolino * xfs_cross_rename()
d31a1825SCarlos Maiolino *
0145225eSBhaskar Chowdhury * responsible for handling RENAME_EXCHANGE flag in renameat2() syscall
d31a1825SCarlos Maiolino */
d31a1825SCarlos MaiolinoSTATIC int
d31a1825SCarlos Maiolinoxfs_cross_rename(
d31a1825SCarlos Maiolino	struct xfs_trans	*tp,
d31a1825SCarlos Maiolino	struct xfs_inode	*dp1,
d31a1825SCarlos Maiolino	struct xfs_name		*name1,
d31a1825SCarlos Maiolino	struct xfs_inode	*ip1,
d31a1825SCarlos Maiolino	struct xfs_inode	*dp2,
d31a1825SCarlos Maiolino	struct xfs_name		*name2,
d31a1825SCarlos Maiolino	struct xfs_inode	*ip2,
d31a1825SCarlos Maiolino	int			spaceres)
d31a1825SCarlos Maiolino{
d31a1825SCarlos Maiolino	int		error = 0;
d31a1825SCarlos Maiolino	int		ip1_flags = 0;
d31a1825SCarlos Maiolino	int		ip2_flags = 0;
d31a1825SCarlos Maiolino	int		dp2_flags = 0;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	/* Swap inode number for dirent in first parent */
381eee69SBrian Foster	error = xfs_dir_replace(tp, dp1, name1, ip2->i_ino, spaceres);
d31a1825SCarlos Maiolino	if (error)
eeacd321SDave Chinner		goto out_trans_abort;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	/* Swap inode number for dirent in second parent */
381eee69SBrian Foster	error = xfs_dir_replace(tp, dp2, name2, ip1->i_ino, spaceres);
d31a1825SCarlos Maiolino	if (error)
eeacd321SDave Chinner		goto out_trans_abort;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	/*
d31a1825SCarlos Maiolino	 * If we're renaming one or more directories across different parents,
d31a1825SCarlos Maiolino	 * update the respective ".." entries (and link counts) to match the new
d31a1825SCarlos Maiolino	 * parents.
d31a1825SCarlos Maiolino	 */
d31a1825SCarlos Maiolino	if (dp1 != dp2) {
d31a1825SCarlos Maiolino		dp2_flags = XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino
c19b3b05SDave Chinner		if (S_ISDIR(VFS_I(ip2)->i_mode)) {
d31a1825SCarlos Maiolino			error = xfs_dir_replace(tp, ip2, &xfs_name_dotdot,
381eee69SBrian Foster						dp1->i_ino, spaceres);
d31a1825SCarlos Maiolino			if (error)
eeacd321SDave Chinner				goto out_trans_abort;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino			/* transfer ip2 ".." reference to dp1 */
c19b3b05SDave Chinner			if (!S_ISDIR(VFS_I(ip1)->i_mode)) {
d31a1825SCarlos Maiolino				error = xfs_droplink(tp, dp2);
d31a1825SCarlos Maiolino				if (error)
eeacd321SDave Chinner					goto out_trans_abort;
91083269SEric Sandeen				xfs_bumplink(tp, dp1);
d31a1825SCarlos Maiolino			}
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino			/*
d31a1825SCarlos Maiolino			 * Although ip1 isn't changed here, userspace needs
d31a1825SCarlos Maiolino			 * to be warned about the change, so that applications
d31a1825SCarlos Maiolino			 * relying on it (like backup ones), will properly
d31a1825SCarlos Maiolino			 * notify the change
d31a1825SCarlos Maiolino			 */
d31a1825SCarlos Maiolino			ip1_flags |= XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino			ip2_flags |= XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino		}
d31a1825SCarlos Maiolino
c19b3b05SDave Chinner		if (S_ISDIR(VFS_I(ip1)->i_mode)) {
d31a1825SCarlos Maiolino			error = xfs_dir_replace(tp, ip1, &xfs_name_dotdot,
381eee69SBrian Foster						dp2->i_ino, spaceres);
d31a1825SCarlos Maiolino			if (error)
eeacd321SDave Chinner				goto out_trans_abort;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino			/* transfer ip1 ".." reference to dp2 */
c19b3b05SDave Chinner			if (!S_ISDIR(VFS_I(ip2)->i_mode)) {
d31a1825SCarlos Maiolino				error = xfs_droplink(tp, dp1);
d31a1825SCarlos Maiolino				if (error)
eeacd321SDave Chinner					goto out_trans_abort;
91083269SEric Sandeen				xfs_bumplink(tp, dp2);
d31a1825SCarlos Maiolino			}
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino			/*
d31a1825SCarlos Maiolino			 * Although ip2 isn't changed here, userspace needs
d31a1825SCarlos Maiolino			 * to be warned about the change, so that applications
d31a1825SCarlos Maiolino			 * relying on it (like backup ones), will properly
d31a1825SCarlos Maiolino			 * notify the change
d31a1825SCarlos Maiolino			 */
d31a1825SCarlos Maiolino			ip1_flags |= XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino			ip2_flags |= XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino		}
d31a1825SCarlos Maiolino	}
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	if (ip1_flags) {
d31a1825SCarlos Maiolino		xfs_trans_ichgtime(tp, ip1, ip1_flags);
d31a1825SCarlos Maiolino		xfs_trans_log_inode(tp, ip1, XFS_ILOG_CORE);
d31a1825SCarlos Maiolino	}
d31a1825SCarlos Maiolino	if (ip2_flags) {
d31a1825SCarlos Maiolino		xfs_trans_ichgtime(tp, ip2, ip2_flags);
d31a1825SCarlos Maiolino		xfs_trans_log_inode(tp, ip2, XFS_ILOG_CORE);
d31a1825SCarlos Maiolino	}
d31a1825SCarlos Maiolino	if (dp2_flags) {
d31a1825SCarlos Maiolino		xfs_trans_ichgtime(tp, dp2, dp2_flags);
d31a1825SCarlos Maiolino		xfs_trans_log_inode(tp, dp2, XFS_ILOG_CORE);
d31a1825SCarlos Maiolino	}
d31a1825SCarlos Maiolino	xfs_trans_ichgtime(tp, dp1, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
d31a1825SCarlos Maiolino	xfs_trans_log_inode(tp, dp1, XFS_ILOG_CORE);
c9cfdb38SBrian Foster	return xfs_finish_rename(tp);
eeacd321SDave Chinner
eeacd321SDave Chinnerout_trans_abort:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
d31a1825SCarlos Maiolino	return error;
d31a1825SCarlos Maiolino}
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino/*
7dcf5c3eSDave Chinner * xfs_rename_alloc_whiteout()
7dcf5c3eSDave Chinner *
b63da6c8SRandy Dunlap * Return a referenced, unlinked, unlocked inode that can be used as a
7dcf5c3eSDave Chinner * whiteout in a rename transaction. We use a tmpfile inode here so that if we
7dcf5c3eSDave Chinner * crash between allocating the inode and linking it into the rename transaction
7dcf5c3eSDave Chinner * recovery will free the inode and we won't leak it.
7dcf5c3eSDave Chinner */
7dcf5c3eSDave Chinnerstatic int
7dcf5c3eSDave Chinnerxfs_rename_alloc_whiteout(
f736d93dSChristoph Hellwig	struct user_namespace	*mnt_userns,
7dcf5c3eSDave Chinner	struct xfs_inode	*dp,
7dcf5c3eSDave Chinner	struct xfs_inode	**wip)
7dcf5c3eSDave Chinner{
7dcf5c3eSDave Chinner	struct xfs_inode	*tmpfile;
7dcf5c3eSDave Chinner	int			error;
7dcf5c3eSDave Chinner
f736d93dSChristoph Hellwig	error = xfs_create_tmpfile(mnt_userns, dp, S_IFCHR | WHITEOUT_MODE,
f736d93dSChristoph Hellwig				   &tmpfile);
7dcf5c3eSDave Chinner	if (error)
7dcf5c3eSDave Chinner		return error;
7dcf5c3eSDave Chinner
22419ac9SBrian Foster	/*
22419ac9SBrian Foster	 * Prepare the tmpfile inode as if it were created through the VFS.
c4a6bf7fSDarrick J. Wong	 * Complete the inode setup and flag it as linkable.  nlink is already
c4a6bf7fSDarrick J. Wong	 * zero, so we can skip the drop_nlink.
22419ac9SBrian Foster	 */
2b3d1d41SChristoph Hellwig	xfs_setup_iops(tmpfile);
7dcf5c3eSDave Chinner	xfs_finish_inode_setup(tmpfile);
7dcf5c3eSDave Chinner	VFS_I(tmpfile)->i_state |= I_LINKABLE;
7dcf5c3eSDave Chinner
7dcf5c3eSDave Chinner	*wip = tmpfile;
7dcf5c3eSDave Chinner	return 0;
7dcf5c3eSDave Chinner}
7dcf5c3eSDave Chinner
7dcf5c3eSDave Chinner/*
f6bba201SDave Chinner * xfs_rename
f6bba201SDave Chinner */
f6bba201SDave Chinnerint
f6bba201SDave Chinnerxfs_rename(
f736d93dSChristoph Hellwig	struct user_namespace	*mnt_userns,
7dcf5c3eSDave Chinner	struct xfs_inode	*src_dp,
f6bba201SDave Chinner	struct xfs_name		*src_name,
7dcf5c3eSDave Chinner	struct xfs_inode	*src_ip,
7dcf5c3eSDave Chinner	struct xfs_inode	*target_dp,
f6bba201SDave Chinner	struct xfs_name		*target_name,
7dcf5c3eSDave Chinner	struct xfs_inode	*target_ip,
d31a1825SCarlos Maiolino	unsigned int		flags)
f6bba201SDave Chinner{
7dcf5c3eSDave Chinner	struct xfs_mount	*mp = src_dp->i_mount;
7dcf5c3eSDave Chinner	struct xfs_trans	*tp;
7dcf5c3eSDave Chinner	struct xfs_inode	*wip = NULL;		/* whiteout inode */
7dcf5c3eSDave Chinner	struct xfs_inode	*inodes[__XFS_SORT_INODES];
6da1b4b1SDarrick J. Wong	int			i;
95afcf5cSDave Chinner	int			num_inodes = __XFS_SORT_INODES;
2b93681fSDave Chinner	bool			new_parent = (src_dp != target_dp);
c19b3b05SDave Chinner	bool			src_is_directory = S_ISDIR(VFS_I(src_ip)->i_mode);
f6bba201SDave Chinner	int			spaceres;
7dcf5c3eSDave Chinner	int			error;
f6bba201SDave Chinner
f6bba201SDave Chinner	trace_xfs_rename(src_dp, target_dp, src_name, target_name);
f6bba201SDave Chinner
eeacd321SDave Chinner	if ((flags & RENAME_EXCHANGE) && !target_ip)
eeacd321SDave Chinner		return -EINVAL;
f6bba201SDave Chinner
7dcf5c3eSDave Chinner	/*
7dcf5c3eSDave Chinner	 * If we are doing a whiteout operation, allocate the whiteout inode
7dcf5c3eSDave Chinner	 * we will be placing at the target and ensure the type is set
7dcf5c3eSDave Chinner	 * appropriately.
7dcf5c3eSDave Chinner	 */
7dcf5c3eSDave Chinner	if (flags & RENAME_WHITEOUT) {
7dcf5c3eSDave Chinner		ASSERT(!(flags & (RENAME_NOREPLACE | RENAME_EXCHANGE)));
f736d93dSChristoph Hellwig		error = xfs_rename_alloc_whiteout(mnt_userns, target_dp, &wip);
7dcf5c3eSDave Chinner		if (error)
7dcf5c3eSDave Chinner			return error;
f6bba201SDave Chinner
7dcf5c3eSDave Chinner		/* setup target dirent info as whiteout */
7dcf5c3eSDave Chinner		src_name->type = XFS_DIR3_FT_CHRDEV;
7dcf5c3eSDave Chinner	}
7dcf5c3eSDave Chinner
7dcf5c3eSDave Chinner	xfs_sort_for_rename(src_dp, target_dp, src_ip, target_ip, wip,
f6bba201SDave Chinner				inodes, &num_inodes);
f6bba201SDave Chinner
f6bba201SDave Chinner	spaceres = XFS_RENAME_SPACE_RES(mp, target_name->len);
253f4911SChristoph Hellwig	error = xfs_trans_alloc(mp, &M_RES(mp)->tr_rename, spaceres, 0, 0, &tp);
2451337dSDave Chinner	if (error == -ENOSPC) {
f6bba201SDave Chinner		spaceres = 0;
253f4911SChristoph Hellwig		error = xfs_trans_alloc(mp, &M_RES(mp)->tr_rename, 0, 0, 0,
253f4911SChristoph Hellwig				&tp);
f6bba201SDave Chinner	}
445883e8SDave Chinner	if (error)
253f4911SChristoph Hellwig		goto out_release_wip;
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Attach the dquots to the inodes
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	error = xfs_qm_vop_rename_dqattach(inodes);
445883e8SDave Chinner	if (error)
445883e8SDave Chinner		goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Lock all the participating inodes. Depending upon whether
f6bba201SDave Chinner	 * the target_name exists in the target directory, and
f6bba201SDave Chinner	 * whether the target directory is the same as the source
f6bba201SDave Chinner	 * directory, we can lock from 2 to 4 inodes.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	xfs_lock_inodes(inodes, num_inodes, XFS_ILOCK_EXCL);
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Join all the inodes to the transaction. From this point on,
f6bba201SDave Chinner	 * we can rely on either trans_commit or trans_cancel to unlock
f6bba201SDave Chinner	 * them.
f6bba201SDave Chinner	 */
65523218SChristoph Hellwig	xfs_trans_ijoin(tp, src_dp, XFS_ILOCK_EXCL);
f6bba201SDave Chinner	if (new_parent)
65523218SChristoph Hellwig		xfs_trans_ijoin(tp, target_dp, XFS_ILOCK_EXCL);
f6bba201SDave Chinner	xfs_trans_ijoin(tp, src_ip, XFS_ILOCK_EXCL);
f6bba201SDave Chinner	if (target_ip)
f6bba201SDave Chinner		xfs_trans_ijoin(tp, target_ip, XFS_ILOCK_EXCL);
7dcf5c3eSDave Chinner	if (wip)
7dcf5c3eSDave Chinner		xfs_trans_ijoin(tp, wip, XFS_ILOCK_EXCL);
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * If we are using project inheritance, we only allow renames
f6bba201SDave Chinner	 * into our tree when the project IDs are the same; else the
f6bba201SDave Chinner	 * tree quota mechanism would be circumvented.
f6bba201SDave Chinner	 */
db07349dSChristoph Hellwig	if (unlikely((target_dp->i_diflags & XFS_DIFLAG_PROJINHERIT) &&
ceaf603cSChristoph Hellwig		     target_dp->i_projid != src_ip->i_projid)) {
2451337dSDave Chinner		error = -EXDEV;
445883e8SDave Chinner		goto out_trans_cancel;
f6bba201SDave Chinner	}
f6bba201SDave Chinner
eeacd321SDave Chinner	/* RENAME_EXCHANGE is unique from here on. */
eeacd321SDave Chinner	if (flags & RENAME_EXCHANGE)
eeacd321SDave Chinner		return xfs_cross_rename(tp, src_dp, src_name, src_ip,
d31a1825SCarlos Maiolino					target_dp, target_name, target_ip,
f16dea54SBrian Foster					spaceres);
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	/*
bc56ad8cSkaixuxia	 * Check for expected errors before we dirty the transaction
bc56ad8cSkaixuxia	 * so we can return an error without a transaction abort.
02092a2fSChandan Babu R	 *
02092a2fSChandan Babu R	 * Extent count overflow check:
02092a2fSChandan Babu R	 *
02092a2fSChandan Babu R	 * From the perspective of src_dp, a rename operation is essentially a
02092a2fSChandan Babu R	 * directory entry remove operation. Hence the only place where we check
02092a2fSChandan Babu R	 * for extent count overflow for src_dp is in
02092a2fSChandan Babu R	 * xfs_bmap_del_extent_real(). xfs_bmap_del_extent_real() returns
02092a2fSChandan Babu R	 * -ENOSPC when it detects a possible extent count overflow and in
02092a2fSChandan Babu R	 * response, the higher layers of directory handling code do the
02092a2fSChandan Babu R	 * following:
02092a2fSChandan Babu R	 * 1. Data/Free blocks: XFS lets these blocks linger until a
02092a2fSChandan Babu R	 *    future remove operation removes them.
02092a2fSChandan Babu R	 * 2. Dabtree blocks: XFS swaps the blocks with the last block in the
02092a2fSChandan Babu R	 *    Leaf space and unmaps the last block.
02092a2fSChandan Babu R	 *
02092a2fSChandan Babu R	 * For target_dp, there are two cases depending on whether the
02092a2fSChandan Babu R	 * destination directory entry exists or not.
02092a2fSChandan Babu R	 *
02092a2fSChandan Babu R	 * When destination directory entry does not exist (i.e. target_ip ==
02092a2fSChandan Babu R	 * NULL), extent count overflow check is performed only when transaction
02092a2fSChandan Babu R	 * has a non-zero sized space reservation associated with it.  With a
02092a2fSChandan Babu R	 * zero-sized space reservation, XFS allows a rename operation to
02092a2fSChandan Babu R	 * continue only when the directory has sufficient free space in its
02092a2fSChandan Babu R	 * data/leaf/free space blocks to hold the new entry.
02092a2fSChandan Babu R	 *
02092a2fSChandan Babu R	 * When destination directory entry exists (i.e. target_ip != NULL), all
02092a2fSChandan Babu R	 * we need to do is change the inode number associated with the already
02092a2fSChandan Babu R	 * existing entry. Hence there is no need to perform an extent count
02092a2fSChandan Babu R	 * overflow check.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	if (target_ip == NULL) {
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * If there's no space reservation, check the entry will
f6bba201SDave Chinner		 * fit before actually inserting it.
f6bba201SDave Chinner		 */
94f3cad5SEric Sandeen		if (!spaceres) {
94f3cad5SEric Sandeen			error = xfs_dir_canenter(tp, target_dp, target_name);
f6bba201SDave Chinner			if (error)
445883e8SDave Chinner				goto out_trans_cancel;
02092a2fSChandan Babu R		} else {
02092a2fSChandan Babu R			error = xfs_iext_count_may_overflow(target_dp,
02092a2fSChandan Babu R					XFS_DATA_FORK,
02092a2fSChandan Babu R					XFS_IEXT_DIR_MANIP_CNT(mp));
02092a2fSChandan Babu R			if (error)
02092a2fSChandan Babu R				goto out_trans_cancel;
94f3cad5SEric Sandeen		}
bc56ad8cSkaixuxia	} else {
bc56ad8cSkaixuxia		/*
bc56ad8cSkaixuxia		 * If target exists and it's a directory, check that whether
bc56ad8cSkaixuxia		 * it can be destroyed.
bc56ad8cSkaixuxia		 */
bc56ad8cSkaixuxia		if (S_ISDIR(VFS_I(target_ip)->i_mode) &&
bc56ad8cSkaixuxia		    (!xfs_dir_isempty(target_ip) ||
bc56ad8cSkaixuxia		     (VFS_I(target_ip)->i_nlink > 2))) {
bc56ad8cSkaixuxia			error = -EEXIST;
bc56ad8cSkaixuxia			goto out_trans_cancel;
bc56ad8cSkaixuxia		}
bc56ad8cSkaixuxia	}
bc56ad8cSkaixuxia
bc56ad8cSkaixuxia	/*
6da1b4b1SDarrick J. Wong	 * Lock the AGI buffers we need to handle bumping the nlink of the
6da1b4b1SDarrick J. Wong	 * whiteout inode off the unlinked list and to handle dropping the
6da1b4b1SDarrick J. Wong	 * nlink of the target inode.  Per locking order rules, do this in
6da1b4b1SDarrick J. Wong	 * increasing AG order and before directory block allocation tries to
6da1b4b1SDarrick J. Wong	 * grab AGFs because we grab AGIs before AGFs.
6da1b4b1SDarrick J. Wong	 *
6da1b4b1SDarrick J. Wong	 * The (vfs) caller must ensure that if src is a directory then
6da1b4b1SDarrick J. Wong	 * target_ip is either null or an empty directory.
6da1b4b1SDarrick J. Wong	 */
6da1b4b1SDarrick J. Wong	for (i = 0; i < num_inodes && inodes[i] != NULL; i++) {
6da1b4b1SDarrick J. Wong		if (inodes[i] == wip ||
6da1b4b1SDarrick J. Wong		    (inodes[i] == target_ip &&
6da1b4b1SDarrick J. Wong		     (VFS_I(target_ip)->i_nlink == 1 || src_is_directory))) {
6da1b4b1SDarrick J. Wong			struct xfs_buf	*bp;
6da1b4b1SDarrick J. Wong			xfs_agnumber_t	agno;
6da1b4b1SDarrick J. Wong
6da1b4b1SDarrick J. Wong			agno = XFS_INO_TO_AGNO(mp, inodes[i]->i_ino);
6da1b4b1SDarrick J. Wong			error = xfs_read_agi(mp, tp, agno, &bp);
6da1b4b1SDarrick J. Wong			if (error)
6da1b4b1SDarrick J. Wong				goto out_trans_cancel;
6da1b4b1SDarrick J. Wong		}
6da1b4b1SDarrick J. Wong	}
6da1b4b1SDarrick J. Wong
6da1b4b1SDarrick J. Wong	/*
bc56ad8cSkaixuxia	 * Directory entry creation below may acquire the AGF. Remove
bc56ad8cSkaixuxia	 * the whiteout from the unlinked list first to preserve correct
bc56ad8cSkaixuxia	 * AGI/AGF locking order. This dirties the transaction so failures
bc56ad8cSkaixuxia	 * after this point will abort and log recovery will clean up the
bc56ad8cSkaixuxia	 * mess.
bc56ad8cSkaixuxia	 *
bc56ad8cSkaixuxia	 * For whiteouts, we need to bump the link count on the whiteout
bc56ad8cSkaixuxia	 * inode. After this point, we have a real link, clear the tmpfile
bc56ad8cSkaixuxia	 * state flag from the inode so it doesn't accidentally get misused
bc56ad8cSkaixuxia	 * in future.
bc56ad8cSkaixuxia	 */
bc56ad8cSkaixuxia	if (wip) {
f40aadb2SDave Chinner		struct xfs_perag	*pag;
f40aadb2SDave Chinner
bc56ad8cSkaixuxia		ASSERT(VFS_I(wip)->i_nlink == 0);
f40aadb2SDave Chinner
f40aadb2SDave Chinner		pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, wip->i_ino));
f40aadb2SDave Chinner		error = xfs_iunlink_remove(tp, pag, wip);
f40aadb2SDave Chinner		xfs_perag_put(pag);
bc56ad8cSkaixuxia		if (error)
bc56ad8cSkaixuxia			goto out_trans_cancel;
bc56ad8cSkaixuxia
bc56ad8cSkaixuxia		xfs_bumplink(tp, wip);
bc56ad8cSkaixuxia		VFS_I(wip)->i_state &= ~I_LINKABLE;
bc56ad8cSkaixuxia	}
bc56ad8cSkaixuxia
bc56ad8cSkaixuxia	/*
bc56ad8cSkaixuxia	 * Set up the target.
bc56ad8cSkaixuxia	 */
bc56ad8cSkaixuxia	if (target_ip == NULL) {
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * If target does not exist and the rename crosses
f6bba201SDave Chinner		 * directories, adjust the target directory link count
f6bba201SDave Chinner		 * to account for the ".." reference from the new entry.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_dir_createname(tp, target_dp, target_name,
381eee69SBrian Foster					   src_ip->i_ino, spaceres);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner		xfs_trans_ichgtime(tp, target_dp,
f6bba201SDave Chinner					XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
f6bba201SDave Chinner
f6bba201SDave Chinner		if (new_parent && src_is_directory) {
91083269SEric Sandeen			xfs_bumplink(tp, target_dp);
f6bba201SDave Chinner		}
f6bba201SDave Chinner	} else { /* target_ip != NULL */
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * Link the source inode under the target name.
f6bba201SDave Chinner		 * If the source inode is a directory and we are moving
f6bba201SDave Chinner		 * it across directories, its ".." entry will be
f6bba201SDave Chinner		 * inconsistent until we replace that down below.
f6bba201SDave Chinner		 *
f6bba201SDave Chinner		 * In case there is already an entry with the same
f6bba201SDave Chinner		 * name at the destination directory, remove it first.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_dir_replace(tp, target_dp, target_name,
381eee69SBrian Foster					src_ip->i_ino, spaceres);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner		xfs_trans_ichgtime(tp, target_dp,
f6bba201SDave Chinner					XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
f6bba201SDave Chinner
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * Decrement the link count on the target since the target
f6bba201SDave Chinner		 * dir no longer points to it.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_droplink(tp, target_ip);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner		if (src_is_directory) {
f6bba201SDave Chinner			/*
f6bba201SDave Chinner			 * Drop the link from the old "." entry.
f6bba201SDave Chinner			 */
f6bba201SDave Chinner			error = xfs_droplink(tp, target_ip);
f6bba201SDave Chinner			if (error)
c8eac49eSBrian Foster				goto out_trans_cancel;
f6bba201SDave Chinner		}
f6bba201SDave Chinner	} /* target_ip != NULL */
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Remove the source.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	if (new_parent && src_is_directory) {
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * Rewrite the ".." entry to point to the new
f6bba201SDave Chinner		 * directory.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_dir_replace(tp, src_ip, &xfs_name_dotdot,
381eee69SBrian Foster					target_dp->i_ino, spaceres);
2451337dSDave Chinner		ASSERT(error != -EEXIST);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner	}
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * We always want to hit the ctime on the source inode.
f6bba201SDave Chinner	 *
f6bba201SDave Chinner	 * This isn't strictly required by the standards since the source
f6bba201SDave Chinner	 * inode isn't really being changed, but old unix file systems did
f6bba201SDave Chinner	 * it and some incremental backup programs won't work without it.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	xfs_trans_ichgtime(tp, src_ip, XFS_ICHGTIME_CHG);
f6bba201SDave Chinner	xfs_trans_log_inode(tp, src_ip, XFS_ILOG_CORE);
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Adjust the link count on src_dp.  This is necessary when
f6bba201SDave Chinner	 * renaming a directory, either within one parent when
f6bba201SDave Chinner	 * the target existed, or across two parent directories.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	if (src_is_directory && (new_parent || target_ip != NULL)) {
f6bba201SDave Chinner
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * Decrement link count on src_directory since the
f6bba201SDave Chinner		 * entry that's moved no longer points to it.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_droplink(tp, src_dp);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner	}
f6bba201SDave Chinner
7dcf5c3eSDave Chinner	/*
7dcf5c3eSDave Chinner	 * For whiteouts, we only need to update the source dirent with the
7dcf5c3eSDave Chinner	 * inode number of the whiteout inode rather than removing it
7dcf5c3eSDave Chinner	 * altogether.
7dcf5c3eSDave Chinner	 */
7dcf5c3eSDave Chinner	if (wip) {
7dcf5c3eSDave Chinner		error = xfs_dir_replace(tp, src_dp, src_name, wip->i_ino,
381eee69SBrian Foster					spaceres);
02092a2fSChandan Babu R	} else {
02092a2fSChandan Babu R		/*
02092a2fSChandan Babu R		 * NOTE: We don't need to check for extent count overflow here
02092a2fSChandan Babu R		 * because the dir remove name code will leave the dir block in
02092a2fSChandan Babu R		 * place if the extent count would overflow.
02092a2fSChandan Babu R		 */
f6bba201SDave Chinner		error = xfs_dir_removename(tp, src_dp, src_name, src_ip->i_ino,
381eee69SBrian Foster					   spaceres);
02092a2fSChandan Babu R	}
02092a2fSChandan Babu R
f6bba201SDave Chinner	if (error)
c8eac49eSBrian Foster		goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner	xfs_trans_ichgtime(tp, src_dp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
f6bba201SDave Chinner	xfs_trans_log_inode(tp, src_dp, XFS_ILOG_CORE);
f6bba201SDave Chinner	if (new_parent)
f6bba201SDave Chinner		xfs_trans_log_inode(tp, target_dp, XFS_ILOG_CORE);
f6bba201SDave Chinner
c9cfdb38SBrian Foster	error = xfs_finish_rename(tp);
7dcf5c3eSDave Chinner	if (wip)
44a8736bSDarrick J. Wong		xfs_irele(wip);
7dcf5c3eSDave Chinner	return error;
f6bba201SDave Chinner
445883e8SDave Chinnerout_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
253f4911SChristoph Hellwigout_release_wip:
7dcf5c3eSDave Chinner	if (wip)
44a8736bSDarrick J. Wong		xfs_irele(wip);
f6bba201SDave Chinner	return error;
f6bba201SDave Chinner}
f6bba201SDave Chinner
e6187b34SDave Chinnerstatic int
e6187b34SDave Chinnerxfs_iflush(
93848a99SChristoph Hellwig	struct xfs_inode	*ip,
93848a99SChristoph Hellwig	struct xfs_buf		*bp)
1da177e4SLinus Torvalds{
93848a99SChristoph Hellwig	struct xfs_inode_log_item *iip = ip->i_itemp;
93848a99SChristoph Hellwig	struct xfs_dinode	*dip;
93848a99SChristoph Hellwig	struct xfs_mount	*mp = ip->i_mount;
f2019299SBrian Foster	int			error;
1da177e4SLinus Torvalds
579aa9caSChristoph Hellwig	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL|XFS_ILOCK_SHARED));
718ecc50SDave Chinner	ASSERT(xfs_iflags_test(ip, XFS_IFLUSHING));
f7e67b20SChristoph Hellwig	ASSERT(ip->i_df.if_format != XFS_DINODE_FMT_BTREE ||
daf83964SChristoph Hellwig	       ip->i_df.if_nextents > XFS_IFORK_MAXEXT(ip, XFS_DATA_FORK));
90c60e16SDave Chinner	ASSERT(iip->ili_item.li_buf == bp);
1da177e4SLinus Torvalds
88ee2df7SChristoph Hellwig	dip = xfs_buf_offset(bp, ip->i_imap.im_boffset);
1da177e4SLinus Torvalds
f2019299SBrian Foster	/*
f2019299SBrian Foster	 * We don't flush the inode if any of the following checks fail, but we
f2019299SBrian Foster	 * do still update the log item and attach to the backing buffer as if
f2019299SBrian Foster	 * the flush happened. This is a formality to facilitate predictable
f2019299SBrian Foster	 * error handling as the caller will shutdown and fail the buffer.
f2019299SBrian Foster	 */
f2019299SBrian Foster	error = -EFSCORRUPTED;
69ef921bSChristoph Hellwig	if (XFS_TEST_ERROR(dip->di_magic != cpu_to_be16(XFS_DINODE_MAGIC),
9e24cfd0SDarrick J. Wong			       mp, XFS_ERRTAG_IFLUSH_1)) {
6a19d939SDave Chinner		xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
c9690043SDarrick J. Wong			"%s: Bad inode %Lu magic number 0x%x, ptr "PTR_FMT,
6a19d939SDave Chinner			__func__, ip->i_ino, be16_to_cpu(dip->di_magic), dip);
f2019299SBrian Foster		goto flush_out;
1da177e4SLinus Torvalds	}
c19b3b05SDave Chinner	if (S_ISREG(VFS_I(ip)->i_mode)) {
1da177e4SLinus Torvalds		if (XFS_TEST_ERROR(
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_EXTENTS &&
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_BTREE,
9e24cfd0SDarrick J. Wong		    mp, XFS_ERRTAG_IFLUSH_3)) {
6a19d939SDave Chinner			xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
c9690043SDarrick J. Wong				"%s: Bad regular inode %Lu, ptr "PTR_FMT,
6a19d939SDave Chinner				__func__, ip->i_ino, ip);
f2019299SBrian Foster			goto flush_out;
1da177e4SLinus Torvalds		}
c19b3b05SDave Chinner	} else if (S_ISDIR(VFS_I(ip)->i_mode)) {
1da177e4SLinus Torvalds		if (XFS_TEST_ERROR(
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_EXTENTS &&
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_BTREE &&
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_LOCAL,
9e24cfd0SDarrick J. Wong		    mp, XFS_ERRTAG_IFLUSH_4)) {
6a19d939SDave Chinner			xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
c9690043SDarrick J. Wong				"%s: Bad directory inode %Lu, ptr "PTR_FMT,
6a19d939SDave Chinner				__func__, ip->i_ino, ip);
f2019299SBrian Foster			goto flush_out;
1da177e4SLinus Torvalds		}
1da177e4SLinus Torvalds	}
daf83964SChristoph Hellwig	if (XFS_TEST_ERROR(ip->i_df.if_nextents + xfs_ifork_nextents(ip->i_afp) >
6e73a545SChristoph Hellwig				ip->i_nblocks, mp, XFS_ERRTAG_IFLUSH_5)) {
6a19d939SDave Chinner		xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
6a19d939SDave Chinner			"%s: detected corrupt incore inode %Lu, "
c9690043SDarrick J. Wong			"total extents = %d, nblocks = %Ld, ptr "PTR_FMT,
6a19d939SDave Chinner			__func__, ip->i_ino,
daf83964SChristoph Hellwig			ip->i_df.if_nextents + xfs_ifork_nextents(ip->i_afp),
6e73a545SChristoph Hellwig			ip->i_nblocks, ip);
f2019299SBrian Foster		goto flush_out;
1da177e4SLinus Torvalds	}
7821ea30SChristoph Hellwig	if (XFS_TEST_ERROR(ip->i_forkoff > mp->m_sb.sb_inodesize,
9e24cfd0SDarrick J. Wong				mp, XFS_ERRTAG_IFLUSH_6)) {
6a19d939SDave Chinner		xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
c9690043SDarrick J. Wong			"%s: bad inode %Lu, forkoff 0x%x, ptr "PTR_FMT,
7821ea30SChristoph Hellwig			__func__, ip->i_ino, ip->i_forkoff, ip);
f2019299SBrian Foster		goto flush_out;
1da177e4SLinus Torvalds	}
e60896d8SDave Chinner
1da177e4SLinus Torvalds	/*
965e0a1aSChristoph Hellwig	 * Inode item log recovery for v2 inodes are dependent on the flushiter
965e0a1aSChristoph Hellwig	 * count for correct sequencing.  We bump the flush iteration count so
965e0a1aSChristoph Hellwig	 * we can detect flushes which postdate a log record during recovery.
965e0a1aSChristoph Hellwig	 * This is redundant as we now log every change and hence this can't
965e0a1aSChristoph Hellwig	 * happen but we need to still do it to ensure backwards compatibility
965e0a1aSChristoph Hellwig	 * with old kernels that predate logging all inode changes.
1da177e4SLinus Torvalds	 */
38c26bfdSDave Chinner	if (!xfs_has_v3inodes(mp))
965e0a1aSChristoph Hellwig		ip->i_flushiter++;
1da177e4SLinus Torvalds
0f45a1b2SChristoph Hellwig	/*
0f45a1b2SChristoph Hellwig	 * If there are inline format data / attr forks attached to this inode,
0f45a1b2SChristoph Hellwig	 * make sure they are not corrupt.
0f45a1b2SChristoph Hellwig	 */
f7e67b20SChristoph Hellwig	if (ip->i_df.if_format == XFS_DINODE_FMT_LOCAL &&
0f45a1b2SChristoph Hellwig	    xfs_ifork_verify_local_data(ip))
0f45a1b2SChristoph Hellwig		goto flush_out;
f7e67b20SChristoph Hellwig	if (ip->i_afp && ip->i_afp->if_format == XFS_DINODE_FMT_LOCAL &&
0f45a1b2SChristoph Hellwig	    xfs_ifork_verify_local_attr(ip))
f2019299SBrian Foster		goto flush_out;
005c5db8SDarrick J. Wong
1da177e4SLinus Torvalds	/*
3987848cSDave Chinner	 * Copy the dirty parts of the inode into the on-disk inode.  We always
3987848cSDave Chinner	 * copy out the core of the inode, because if the inode is dirty at all
3987848cSDave Chinner	 * the core must be.
1da177e4SLinus Torvalds	 */
93f958f9SDave Chinner	xfs_inode_to_disk(ip, dip, iip->ili_item.li_lsn);
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/* Wrap, we never let the log put out DI_MAX_FLUSH */
38c26bfdSDave Chinner	if (!xfs_has_v3inodes(mp)) {
965e0a1aSChristoph Hellwig		if (ip->i_flushiter == DI_MAX_FLUSH)
965e0a1aSChristoph Hellwig			ip->i_flushiter = 0;
ee7b83fdSChristoph Hellwig	}
1da177e4SLinus Torvalds
005c5db8SDarrick J. Wong	xfs_iflush_fork(ip, dip, iip, XFS_DATA_FORK);
005c5db8SDarrick J. Wong	if (XFS_IFORK_Q(ip))
005c5db8SDarrick J. Wong		xfs_iflush_fork(ip, dip, iip, XFS_ATTR_FORK);
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
f5d8d5c4SChristoph Hellwig	 * We've recorded everything logged in the inode, so we'd like to clear
f5d8d5c4SChristoph Hellwig	 * the ili_fields bits so we don't log and flush things unnecessarily.
f5d8d5c4SChristoph Hellwig	 * However, we can't stop logging all this information until the data
f5d8d5c4SChristoph Hellwig	 * we've copied into the disk buffer is written to disk.  If we did we
f5d8d5c4SChristoph Hellwig	 * might overwrite the copy of the inode in the log with all the data
f5d8d5c4SChristoph Hellwig	 * after re-logging only part of it, and in the face of a crash we
f5d8d5c4SChristoph Hellwig	 * wouldn't have all the data we need to recover.
1da177e4SLinus Torvalds	 *
f5d8d5c4SChristoph Hellwig	 * What we do is move the bits to the ili_last_fields field.  When
f5d8d5c4SChristoph Hellwig	 * logging the inode, these bits are moved back to the ili_fields field.
664ffb8aSChristoph Hellwig	 * In the xfs_buf_inode_iodone() routine we clear ili_last_fields, since
664ffb8aSChristoph Hellwig	 * we know that the information those bits represent is permanently on
f5d8d5c4SChristoph Hellwig	 * disk.  As long as the flush completes before the inode is logged
f5d8d5c4SChristoph Hellwig	 * again, then both ili_fields and ili_last_fields will be cleared.
1da177e4SLinus Torvalds	 */
f2019299SBrian Foster	error = 0;
f2019299SBrian Fosterflush_out:
1319ebefSDave Chinner	spin_lock(&iip->ili_lock);
f5d8d5c4SChristoph Hellwig	iip->ili_last_fields = iip->ili_fields;
f5d8d5c4SChristoph Hellwig	iip->ili_fields = 0;
fc0561ceSDave Chinner	iip->ili_fsync_fields = 0;
1319ebefSDave Chinner	spin_unlock(&iip->ili_lock);
1da177e4SLinus Torvalds
1319ebefSDave Chinner	/*
1319ebefSDave Chinner	 * Store the current LSN of the inode so that we can tell whether the
664ffb8aSChristoph Hellwig	 * item has moved in the AIL from xfs_buf_inode_iodone().
1319ebefSDave Chinner	 */
7b2e2a31SDavid Chinner	xfs_trans_ail_copy_lsn(mp->m_ail, &iip->ili_flush_lsn,
7b2e2a31SDavid Chinner				&iip->ili_item.li_lsn);
1da177e4SLinus Torvalds
93848a99SChristoph Hellwig	/* generate the checksum. */
93848a99SChristoph Hellwig	xfs_dinode_calc_crc(mp, dip);
f2019299SBrian Foster	return error;
1da177e4SLinus Torvalds}
44a8736bSDarrick J. Wong
e6187b34SDave Chinner/*
e6187b34SDave Chinner * Non-blocking flush of dirty inode metadata into the backing buffer.
e6187b34SDave Chinner *
e6187b34SDave Chinner * The caller must have a reference to the inode and hold the cluster buffer
e6187b34SDave Chinner * locked. The function will walk across all the inodes on the cluster buffer it
e6187b34SDave Chinner * can find and lock without blocking, and flush them to the cluster buffer.
e6187b34SDave Chinner *
5717ea4dSDave Chinner * On successful flushing of at least one inode, the caller must write out the
5717ea4dSDave Chinner * buffer and release it. If no inodes are flushed, -EAGAIN will be returned and
5717ea4dSDave Chinner * the caller needs to release the buffer. On failure, the filesystem will be
5717ea4dSDave Chinner * shut down, the buffer will have been unlocked and released, and EFSCORRUPTED
5717ea4dSDave Chinner * will be returned.
e6187b34SDave Chinner */
e6187b34SDave Chinnerint
e6187b34SDave Chinnerxfs_iflush_cluster(
e6187b34SDave Chinner	struct xfs_buf		*bp)
e6187b34SDave Chinner{
5717ea4dSDave Chinner	struct xfs_mount	*mp = bp->b_mount;
5717ea4dSDave Chinner	struct xfs_log_item	*lip, *n;
5717ea4dSDave Chinner	struct xfs_inode	*ip;
5717ea4dSDave Chinner	struct xfs_inode_log_item *iip;
e6187b34SDave Chinner	int			clcount = 0;
5717ea4dSDave Chinner	int			error = 0;
e6187b34SDave Chinner
e6187b34SDave Chinner	/*
5717ea4dSDave Chinner	 * We must use the safe variant here as on shutdown xfs_iflush_abort()
5717ea4dSDave Chinner	 * can remove itself from the list.
e6187b34SDave Chinner	 */
5717ea4dSDave Chinner	list_for_each_entry_safe(lip, n, &bp->b_li_list, li_bio_list) {
5717ea4dSDave Chinner		iip = (struct xfs_inode_log_item *)lip;
5717ea4dSDave Chinner		ip = iip->ili_inode;
5717ea4dSDave Chinner
5717ea4dSDave Chinner		/*
5717ea4dSDave Chinner		 * Quick and dirty check to avoid locks if possible.
5717ea4dSDave Chinner		 */
718ecc50SDave Chinner		if (__xfs_iflags_test(ip, XFS_IRECLAIM | XFS_IFLUSHING))
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner		if (xfs_ipincount(ip))
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner
5717ea4dSDave Chinner		/*
5717ea4dSDave Chinner		 * The inode is still attached to the buffer, which means it is
5717ea4dSDave Chinner		 * dirty but reclaim might try to grab it. Check carefully for
5717ea4dSDave Chinner		 * that, and grab the ilock while still holding the i_flags_lock
5717ea4dSDave Chinner		 * to guarantee reclaim will not be able to reclaim this inode
5717ea4dSDave Chinner		 * once we drop the i_flags_lock.
5717ea4dSDave Chinner		 */
5717ea4dSDave Chinner		spin_lock(&ip->i_flags_lock);
5717ea4dSDave Chinner		ASSERT(!__xfs_iflags_test(ip, XFS_ISTALE));
718ecc50SDave Chinner		if (__xfs_iflags_test(ip, XFS_IRECLAIM | XFS_IFLUSHING)) {
5717ea4dSDave Chinner			spin_unlock(&ip->i_flags_lock);
e6187b34SDave Chinner			continue;
e6187b34SDave Chinner		}
e6187b34SDave Chinner
e6187b34SDave Chinner		/*
5717ea4dSDave Chinner		 * ILOCK will pin the inode against reclaim and prevent
5717ea4dSDave Chinner		 * concurrent transactions modifying the inode while we are
718ecc50SDave Chinner		 * flushing the inode. If we get the lock, set the flushing
718ecc50SDave Chinner		 * state before we drop the i_flags_lock.
e6187b34SDave Chinner		 */
5717ea4dSDave Chinner		if (!xfs_ilock_nowait(ip, XFS_ILOCK_SHARED)) {
5717ea4dSDave Chinner			spin_unlock(&ip->i_flags_lock);
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner		}
718ecc50SDave Chinner		__xfs_iflags_set(ip, XFS_IFLUSHING);
5717ea4dSDave Chinner		spin_unlock(&ip->i_flags_lock);
5717ea4dSDave Chinner
5717ea4dSDave Chinner		/*
5717ea4dSDave Chinner		 * Abort flushing this inode if we are shut down because the
5717ea4dSDave Chinner		 * inode may not currently be in the AIL. This can occur when
5717ea4dSDave Chinner		 * log I/O failure unpins the inode without inserting into the
5717ea4dSDave Chinner		 * AIL, leaving a dirty/unpinned inode attached to the buffer
5717ea4dSDave Chinner		 * that otherwise looks like it should be flushed.
5717ea4dSDave Chinner		 */
5717ea4dSDave Chinner		if (XFS_FORCED_SHUTDOWN(mp)) {
5717ea4dSDave Chinner			xfs_iunpin_wait(ip);
5717ea4dSDave Chinner			xfs_iflush_abort(ip);
5717ea4dSDave Chinner			xfs_iunlock(ip, XFS_ILOCK_SHARED);
5717ea4dSDave Chinner			error = -EIO;
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner		}
5717ea4dSDave Chinner
5717ea4dSDave Chinner		/* don't block waiting on a log force to unpin dirty inodes */
5717ea4dSDave Chinner		if (xfs_ipincount(ip)) {
718ecc50SDave Chinner			xfs_iflags_clear(ip, XFS_IFLUSHING);
5717ea4dSDave Chinner			xfs_iunlock(ip, XFS_ILOCK_SHARED);
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner		}
5717ea4dSDave Chinner
5717ea4dSDave Chinner		if (!xfs_inode_clean(ip))
5717ea4dSDave Chinner			error = xfs_iflush(ip, bp);
5717ea4dSDave Chinner		else
718ecc50SDave Chinner			xfs_iflags_clear(ip, XFS_IFLUSHING);
5717ea4dSDave Chinner		xfs_iunlock(ip, XFS_ILOCK_SHARED);
5717ea4dSDave Chinner		if (error)
e6187b34SDave Chinner			break;
e6187b34SDave Chinner		clcount++;
e6187b34SDave Chinner	}
e6187b34SDave Chinner
e6187b34SDave Chinner	if (error) {
e6187b34SDave Chinner		bp->b_flags |= XBF_ASYNC;
e6187b34SDave Chinner		xfs_buf_ioend_fail(bp);
e6187b34SDave Chinner		xfs_force_shutdown(mp, SHUTDOWN_CORRUPT_INCORE);
e6187b34SDave Chinner		return error;
e6187b34SDave Chinner	}
e6187b34SDave Chinner
5717ea4dSDave Chinner	if (!clcount)
5717ea4dSDave Chinner		return -EAGAIN;
5717ea4dSDave Chinner
5717ea4dSDave Chinner	XFS_STATS_INC(mp, xs_icluster_flushcnt);
5717ea4dSDave Chinner	XFS_STATS_ADD(mp, xs_icluster_flushinode, clcount);
5717ea4dSDave Chinner	return 0;
5717ea4dSDave Chinner
5717ea4dSDave Chinner}
5717ea4dSDave Chinner
44a8736bSDarrick J. Wong/* Release an inode. */
44a8736bSDarrick J. Wongvoid
44a8736bSDarrick J. Wongxfs_irele(
44a8736bSDarrick J. Wong	struct xfs_inode	*ip)
44a8736bSDarrick J. Wong{
44a8736bSDarrick J. Wong	trace_xfs_irele(ip, _RET_IP_);
44a8736bSDarrick J. Wong	iput(VFS_I(ip));
44a8736bSDarrick J. Wong}
54fbdd10SChristoph Hellwig
54fbdd10SChristoph Hellwig/*
54fbdd10SChristoph Hellwig * Ensure all commited transactions touching the inode are written to the log.
54fbdd10SChristoph Hellwig */
54fbdd10SChristoph Hellwigint
54fbdd10SChristoph Hellwigxfs_log_force_inode(
54fbdd10SChristoph Hellwig	struct xfs_inode	*ip)
54fbdd10SChristoph Hellwig{
5f9b4b0dSDave Chinner	xfs_csn_t		seq = 0;
54fbdd10SChristoph Hellwig
54fbdd10SChristoph Hellwig	xfs_ilock(ip, XFS_ILOCK_SHARED);
54fbdd10SChristoph Hellwig	if (xfs_ipincount(ip))
5f9b4b0dSDave Chinner		seq = ip->i_itemp->ili_commit_seq;
54fbdd10SChristoph Hellwig	xfs_iunlock(ip, XFS_ILOCK_SHARED);
54fbdd10SChristoph Hellwig
5f9b4b0dSDave Chinner	if (!seq)
54fbdd10SChristoph Hellwig		return 0;
5f9b4b0dSDave Chinner	return xfs_log_force_seq(ip->i_mount, seq, XFS_LOG_SYNC, NULL);
54fbdd10SChristoph Hellwig}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong/*
e2aaee9cSDarrick J. Wong * Grab the exclusive iolock for a data copy from src to dest, making sure to
e2aaee9cSDarrick J. Wong * abide vfs locking order (lowest pointer value goes first) and breaking the
e2aaee9cSDarrick J. Wong * layout leases before proceeding.  The loop is needed because we cannot call
e2aaee9cSDarrick J. Wong * the blocking break_layout() with the iolocks held, and therefore have to
e2aaee9cSDarrick J. Wong * back out both locks.
e2aaee9cSDarrick J. Wong */
e2aaee9cSDarrick J. Wongstatic int
e2aaee9cSDarrick J. Wongxfs_iolock_two_inodes_and_break_layout(
e2aaee9cSDarrick J. Wong	struct inode		*src,
e2aaee9cSDarrick J. Wong	struct inode		*dest)
e2aaee9cSDarrick J. Wong{
e2aaee9cSDarrick J. Wong	int			error;
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	if (src > dest)
e2aaee9cSDarrick J. Wong		swap(src, dest);
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wongretry:
e2aaee9cSDarrick J. Wong	/* Wait to break both inodes' layouts before we start locking. */
e2aaee9cSDarrick J. Wong	error = break_layout(src, true);
e2aaee9cSDarrick J. Wong	if (error)
e2aaee9cSDarrick J. Wong		return error;
e2aaee9cSDarrick J. Wong	if (src != dest) {
e2aaee9cSDarrick J. Wong		error = break_layout(dest, true);
e2aaee9cSDarrick J. Wong		if (error)
e2aaee9cSDarrick J. Wong			return error;
e2aaee9cSDarrick J. Wong	}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	/* Lock one inode and make sure nobody got in and leased it. */
e2aaee9cSDarrick J. Wong	inode_lock(src);
e2aaee9cSDarrick J. Wong	error = break_layout(src, false);
e2aaee9cSDarrick J. Wong	if (error) {
e2aaee9cSDarrick J. Wong		inode_unlock(src);
e2aaee9cSDarrick J. Wong		if (error == -EWOULDBLOCK)
e2aaee9cSDarrick J. Wong			goto retry;
e2aaee9cSDarrick J. Wong		return error;
e2aaee9cSDarrick J. Wong	}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	if (src == dest)
e2aaee9cSDarrick J. Wong		return 0;
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	/* Lock the other inode and make sure nobody got in and leased it. */
e2aaee9cSDarrick J. Wong	inode_lock_nested(dest, I_MUTEX_NONDIR2);
e2aaee9cSDarrick J. Wong	error = break_layout(dest, false);
e2aaee9cSDarrick J. Wong	if (error) {
e2aaee9cSDarrick J. Wong		inode_unlock(src);
e2aaee9cSDarrick J. Wong		inode_unlock(dest);
e2aaee9cSDarrick J. Wong		if (error == -EWOULDBLOCK)
e2aaee9cSDarrick J. Wong			goto retry;
e2aaee9cSDarrick J. Wong		return error;
e2aaee9cSDarrick J. Wong	}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	return 0;
e2aaee9cSDarrick J. Wong}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong/*
e2aaee9cSDarrick J. Wong * Lock two inodes so that userspace cannot initiate I/O via file syscalls or
e2aaee9cSDarrick J. Wong * mmap activity.
e2aaee9cSDarrick J. Wong */
e2aaee9cSDarrick J. Wongint
e2aaee9cSDarrick J. Wongxfs_ilock2_io_mmap(
e2aaee9cSDarrick J. Wong	struct xfs_inode	*ip1,
e2aaee9cSDarrick J. Wong	struct xfs_inode	*ip2)
e2aaee9cSDarrick J. Wong{
e2aaee9cSDarrick J. Wong	int			ret;
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	ret = xfs_iolock_two_inodes_and_break_layout(VFS_I(ip1), VFS_I(ip2));
e2aaee9cSDarrick J. Wong	if (ret)
e2aaee9cSDarrick J. Wong		return ret;
e2aaee9cSDarrick J. Wong	if (ip1 == ip2)
e2aaee9cSDarrick J. Wong		xfs_ilock(ip1, XFS_MMAPLOCK_EXCL);
e2aaee9cSDarrick J. Wong	else
e2aaee9cSDarrick J. Wong		xfs_lock_two_inodes(ip1, XFS_MMAPLOCK_EXCL,
e2aaee9cSDarrick J. Wong				    ip2, XFS_MMAPLOCK_EXCL);
e2aaee9cSDarrick J. Wong	return 0;
e2aaee9cSDarrick J. Wong}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong/* Unlock both inodes to allow IO and mmap activity. */
e2aaee9cSDarrick J. Wongvoid
e2aaee9cSDarrick J. Wongxfs_iunlock2_io_mmap(
e2aaee9cSDarrick J. Wong	struct xfs_inode	*ip1,
e2aaee9cSDarrick J. Wong	struct xfs_inode	*ip2)
e2aaee9cSDarrick J. Wong{
e2aaee9cSDarrick J. Wong	bool			same_inode = (ip1 == ip2);
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	xfs_iunlock(ip2, XFS_MMAPLOCK_EXCL);
e2aaee9cSDarrick J. Wong	if (!same_inode)
e2aaee9cSDarrick J. Wong		xfs_iunlock(ip1, XFS_MMAPLOCK_EXCL);
e2aaee9cSDarrick J. Wong	inode_unlock(VFS_I(ip2));
e2aaee9cSDarrick J. Wong	if (!same_inode)
e2aaee9cSDarrick J. Wong		inode_unlock(VFS_I(ip1));
e2aaee9cSDarrick J. Wong}