fs/xfs/xfs_inode.c

0b61f8a4SDave Chinner// SPDX-License-Identifier: GPL-2.0
1da177e4SLinus Torvalds/*
3e57ecf6SOlaf Weber * Copyright (c) 2000-2006 Silicon Graphics, Inc.
7b718769SNathan Scott * All Rights Reserved.
1da177e4SLinus Torvalds */
f0e28280SJeff Layton#include <linux/iversion.h>
40ebd81dSRobert P. J. Day
1da177e4SLinus Torvalds#include "xfs.h"
a844f451SNathan Scott#include "xfs_fs.h"
70a9883cSDave Chinner#include "xfs_shared.h"
239880efSDave Chinner#include "xfs_format.h"
239880efSDave Chinner#include "xfs_log_format.h"
239880efSDave Chinner#include "xfs_trans_resv.h"
1da177e4SLinus Torvalds#include "xfs_mount.h"
3ab78df2SDarrick J. Wong#include "xfs_defer.h"
a4fbe6abSDave Chinner#include "xfs_inode.h"
c24b5dfaSDave Chinner#include "xfs_dir2.h"
c24b5dfaSDave Chinner#include "xfs_attr.h"
239880efSDave Chinner#include "xfs_trans_space.h"
239880efSDave Chinner#include "xfs_trans.h"
1da177e4SLinus Torvalds#include "xfs_buf_item.h"
a844f451SNathan Scott#include "xfs_inode_item.h"
784eb7d8SDave Chinner#include "xfs_iunlink_item.h"
a844f451SNathan Scott#include "xfs_ialloc.h"
a844f451SNathan Scott#include "xfs_bmap.h"
68988114SDave Chinner#include "xfs_bmap_util.h"
e9e899a2SDarrick J. Wong#include "xfs_errortag.h"
1da177e4SLinus Torvalds#include "xfs_error.h"
1da177e4SLinus Torvalds#include "xfs_quota.h"
2a82b8beSDavid Chinner#include "xfs_filestream.h"
0b1b213fSChristoph Hellwig#include "xfs_trace.h"
33479e05SDave Chinner#include "xfs_icache.h"
c24b5dfaSDave Chinner#include "xfs_symlink.h"
239880efSDave Chinner#include "xfs_trans_priv.h"
239880efSDave Chinner#include "xfs_log.h"
a4fbe6abSDave Chinner#include "xfs_bmap_btree.h"
aa8968f2SDarrick J. Wong#include "xfs_reflink.h"
9bbafc71SDave Chinner#include "xfs_ag.h"
01728b44SDave Chinner#include "xfs_log_priv.h"
1da177e4SLinus Torvalds
182696fbSDarrick J. Wongstruct kmem_cache *xfs_inode_cache;
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
8f04c47aSChristoph Hellwig * Used in xfs_itruncate_extents().  This is the maximum number of extents
1da177e4SLinus Torvalds * freed from a file in a single transaction.
1da177e4SLinus Torvalds */
1da177e4SLinus Torvalds#define	XFS_ITRUNC_MAX_EXTENTS	2
1da177e4SLinus Torvalds
54d7b5c1SDave ChinnerSTATIC int xfs_iunlink(struct xfs_trans *, struct xfs_inode *);
f40aadb2SDave ChinnerSTATIC int xfs_iunlink_remove(struct xfs_trans *tp, struct xfs_perag *pag,
f40aadb2SDave Chinner	struct xfs_inode *);
ab297431SZhi Yong Wu
2a0ec1d9SDave Chinner/*
2a0ec1d9SDave Chinner * helper function to extract extent size hint from inode
2a0ec1d9SDave Chinner */
2a0ec1d9SDave Chinnerxfs_extlen_t
2a0ec1d9SDave Chinnerxfs_get_extsz_hint(
2a0ec1d9SDave Chinner	struct xfs_inode	*ip)
2a0ec1d9SDave Chinner{
bdb2ed2dSChristoph Hellwig	/*
bdb2ed2dSChristoph Hellwig	 * No point in aligning allocations if we need to COW to actually
bdb2ed2dSChristoph Hellwig	 * write to them.
bdb2ed2dSChristoph Hellwig	 */
bdb2ed2dSChristoph Hellwig	if (xfs_is_always_cow_inode(ip))
bdb2ed2dSChristoph Hellwig		return 0;
db07349dSChristoph Hellwig	if ((ip->i_diflags & XFS_DIFLAG_EXTSIZE) && ip->i_extsize)
031474c2SChristoph Hellwig		return ip->i_extsize;
2a0ec1d9SDave Chinner	if (XFS_IS_REALTIME_INODE(ip))
2a0ec1d9SDave Chinner		return ip->i_mount->m_sb.sb_rextsize;
2a0ec1d9SDave Chinner	return 0;
2a0ec1d9SDave Chinner}
2a0ec1d9SDave Chinner
fa96acadSDave Chinner/*
f7ca3522SDarrick J. Wong * Helper function to extract CoW extent size hint from inode.
f7ca3522SDarrick J. Wong * Between the extent size hint and the CoW extent size hint, we
e153aa79SDarrick J. Wong * return the greater of the two.  If the value is zero (automatic),
e153aa79SDarrick J. Wong * use the default size.
f7ca3522SDarrick J. Wong */
f7ca3522SDarrick J. Wongxfs_extlen_t
f7ca3522SDarrick J. Wongxfs_get_cowextsz_hint(
f7ca3522SDarrick J. Wong	struct xfs_inode	*ip)
f7ca3522SDarrick J. Wong{
f7ca3522SDarrick J. Wong	xfs_extlen_t		a, b;
f7ca3522SDarrick J. Wong
f7ca3522SDarrick J. Wong	a = 0;
3e09ab8fSChristoph Hellwig	if (ip->i_diflags2 & XFS_DIFLAG2_COWEXTSIZE)
b33ce57dSChristoph Hellwig		a = ip->i_cowextsize;
f7ca3522SDarrick J. Wong	b = xfs_get_extsz_hint(ip);
f7ca3522SDarrick J. Wong
e153aa79SDarrick J. Wong	a = max(a, b);
e153aa79SDarrick J. Wong	if (a == 0)
e153aa79SDarrick J. Wong		return XFS_DEFAULT_COWEXTSZ_HINT;
f7ca3522SDarrick J. Wong	return a;
f7ca3522SDarrick J. Wong}
f7ca3522SDarrick J. Wong
f7ca3522SDarrick J. Wong/*
efa70be1SChristoph Hellwig * These two are wrapper routines around the xfs_ilock() routine used to
efa70be1SChristoph Hellwig * centralize some grungy code.  They are used in places that wish to lock the
efa70be1SChristoph Hellwig * inode solely for reading the extents.  The reason these places can't just
efa70be1SChristoph Hellwig * call xfs_ilock(ip, XFS_ILOCK_SHARED) is that the inode lock also guards to
efa70be1SChristoph Hellwig * bringing in of the extents from disk for a file in b-tree format.  If the
efa70be1SChristoph Hellwig * inode is in b-tree format, then we need to lock the inode exclusively until
efa70be1SChristoph Hellwig * the extents are read in.  Locking it exclusively all the time would limit
efa70be1SChristoph Hellwig * our parallelism unnecessarily, though.  What we do instead is check to see
efa70be1SChristoph Hellwig * if the extents have been read in yet, and only lock the inode exclusively
efa70be1SChristoph Hellwig * if they have not.
fa96acadSDave Chinner *
efa70be1SChristoph Hellwig * The functions return a value which should be given to the corresponding
01f4f327SChristoph Hellwig * xfs_iunlock() call.
fa96acadSDave Chinner */
fa96acadSDave Chinneruint
309ecac8SChristoph Hellwigxfs_ilock_data_map_shared(
309ecac8SChristoph Hellwig	struct xfs_inode	*ip)
fa96acadSDave Chinner{
309ecac8SChristoph Hellwig	uint			lock_mode = XFS_ILOCK_SHARED;
fa96acadSDave Chinner
b2197a36SChristoph Hellwig	if (xfs_need_iread_extents(&ip->i_df))
fa96acadSDave Chinner		lock_mode = XFS_ILOCK_EXCL;
fa96acadSDave Chinner	xfs_ilock(ip, lock_mode);
fa96acadSDave Chinner	return lock_mode;
fa96acadSDave Chinner}
fa96acadSDave Chinner
efa70be1SChristoph Hellwiguint
efa70be1SChristoph Hellwigxfs_ilock_attr_map_shared(
efa70be1SChristoph Hellwig	struct xfs_inode	*ip)
fa96acadSDave Chinner{
efa70be1SChristoph Hellwig	uint			lock_mode = XFS_ILOCK_SHARED;
efa70be1SChristoph Hellwig
932b42c6SDarrick J. Wong	if (xfs_inode_has_attr_fork(ip) && xfs_need_iread_extents(&ip->i_af))
efa70be1SChristoph Hellwig		lock_mode = XFS_ILOCK_EXCL;
efa70be1SChristoph Hellwig	xfs_ilock(ip, lock_mode);
efa70be1SChristoph Hellwig	return lock_mode;
fa96acadSDave Chinner}
fa96acadSDave Chinner
fa96acadSDave Chinner/*
ca76a761SKaixu Xia * You can't set both SHARED and EXCL for the same lock,
ca76a761SKaixu Xia * and only XFS_IOLOCK_SHARED, XFS_IOLOCK_EXCL, XFS_MMAPLOCK_SHARED,
ca76a761SKaixu Xia * XFS_MMAPLOCK_EXCL, XFS_ILOCK_SHARED, XFS_ILOCK_EXCL are valid values
ca76a761SKaixu Xia * to set in lock_flags.
ca76a761SKaixu Xia */
ca76a761SKaixu Xiastatic inline void
ca76a761SKaixu Xiaxfs_lock_flags_assert(
ca76a761SKaixu Xia	uint		lock_flags)
ca76a761SKaixu Xia{
ca76a761SKaixu Xia	ASSERT((lock_flags & (XFS_IOLOCK_SHARED | XFS_IOLOCK_EXCL)) !=
ca76a761SKaixu Xia		(XFS_IOLOCK_SHARED | XFS_IOLOCK_EXCL));
ca76a761SKaixu Xia	ASSERT((lock_flags & (XFS_MMAPLOCK_SHARED | XFS_MMAPLOCK_EXCL)) !=
ca76a761SKaixu Xia		(XFS_MMAPLOCK_SHARED | XFS_MMAPLOCK_EXCL));
ca76a761SKaixu Xia	ASSERT((lock_flags & (XFS_ILOCK_SHARED | XFS_ILOCK_EXCL)) !=
ca76a761SKaixu Xia		(XFS_ILOCK_SHARED | XFS_ILOCK_EXCL));
ca76a761SKaixu Xia	ASSERT((lock_flags & ~(XFS_LOCK_MASK | XFS_LOCK_SUBCLASS_MASK)) == 0);
ca76a761SKaixu Xia	ASSERT(lock_flags != 0);
ca76a761SKaixu Xia}
ca76a761SKaixu Xia
ca76a761SKaixu Xia/*
65523218SChristoph Hellwig * In addition to i_rwsem in the VFS inode, the xfs inode contains 2
2433480aSJan Kara * multi-reader locks: invalidate_lock and the i_lock.  This routine allows
65523218SChristoph Hellwig * various combinations of the locks to be obtained.
fa96acadSDave Chinner *
653c60b6SDave Chinner * The 3 locks should always be ordered so that the IO lock is obtained first,
653c60b6SDave Chinner * the mmap lock second and the ilock last in order to prevent deadlock.
fa96acadSDave Chinner *
653c60b6SDave Chinner * Basic locking order:
653c60b6SDave Chinner *
2433480aSJan Kara * i_rwsem -> invalidate_lock -> page_lock -> i_ilock
653c60b6SDave Chinner *
c1e8d7c6SMichel Lespinasse * mmap_lock locking order:
653c60b6SDave Chinner *
c1e8d7c6SMichel Lespinasse * i_rwsem -> page lock -> mmap_lock
2433480aSJan Kara * mmap_lock -> invalidate_lock -> page_lock
653c60b6SDave Chinner *
c1e8d7c6SMichel Lespinasse * The difference in mmap_lock locking order mean that we cannot hold the
2433480aSJan Kara * invalidate_lock over syscall based read(2)/write(2) based IO. These IO paths
2433480aSJan Kara * can fault in pages during copy in/out (for buffered IO) or require the
2433480aSJan Kara * mmap_lock in get_user_pages() to map the user pages into the kernel address
2433480aSJan Kara * space for direct IO. Similarly the i_rwsem cannot be taken inside a page
2433480aSJan Kara * fault because page faults already hold the mmap_lock.
653c60b6SDave Chinner *
653c60b6SDave Chinner * Hence to serialise fully against both syscall and mmap based IO, we need to
2433480aSJan Kara * take both the i_rwsem and the invalidate_lock. These locks should *only* be
2433480aSJan Kara * both taken in places where we need to invalidate the page cache in a race
653c60b6SDave Chinner * free manner (e.g. truncate, hole punch and other extent manipulation
653c60b6SDave Chinner * functions).
fa96acadSDave Chinner */
fa96acadSDave Chinnervoid
fa96acadSDave Chinnerxfs_ilock(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
fa96acadSDave Chinner	trace_xfs_ilock(ip, lock_flags, _RET_IP_);
fa96acadSDave Chinner
ca76a761SKaixu Xia	xfs_lock_flags_assert(lock_flags);
fa96acadSDave Chinner
65523218SChristoph Hellwig	if (lock_flags & XFS_IOLOCK_EXCL) {
65523218SChristoph Hellwig		down_write_nested(&VFS_I(ip)->i_rwsem,
65523218SChristoph Hellwig				  XFS_IOLOCK_DEP(lock_flags));
65523218SChristoph Hellwig	} else if (lock_flags & XFS_IOLOCK_SHARED) {
65523218SChristoph Hellwig		down_read_nested(&VFS_I(ip)->i_rwsem,
65523218SChristoph Hellwig				 XFS_IOLOCK_DEP(lock_flags));
65523218SChristoph Hellwig	}
fa96acadSDave Chinner
2433480aSJan Kara	if (lock_flags & XFS_MMAPLOCK_EXCL) {
2433480aSJan Kara		down_write_nested(&VFS_I(ip)->i_mapping->invalidate_lock,
2433480aSJan Kara				  XFS_MMAPLOCK_DEP(lock_flags));
2433480aSJan Kara	} else if (lock_flags & XFS_MMAPLOCK_SHARED) {
2433480aSJan Kara		down_read_nested(&VFS_I(ip)->i_mapping->invalidate_lock,
2433480aSJan Kara				 XFS_MMAPLOCK_DEP(lock_flags));
2433480aSJan Kara	}
653c60b6SDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_ILOCK_EXCL)
fa96acadSDave Chinner		mrupdate_nested(&ip->i_lock, XFS_ILOCK_DEP(lock_flags));
fa96acadSDave Chinner	else if (lock_flags & XFS_ILOCK_SHARED)
fa96acadSDave Chinner		mraccess_nested(&ip->i_lock, XFS_ILOCK_DEP(lock_flags));
fa96acadSDave Chinner}
fa96acadSDave Chinner
fa96acadSDave Chinner/*
fa96acadSDave Chinner * This is just like xfs_ilock(), except that the caller
fa96acadSDave Chinner * is guaranteed not to sleep.  It returns 1 if it gets
fa96acadSDave Chinner * the requested locks and 0 otherwise.  If the IO lock is
fa96acadSDave Chinner * obtained but the inode lock cannot be, then the IO lock
fa96acadSDave Chinner * is dropped before returning.
fa96acadSDave Chinner *
fa96acadSDave Chinner * ip -- the inode being locked
fa96acadSDave Chinner * lock_flags -- this parameter indicates the inode's locks to be
fa96acadSDave Chinner *       to be locked.  See the comment for xfs_ilock() for a list
fa96acadSDave Chinner *	 of valid values.
fa96acadSDave Chinner */
fa96acadSDave Chinnerint
fa96acadSDave Chinnerxfs_ilock_nowait(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
fa96acadSDave Chinner	trace_xfs_ilock_nowait(ip, lock_flags, _RET_IP_);
fa96acadSDave Chinner
ca76a761SKaixu Xia	xfs_lock_flags_assert(lock_flags);
fa96acadSDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_IOLOCK_EXCL) {
65523218SChristoph Hellwig		if (!down_write_trylock(&VFS_I(ip)->i_rwsem))
fa96acadSDave Chinner			goto out;
fa96acadSDave Chinner	} else if (lock_flags & XFS_IOLOCK_SHARED) {
65523218SChristoph Hellwig		if (!down_read_trylock(&VFS_I(ip)->i_rwsem))
fa96acadSDave Chinner			goto out;
fa96acadSDave Chinner	}
653c60b6SDave Chinner
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL) {
2433480aSJan Kara		if (!down_write_trylock(&VFS_I(ip)->i_mapping->invalidate_lock))
653c60b6SDave Chinner			goto out_undo_iolock;
653c60b6SDave Chinner	} else if (lock_flags & XFS_MMAPLOCK_SHARED) {
2433480aSJan Kara		if (!down_read_trylock(&VFS_I(ip)->i_mapping->invalidate_lock))
653c60b6SDave Chinner			goto out_undo_iolock;
653c60b6SDave Chinner	}
653c60b6SDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_ILOCK_EXCL) {
fa96acadSDave Chinner		if (!mrtryupdate(&ip->i_lock))
653c60b6SDave Chinner			goto out_undo_mmaplock;
fa96acadSDave Chinner	} else if (lock_flags & XFS_ILOCK_SHARED) {
fa96acadSDave Chinner		if (!mrtryaccess(&ip->i_lock))
653c60b6SDave Chinner			goto out_undo_mmaplock;
fa96acadSDave Chinner	}
fa96acadSDave Chinner	return 1;
fa96acadSDave Chinner
653c60b6SDave Chinnerout_undo_mmaplock:
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL)
2433480aSJan Kara		up_write(&VFS_I(ip)->i_mapping->invalidate_lock);
653c60b6SDave Chinner	else if (lock_flags & XFS_MMAPLOCK_SHARED)
2433480aSJan Kara		up_read(&VFS_I(ip)->i_mapping->invalidate_lock);
fa96acadSDave Chinnerout_undo_iolock:
fa96acadSDave Chinner	if (lock_flags & XFS_IOLOCK_EXCL)
65523218SChristoph Hellwig		up_write(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner	else if (lock_flags & XFS_IOLOCK_SHARED)
65523218SChristoph Hellwig		up_read(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinnerout:
fa96acadSDave Chinner	return 0;
fa96acadSDave Chinner}
fa96acadSDave Chinner
fa96acadSDave Chinner/*
fa96acadSDave Chinner * xfs_iunlock() is used to drop the inode locks acquired with
fa96acadSDave Chinner * xfs_ilock() and xfs_ilock_nowait().  The caller must pass
fa96acadSDave Chinner * in the flags given to xfs_ilock() or xfs_ilock_nowait() so
fa96acadSDave Chinner * that we know which locks to drop.
fa96acadSDave Chinner *
fa96acadSDave Chinner * ip -- the inode being unlocked
fa96acadSDave Chinner * lock_flags -- this parameter indicates the inode's locks to be
fa96acadSDave Chinner *       to be unlocked.  See the comment for xfs_ilock() for a list
fa96acadSDave Chinner *	 of valid values for this parameter.
fa96acadSDave Chinner *
fa96acadSDave Chinner */
fa96acadSDave Chinnervoid
fa96acadSDave Chinnerxfs_iunlock(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
ca76a761SKaixu Xia	xfs_lock_flags_assert(lock_flags);
fa96acadSDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_IOLOCK_EXCL)
65523218SChristoph Hellwig		up_write(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner	else if (lock_flags & XFS_IOLOCK_SHARED)
65523218SChristoph Hellwig		up_read(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL)
2433480aSJan Kara		up_write(&VFS_I(ip)->i_mapping->invalidate_lock);
653c60b6SDave Chinner	else if (lock_flags & XFS_MMAPLOCK_SHARED)
2433480aSJan Kara		up_read(&VFS_I(ip)->i_mapping->invalidate_lock);
653c60b6SDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_ILOCK_EXCL)
fa96acadSDave Chinner		mrunlock_excl(&ip->i_lock);
fa96acadSDave Chinner	else if (lock_flags & XFS_ILOCK_SHARED)
fa96acadSDave Chinner		mrunlock_shared(&ip->i_lock);
fa96acadSDave Chinner
fa96acadSDave Chinner	trace_xfs_iunlock(ip, lock_flags, _RET_IP_);
fa96acadSDave Chinner}
fa96acadSDave Chinner
fa96acadSDave Chinner/*
fa96acadSDave Chinner * give up write locks.  the i/o lock cannot be held nested
fa96acadSDave Chinner * if it is being demoted.
fa96acadSDave Chinner */
fa96acadSDave Chinnervoid
fa96acadSDave Chinnerxfs_ilock_demote(
fa96acadSDave Chinner	xfs_inode_t		*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
653c60b6SDave Chinner	ASSERT(lock_flags & (XFS_IOLOCK_EXCL|XFS_MMAPLOCK_EXCL|XFS_ILOCK_EXCL));
653c60b6SDave Chinner	ASSERT((lock_flags &
653c60b6SDave Chinner		~(XFS_IOLOCK_EXCL|XFS_MMAPLOCK_EXCL|XFS_ILOCK_EXCL)) == 0);
fa96acadSDave Chinner
fa96acadSDave Chinner	if (lock_flags & XFS_ILOCK_EXCL)
fa96acadSDave Chinner		mrdemote(&ip->i_lock);
653c60b6SDave Chinner	if (lock_flags & XFS_MMAPLOCK_EXCL)
2433480aSJan Kara		downgrade_write(&VFS_I(ip)->i_mapping->invalidate_lock);
fa96acadSDave Chinner	if (lock_flags & XFS_IOLOCK_EXCL)
65523218SChristoph Hellwig		downgrade_write(&VFS_I(ip)->i_rwsem);
fa96acadSDave Chinner
fa96acadSDave Chinner	trace_xfs_ilock_demote(ip, lock_flags, _RET_IP_);
fa96acadSDave Chinner}
fa96acadSDave Chinner
742ae1e3SDave Chinner#if defined(DEBUG) || defined(XFS_WARN)
e31cbde7SPavel Reichlstatic inline bool
e31cbde7SPavel Reichl__xfs_rwsem_islocked(
e31cbde7SPavel Reichl	struct rw_semaphore	*rwsem,
e31cbde7SPavel Reichl	bool			shared)
e31cbde7SPavel Reichl{
e31cbde7SPavel Reichl	if (!debug_locks)
e31cbde7SPavel Reichl		return rwsem_is_locked(rwsem);
e31cbde7SPavel Reichl
e31cbde7SPavel Reichl	if (!shared)
e31cbde7SPavel Reichl		return lockdep_is_held_type(rwsem, 0);
e31cbde7SPavel Reichl
e31cbde7SPavel Reichl	/*
e31cbde7SPavel Reichl	 * We are checking that the lock is held at least in shared
e31cbde7SPavel Reichl	 * mode but don't care that it might be held exclusively
e31cbde7SPavel Reichl	 * (i.e. shared | excl). Hence we check if the lock is held
e31cbde7SPavel Reichl	 * in any mode rather than an explicit shared mode.
e31cbde7SPavel Reichl	 */
e31cbde7SPavel Reichl	return lockdep_is_held_type(rwsem, -1);
e31cbde7SPavel Reichl}
e31cbde7SPavel Reichl
e31cbde7SPavel Reichlbool
fa96acadSDave Chinnerxfs_isilocked(
e31cbde7SPavel Reichl	struct xfs_inode	*ip,
fa96acadSDave Chinner	uint			lock_flags)
fa96acadSDave Chinner{
fa96acadSDave Chinner	if (lock_flags & (XFS_ILOCK_EXCL|XFS_ILOCK_SHARED)) {
fa96acadSDave Chinner		if (!(lock_flags & XFS_ILOCK_SHARED))
fa96acadSDave Chinner			return !!ip->i_lock.mr_writer;
fa96acadSDave Chinner		return rwsem_is_locked(&ip->i_lock.mr_lock);
fa96acadSDave Chinner	}
fa96acadSDave Chinner
653c60b6SDave Chinner	if (lock_flags & (XFS_MMAPLOCK_EXCL|XFS_MMAPLOCK_SHARED)) {
82af8806SKaixu Xia		return __xfs_rwsem_islocked(&VFS_I(ip)->i_mapping->invalidate_lock,
82af8806SKaixu Xia				(lock_flags & XFS_MMAPLOCK_SHARED));
653c60b6SDave Chinner	}
653c60b6SDave Chinner
fa96acadSDave Chinner	if (lock_flags & (XFS_IOLOCK_EXCL | XFS_IOLOCK_SHARED)) {
e31cbde7SPavel Reichl		return __xfs_rwsem_islocked(&VFS_I(ip)->i_rwsem,
e31cbde7SPavel Reichl				(lock_flags & XFS_IOLOCK_SHARED));
fa96acadSDave Chinner	}
fa96acadSDave Chinner
fa96acadSDave Chinner	ASSERT(0);
e31cbde7SPavel Reichl	return false;
fa96acadSDave Chinner}
fa96acadSDave Chinner#endif
fa96acadSDave Chinner
b6a9947eSDave Chinner/*
b6a9947eSDave Chinner * xfs_lockdep_subclass_ok() is only used in an ASSERT, so is only called when
b6a9947eSDave Chinner * DEBUG or XFS_WARN is set. And MAX_LOCKDEP_SUBCLASSES is then only defined
b6a9947eSDave Chinner * when CONFIG_LOCKDEP is set. Hence the complex define below to avoid build
b6a9947eSDave Chinner * errors and warnings.
b6a9947eSDave Chinner */
b6a9947eSDave Chinner#if (defined(DEBUG) || defined(XFS_WARN)) && defined(CONFIG_LOCKDEP)
3403ccc0SDave Chinnerstatic bool
3403ccc0SDave Chinnerxfs_lockdep_subclass_ok(
3403ccc0SDave Chinner	int subclass)
3403ccc0SDave Chinner{
3403ccc0SDave Chinner	return subclass < MAX_LOCKDEP_SUBCLASSES;
3403ccc0SDave Chinner}
3403ccc0SDave Chinner#else
3403ccc0SDave Chinner#define xfs_lockdep_subclass_ok(subclass)	(true)
3403ccc0SDave Chinner#endif
3403ccc0SDave Chinner
c24b5dfaSDave Chinner/*
653c60b6SDave Chinner * Bump the subclass so xfs_lock_inodes() acquires each lock with a different
0952c818SDave Chinner * value. This can be called for any type of inode lock combination, including
0952c818SDave Chinner * parent locking. Care must be taken to ensure we don't overrun the subclass
0952c818SDave Chinner * storage fields in the class mask we build.
c24b5dfaSDave Chinner */
a1033753SDave Chinnerstatic inline uint
a1033753SDave Chinnerxfs_lock_inumorder(
a1033753SDave Chinner	uint	lock_mode,
a1033753SDave Chinner	uint	subclass)
c24b5dfaSDave Chinner{
a1033753SDave Chinner	uint	class = 0;
0952c818SDave Chinner
0952c818SDave Chinner	ASSERT(!(lock_mode & (XFS_ILOCK_PARENT | XFS_ILOCK_RTBITMAP |
0952c818SDave Chinner			      XFS_ILOCK_RTSUM)));
3403ccc0SDave Chinner	ASSERT(xfs_lockdep_subclass_ok(subclass));
0952c818SDave Chinner
653c60b6SDave Chinner	if (lock_mode & (XFS_IOLOCK_SHARED|XFS_IOLOCK_EXCL)) {
0952c818SDave Chinner		ASSERT(subclass <= XFS_IOLOCK_MAX_SUBCLASS);
0952c818SDave Chinner		class += subclass << XFS_IOLOCK_SHIFT;
653c60b6SDave Chinner	}
653c60b6SDave Chinner
653c60b6SDave Chinner	if (lock_mode & (XFS_MMAPLOCK_SHARED|XFS_MMAPLOCK_EXCL)) {
0952c818SDave Chinner		ASSERT(subclass <= XFS_MMAPLOCK_MAX_SUBCLASS);
0952c818SDave Chinner		class += subclass << XFS_MMAPLOCK_SHIFT;
653c60b6SDave Chinner	}
653c60b6SDave Chinner
0952c818SDave Chinner	if (lock_mode & (XFS_ILOCK_SHARED|XFS_ILOCK_EXCL)) {
0952c818SDave Chinner		ASSERT(subclass <= XFS_ILOCK_MAX_SUBCLASS);
0952c818SDave Chinner		class += subclass << XFS_ILOCK_SHIFT;
0952c818SDave Chinner	}
c24b5dfaSDave Chinner
0952c818SDave Chinner	return (lock_mode & ~XFS_LOCK_SUBCLASS_MASK) | class;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner/*
95afcf5cSDave Chinner * The following routine will lock n inodes in exclusive mode.  We assume the
95afcf5cSDave Chinner * caller calls us with the inodes in i_ino order.
c24b5dfaSDave Chinner *
95afcf5cSDave Chinner * We need to detect deadlock where an inode that we lock is in the AIL and we
95afcf5cSDave Chinner * start waiting for another inode that is locked by a thread in a long running
95afcf5cSDave Chinner * transaction (such as truncate). This can result in deadlock since the long
95afcf5cSDave Chinner * running trans might need to wait for the inode we just locked in order to
95afcf5cSDave Chinner * push the tail and free space in the log.
0952c818SDave Chinner *
0952c818SDave Chinner * xfs_lock_inodes() can only be used to lock one type of lock at a time -
0952c818SDave Chinner * the iolock, the mmaplock or the ilock, but not more than one at a time. If we
0952c818SDave Chinner * lock more than one at a time, lockdep will report false positives saying we
0952c818SDave Chinner * have violated locking orders.
c24b5dfaSDave Chinner */
0d5a75e9SEric Sandeenstatic void
c24b5dfaSDave Chinnerxfs_lock_inodes(
efe2330fSChristoph Hellwig	struct xfs_inode	**ips,
c24b5dfaSDave Chinner	int			inodes,
c24b5dfaSDave Chinner	uint			lock_mode)
c24b5dfaSDave Chinner{
a1033753SDave Chinner	int			attempts = 0;
a1033753SDave Chinner	uint			i;
a1033753SDave Chinner	int			j;
a1033753SDave Chinner	bool			try_lock;
efe2330fSChristoph Hellwig	struct xfs_log_item	*lp;
c24b5dfaSDave Chinner
0952c818SDave Chinner	/*
0952c818SDave Chinner	 * Currently supports between 2 and 5 inodes with exclusive locking.  We
0952c818SDave Chinner	 * support an arbitrary depth of locking here, but absolute limits on
b63da6c8SRandy Dunlap	 * inodes depend on the type of locking and the limits placed by
0952c818SDave Chinner	 * lockdep annotations in xfs_lock_inumorder.  These are all checked by
0952c818SDave Chinner	 * the asserts.
0952c818SDave Chinner	 */
95afcf5cSDave Chinner	ASSERT(ips && inodes >= 2 && inodes <= 5);
0952c818SDave Chinner	ASSERT(lock_mode & (XFS_IOLOCK_EXCL | XFS_MMAPLOCK_EXCL |
0952c818SDave Chinner			    XFS_ILOCK_EXCL));
0952c818SDave Chinner	ASSERT(!(lock_mode & (XFS_IOLOCK_SHARED | XFS_MMAPLOCK_SHARED |
0952c818SDave Chinner			      XFS_ILOCK_SHARED)));
0952c818SDave Chinner	ASSERT(!(lock_mode & XFS_MMAPLOCK_EXCL) ||
0952c818SDave Chinner		inodes <= XFS_MMAPLOCK_MAX_SUBCLASS + 1);
0952c818SDave Chinner	ASSERT(!(lock_mode & XFS_ILOCK_EXCL) ||
0952c818SDave Chinner		inodes <= XFS_ILOCK_MAX_SUBCLASS + 1);
0952c818SDave Chinner
0952c818SDave Chinner	if (lock_mode & XFS_IOLOCK_EXCL) {
0952c818SDave Chinner		ASSERT(!(lock_mode & (XFS_MMAPLOCK_EXCL | XFS_ILOCK_EXCL)));
0952c818SDave Chinner	} else if (lock_mode & XFS_MMAPLOCK_EXCL)
0952c818SDave Chinner		ASSERT(!(lock_mode & XFS_ILOCK_EXCL));
c24b5dfaSDave Chinner
c24b5dfaSDave Chinneragain:
a1033753SDave Chinner	try_lock = false;
a1033753SDave Chinner	i = 0;
c24b5dfaSDave Chinner	for (; i < inodes; i++) {
c24b5dfaSDave Chinner		ASSERT(ips[i]);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		if (i && (ips[i] == ips[i - 1]))	/* Already locked */
c24b5dfaSDave Chinner			continue;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		/*
95afcf5cSDave Chinner		 * If try_lock is not set yet, make sure all locked inodes are
95afcf5cSDave Chinner		 * not in the AIL.  If any are, set try_lock to be used later.
c24b5dfaSDave Chinner		 */
c24b5dfaSDave Chinner		if (!try_lock) {
c24b5dfaSDave Chinner			for (j = (i - 1); j >= 0 && !try_lock; j--) {
b3b14aacSChristoph Hellwig				lp = &ips[j]->i_itemp->ili_item;
22525c17SDave Chinner				if (lp && test_bit(XFS_LI_IN_AIL, &lp->li_flags))
a1033753SDave Chinner					try_lock = true;
c24b5dfaSDave Chinner			}
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		/*
c24b5dfaSDave Chinner		 * If any of the previous locks we have locked is in the AIL,
c24b5dfaSDave Chinner		 * we must TRY to get the second and subsequent locks. If
c24b5dfaSDave Chinner		 * we can't get any, we must release all we have
c24b5dfaSDave Chinner		 * and try again.
c24b5dfaSDave Chinner		 */
95afcf5cSDave Chinner		if (!try_lock) {
95afcf5cSDave Chinner			xfs_ilock(ips[i], xfs_lock_inumorder(lock_mode, i));
95afcf5cSDave Chinner			continue;
95afcf5cSDave Chinner		}
c24b5dfaSDave Chinner
95afcf5cSDave Chinner		/* try_lock means we have an inode locked that is in the AIL. */
c24b5dfaSDave Chinner		ASSERT(i != 0);
95afcf5cSDave Chinner		if (xfs_ilock_nowait(ips[i], xfs_lock_inumorder(lock_mode, i)))
95afcf5cSDave Chinner			continue;
95afcf5cSDave Chinner
95afcf5cSDave Chinner		/*
95afcf5cSDave Chinner		 * Unlock all previous guys and try again.  xfs_iunlock will try
95afcf5cSDave Chinner		 * to push the tail if the inode is in the AIL.
95afcf5cSDave Chinner		 */
c24b5dfaSDave Chinner		attempts++;
c24b5dfaSDave Chinner		for (j = i - 1; j >= 0; j--) {
c24b5dfaSDave Chinner			/*
95afcf5cSDave Chinner			 * Check to see if we've already unlocked this one.  Not
95afcf5cSDave Chinner			 * the first one going back, and the inode ptr is the
95afcf5cSDave Chinner			 * same.
c24b5dfaSDave Chinner			 */
95afcf5cSDave Chinner			if (j != (i - 1) && ips[j] == ips[j + 1])
c24b5dfaSDave Chinner				continue;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner			xfs_iunlock(ips[j], lock_mode);
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		if ((attempts % 5) == 0) {
c24b5dfaSDave Chinner			delay(1); /* Don't just spin the CPU */
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner		goto again;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner/*
d2c292d8SJan Kara * xfs_lock_two_inodes() can only be used to lock ilock. The iolock and
d2c292d8SJan Kara * mmaplock must be double-locked separately since we use i_rwsem and
d2c292d8SJan Kara * invalidate_lock for that. We now support taking one lock EXCL and the
d2c292d8SJan Kara * other SHARED.
c24b5dfaSDave Chinner */
c24b5dfaSDave Chinnervoid
c24b5dfaSDave Chinnerxfs_lock_two_inodes(
7c2d238aSDarrick J. Wong	struct xfs_inode	*ip0,
7c2d238aSDarrick J. Wong	uint			ip0_mode,
7c2d238aSDarrick J. Wong	struct xfs_inode	*ip1,
7c2d238aSDarrick J. Wong	uint			ip1_mode)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	int			attempts = 0;
efe2330fSChristoph Hellwig	struct xfs_log_item	*lp;
c24b5dfaSDave Chinner
7c2d238aSDarrick J. Wong	ASSERT(hweight32(ip0_mode) == 1);
7c2d238aSDarrick J. Wong	ASSERT(hweight32(ip1_mode) == 1);
7c2d238aSDarrick J. Wong	ASSERT(!(ip0_mode & (XFS_IOLOCK_SHARED|XFS_IOLOCK_EXCL)));
7c2d238aSDarrick J. Wong	ASSERT(!(ip1_mode & (XFS_IOLOCK_SHARED|XFS_IOLOCK_EXCL)));
d2c292d8SJan Kara	ASSERT(!(ip0_mode & (XFS_MMAPLOCK_SHARED|XFS_MMAPLOCK_EXCL)));
d2c292d8SJan Kara	ASSERT(!(ip1_mode & (XFS_MMAPLOCK_SHARED|XFS_MMAPLOCK_EXCL)));
c24b5dfaSDave Chinner	ASSERT(ip0->i_ino != ip1->i_ino);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (ip0->i_ino > ip1->i_ino) {
2a09b575SChangcheng Deng		swap(ip0, ip1);
2a09b575SChangcheng Deng		swap(ip0_mode, ip1_mode);
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner again:
7c2d238aSDarrick J. Wong	xfs_ilock(ip0, xfs_lock_inumorder(ip0_mode, 0));
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If the first lock we have locked is in the AIL, we must TRY to get
c24b5dfaSDave Chinner	 * the second lock. If we can't get it, we must release the first one
c24b5dfaSDave Chinner	 * and try again.
c24b5dfaSDave Chinner	 */
b3b14aacSChristoph Hellwig	lp = &ip0->i_itemp->ili_item;
22525c17SDave Chinner	if (lp && test_bit(XFS_LI_IN_AIL, &lp->li_flags)) {
7c2d238aSDarrick J. Wong		if (!xfs_ilock_nowait(ip1, xfs_lock_inumorder(ip1_mode, 1))) {
7c2d238aSDarrick J. Wong			xfs_iunlock(ip0, ip0_mode);
c24b5dfaSDave Chinner			if ((++attempts % 5) == 0)
c24b5dfaSDave Chinner				delay(1); /* Don't just spin the CPU */
c24b5dfaSDave Chinner			goto again;
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner	} else {
7c2d238aSDarrick J. Wong		xfs_ilock(ip1, xfs_lock_inumorder(ip1_mode, 1));
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
1da177e4SLinus Torvaldsuint
1da177e4SLinus Torvaldsxfs_ip2xflags(
58f88ca2SDave Chinner	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
4422501dSChristoph Hellwig	uint			flags = 0;
1da177e4SLinus Torvalds
4422501dSChristoph Hellwig	if (ip->i_diflags & XFS_DIFLAG_ANY) {
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_REALTIME)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_REALTIME;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_PREALLOC)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_PREALLOC;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_IMMUTABLE)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_IMMUTABLE;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_APPEND)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_APPEND;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_SYNC)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_SYNC;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_NOATIME)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_NOATIME;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_NODUMP)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_NODUMP;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_RTINHERIT)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_RTINHERIT;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_PROJINHERIT)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_PROJINHERIT;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_NOSYMLINKS)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_NOSYMLINKS;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_EXTSIZE)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_EXTSIZE;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_EXTSZINHERIT)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_EXTSZINHERIT;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_NODEFRAG)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_NODEFRAG;
4422501dSChristoph Hellwig		if (ip->i_diflags & XFS_DIFLAG_FILESTREAM)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_FILESTREAM;
4422501dSChristoph Hellwig	}
4422501dSChristoph Hellwig
4422501dSChristoph Hellwig	if (ip->i_diflags2 & XFS_DIFLAG2_ANY) {
4422501dSChristoph Hellwig		if (ip->i_diflags2 & XFS_DIFLAG2_DAX)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_DAX;
4422501dSChristoph Hellwig		if (ip->i_diflags2 & XFS_DIFLAG2_COWEXTSIZE)
4422501dSChristoph Hellwig			flags |= FS_XFLAG_COWEXTSIZE;
4422501dSChristoph Hellwig	}
4422501dSChristoph Hellwig
932b42c6SDarrick J. Wong	if (xfs_inode_has_attr_fork(ip))
4422501dSChristoph Hellwig		flags |= FS_XFLAG_HASATTR;
4422501dSChristoph Hellwig	return flags;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
c24b5dfaSDave Chinner * Lookups up an inode from "name". If ci_name is not NULL, then a CI match
c24b5dfaSDave Chinner * is allowed, otherwise it has to be an exact match. If a CI match is found,
c24b5dfaSDave Chinner * ci_name->name will point to a the actual name (caller must free) or
c24b5dfaSDave Chinner * will be set to NULL if an exact match is found.
c24b5dfaSDave Chinner */
c24b5dfaSDave Chinnerint
c24b5dfaSDave Chinnerxfs_lookup(
996b2329SDarrick J. Wong	struct xfs_inode	*dp,
996b2329SDarrick J. Wong	const struct xfs_name	*name,
996b2329SDarrick J. Wong	struct xfs_inode	**ipp,
c24b5dfaSDave Chinner	struct xfs_name		*ci_name)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	xfs_ino_t		inum;
c24b5dfaSDave Chinner	int			error;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	trace_xfs_lookup(dp, name);
c24b5dfaSDave Chinner
75c8c50fSDave Chinner	if (xfs_is_shutdown(dp->i_mount))
2451337dSDave Chinner		return -EIO;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	error = xfs_dir_lookup(NULL, dp, name, &inum, ci_name);
c24b5dfaSDave Chinner	if (error)
dbad7c99SDave Chinner		goto out_unlock;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	error = xfs_iget(dp->i_mount, NULL, inum, 0, 0, ipp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto out_free_name;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinnerout_free_name:
c24b5dfaSDave Chinner	if (ci_name)
c24b5dfaSDave Chinner		kmem_free(ci_name->name);
dbad7c99SDave Chinnerout_unlock:
c24b5dfaSDave Chinner	*ipp = NULL;
c24b5dfaSDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
8a569d71SDarrick J. Wong/* Propagate di_flags from a parent inode to a child inode. */
8a569d71SDarrick J. Wongstatic void
8a569d71SDarrick J. Wongxfs_inode_inherit_flags(
8a569d71SDarrick J. Wong	struct xfs_inode	*ip,
8a569d71SDarrick J. Wong	const struct xfs_inode	*pip)
8a569d71SDarrick J. Wong{
8a569d71SDarrick J. Wong	unsigned int		di_flags = 0;
603f000bSDarrick J. Wong	xfs_failaddr_t		failaddr;
8a569d71SDarrick J. Wong	umode_t			mode = VFS_I(ip)->i_mode;
8a569d71SDarrick J. Wong
8a569d71SDarrick J. Wong	if (S_ISDIR(mode)) {
db07349dSChristoph Hellwig		if (pip->i_diflags & XFS_DIFLAG_RTINHERIT)
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_RTINHERIT;
db07349dSChristoph Hellwig		if (pip->i_diflags & XFS_DIFLAG_EXTSZINHERIT) {
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_EXTSZINHERIT;
031474c2SChristoph Hellwig			ip->i_extsize = pip->i_extsize;
8a569d71SDarrick J. Wong		}
db07349dSChristoph Hellwig		if (pip->i_diflags & XFS_DIFLAG_PROJINHERIT)
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_PROJINHERIT;
8a569d71SDarrick J. Wong	} else if (S_ISREG(mode)) {
db07349dSChristoph Hellwig		if ((pip->i_diflags & XFS_DIFLAG_RTINHERIT) &&
38c26bfdSDave Chinner		    xfs_has_realtime(ip->i_mount))
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_REALTIME;
db07349dSChristoph Hellwig		if (pip->i_diflags & XFS_DIFLAG_EXTSZINHERIT) {
8a569d71SDarrick J. Wong			di_flags |= XFS_DIFLAG_EXTSIZE;
031474c2SChristoph Hellwig			ip->i_extsize = pip->i_extsize;
8a569d71SDarrick J. Wong		}
8a569d71SDarrick J. Wong	}
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_NOATIME) &&
8a569d71SDarrick J. Wong	    xfs_inherit_noatime)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_NOATIME;
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_NODUMP) &&
8a569d71SDarrick J. Wong	    xfs_inherit_nodump)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_NODUMP;
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_SYNC) &&
8a569d71SDarrick J. Wong	    xfs_inherit_sync)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_SYNC;
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_NOSYMLINKS) &&
8a569d71SDarrick J. Wong	    xfs_inherit_nosymlinks)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_NOSYMLINKS;
db07349dSChristoph Hellwig	if ((pip->i_diflags & XFS_DIFLAG_NODEFRAG) &&
8a569d71SDarrick J. Wong	    xfs_inherit_nodefrag)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_NODEFRAG;
db07349dSChristoph Hellwig	if (pip->i_diflags & XFS_DIFLAG_FILESTREAM)
8a569d71SDarrick J. Wong		di_flags |= XFS_DIFLAG_FILESTREAM;
8a569d71SDarrick J. Wong
db07349dSChristoph Hellwig	ip->i_diflags |= di_flags;
603f000bSDarrick J. Wong
603f000bSDarrick J. Wong	/*
603f000bSDarrick J. Wong	 * Inode verifiers on older kernels only check that the extent size
603f000bSDarrick J. Wong	 * hint is an integer multiple of the rt extent size on realtime files.
603f000bSDarrick J. Wong	 * They did not check the hint alignment on a directory with both
603f000bSDarrick J. Wong	 * rtinherit and extszinherit flags set.  If the misaligned hint is
603f000bSDarrick J. Wong	 * propagated from a directory into a new realtime file, new file
603f000bSDarrick J. Wong	 * allocations will fail due to math errors in the rt allocator and/or
603f000bSDarrick J. Wong	 * trip the verifiers.  Validate the hint settings in the new file so
603f000bSDarrick J. Wong	 * that we don't let broken hints propagate.
603f000bSDarrick J. Wong	 */
603f000bSDarrick J. Wong	failaddr = xfs_inode_validate_extsize(ip->i_mount, ip->i_extsize,
603f000bSDarrick J. Wong			VFS_I(ip)->i_mode, ip->i_diflags);
603f000bSDarrick J. Wong	if (failaddr) {
603f000bSDarrick J. Wong		ip->i_diflags &= ~(XFS_DIFLAG_EXTSIZE |
603f000bSDarrick J. Wong				   XFS_DIFLAG_EXTSZINHERIT);
603f000bSDarrick J. Wong		ip->i_extsize = 0;
603f000bSDarrick J. Wong	}
8a569d71SDarrick J. Wong}
8a569d71SDarrick J. Wong
8a569d71SDarrick J. Wong/* Propagate di_flags2 from a parent inode to a child inode. */
8a569d71SDarrick J. Wongstatic void
8a569d71SDarrick J. Wongxfs_inode_inherit_flags2(
8a569d71SDarrick J. Wong	struct xfs_inode	*ip,
8a569d71SDarrick J. Wong	const struct xfs_inode	*pip)
8a569d71SDarrick J. Wong{
603f000bSDarrick J. Wong	xfs_failaddr_t		failaddr;
603f000bSDarrick J. Wong
3e09ab8fSChristoph Hellwig	if (pip->i_diflags2 & XFS_DIFLAG2_COWEXTSIZE) {
3e09ab8fSChristoph Hellwig		ip->i_diflags2 |= XFS_DIFLAG2_COWEXTSIZE;
b33ce57dSChristoph Hellwig		ip->i_cowextsize = pip->i_cowextsize;
8a569d71SDarrick J. Wong	}
3e09ab8fSChristoph Hellwig	if (pip->i_diflags2 & XFS_DIFLAG2_DAX)
3e09ab8fSChristoph Hellwig		ip->i_diflags2 |= XFS_DIFLAG2_DAX;
603f000bSDarrick J. Wong
603f000bSDarrick J. Wong	/* Don't let invalid cowextsize hints propagate. */
603f000bSDarrick J. Wong	failaddr = xfs_inode_validate_cowextsize(ip->i_mount, ip->i_cowextsize,
603f000bSDarrick J. Wong			VFS_I(ip)->i_mode, ip->i_diflags, ip->i_diflags2);
603f000bSDarrick J. Wong	if (failaddr) {
603f000bSDarrick J. Wong		ip->i_diflags2 &= ~XFS_DIFLAG2_COWEXTSIZE;
603f000bSDarrick J. Wong		ip->i_cowextsize = 0;
603f000bSDarrick J. Wong	}
8a569d71SDarrick J. Wong}
8a569d71SDarrick J. Wong
c24b5dfaSDave Chinner/*
1abcf261SDave Chinner * Initialise a newly allocated inode and return the in-core inode to the
1abcf261SDave Chinner * caller locked exclusively.
1da177e4SLinus Torvalds */
b652afd9SDave Chinnerint
1abcf261SDave Chinnerxfs_init_new_inode(
f2d40141SChristian Brauner	struct mnt_idmap	*idmap,
1abcf261SDave Chinner	struct xfs_trans	*tp,
1abcf261SDave Chinner	struct xfs_inode	*pip,
1abcf261SDave Chinner	xfs_ino_t		ino,
576b1d67SAl Viro	umode_t			mode,
31b084aeSNathan Scott	xfs_nlink_t		nlink,
66f36464SChristoph Hellwig	dev_t			rdev,
6743099cSArkadiusz Mi?kiewicz	prid_t			prid,
e6a688c3SDave Chinner	bool			init_xattrs,
1abcf261SDave Chinner	struct xfs_inode	**ipp)
1da177e4SLinus Torvalds{
01ea173eSChristoph Hellwig	struct inode		*dir = pip ? VFS_I(pip) : NULL;
93848a99SChristoph Hellwig	struct xfs_mount	*mp = tp->t_mountp;
1abcf261SDave Chinner	struct xfs_inode	*ip;
1abcf261SDave Chinner	unsigned int		flags;
1da177e4SLinus Torvalds	int			error;
95582b00SDeepa Dinamani	struct timespec64	tv;
3987848cSDave Chinner	struct inode		*inode;
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
8b26984dSDave Chinner	 * Protect against obviously corrupt allocation btree records. Later
8b26984dSDave Chinner	 * xfs_iget checks will catch re-allocation of other active in-memory
8b26984dSDave Chinner	 * and on-disk inodes. If we don't catch reallocating the parent inode
8b26984dSDave Chinner	 * here we will deadlock in xfs_iget() so we have to do these checks
8b26984dSDave Chinner	 * first.
8b26984dSDave Chinner	 */
8b26984dSDave Chinner	if ((pip && ino == pip->i_ino) || !xfs_verify_dir_ino(mp, ino)) {
8b26984dSDave Chinner		xfs_alert(mp, "Allocated a known in-use inode 0x%llx!", ino);
8b26984dSDave Chinner		return -EFSCORRUPTED;
8b26984dSDave Chinner	}
8b26984dSDave Chinner
8b26984dSDave Chinner	/*
1abcf261SDave Chinner	 * Get the in-core inode with the lock held exclusively to prevent
1abcf261SDave Chinner	 * others from looking at until we're done.
1da177e4SLinus Torvalds	 */
1abcf261SDave Chinner	error = xfs_iget(mp, tp, ino, XFS_IGET_CREATE, XFS_ILOCK_EXCL, &ip);
bf904248SDavid Chinner	if (error)
1da177e4SLinus Torvalds		return error;
1abcf261SDave Chinner
1da177e4SLinus Torvalds	ASSERT(ip != NULL);
3987848cSDave Chinner	inode = VFS_I(ip);
54d7b5c1SDave Chinner	set_nlink(inode, nlink);
66f36464SChristoph Hellwig	inode->i_rdev = rdev;
ceaf603cSChristoph Hellwig	ip->i_projid = prid;
1da177e4SLinus Torvalds
0560f31aSDave Chinner	if (dir && !(dir->i_mode & S_ISGID) && xfs_has_grpid(mp)) {
c14329d3SChristian Brauner		inode_fsuid_set(inode, idmap);
01ea173eSChristoph Hellwig		inode->i_gid = dir->i_gid;
01ea173eSChristoph Hellwig		inode->i_mode = mode;
3d8f2821SChristoph Hellwig	} else {
f2d40141SChristian Brauner		inode_init_owner(idmap, inode, dir, mode);
1da177e4SLinus Torvalds	}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
1da177e4SLinus Torvalds	 * If the group ID of the new file does not match the effective group
1da177e4SLinus Torvalds	 * ID or one of the supplementary group IDs, the S_ISGID bit is cleared
1da177e4SLinus Torvalds	 * (and only if the irix_sgid_inherit compatibility variable is set).
1da177e4SLinus Torvalds	 */
42b7cc11SChristian Brauner	if (irix_sgid_inherit && (inode->i_mode & S_ISGID) &&
e67fe633SChristian Brauner	    !vfsgid_in_group_p(i_gid_into_vfsgid(idmap, inode)))
c19b3b05SDave Chinner		inode->i_mode &= ~S_ISGID;
1da177e4SLinus Torvalds
13d2c10bSChristoph Hellwig	ip->i_disk_size = 0;
daf83964SChristoph Hellwig	ip->i_df.if_nextents = 0;
6e73a545SChristoph Hellwig	ASSERT(ip->i_nblocks == 0);
dff35fd4SChristoph Hellwig
a0a415e3SJeff Layton	tv = inode_set_ctime_current(inode);
3987848cSDave Chinner	inode->i_mtime = tv;
3987848cSDave Chinner	inode->i_atime = tv;
dff35fd4SChristoph Hellwig
031474c2SChristoph Hellwig	ip->i_extsize = 0;
db07349dSChristoph Hellwig	ip->i_diflags = 0;
93848a99SChristoph Hellwig
38c26bfdSDave Chinner	if (xfs_has_v3inodes(mp)) {
f0e28280SJeff Layton		inode_set_iversion(inode, 1);
b33ce57dSChristoph Hellwig		ip->i_cowextsize = 0;
e98d5e88SChristoph Hellwig		ip->i_crtime = tv;
93848a99SChristoph Hellwig	}
93848a99SChristoph Hellwig
1da177e4SLinus Torvalds	flags = XFS_ILOG_CORE;
1da177e4SLinus Torvalds	switch (mode & S_IFMT) {
1da177e4SLinus Torvalds	case S_IFIFO:
1da177e4SLinus Torvalds	case S_IFCHR:
1da177e4SLinus Torvalds	case S_IFBLK:
1da177e4SLinus Torvalds	case S_IFSOCK:
f7e67b20SChristoph Hellwig		ip->i_df.if_format = XFS_DINODE_FMT_DEV;
1da177e4SLinus Torvalds		flags |= XFS_ILOG_DEV;
1da177e4SLinus Torvalds		break;
1da177e4SLinus Torvalds	case S_IFREG:
1da177e4SLinus Torvalds	case S_IFDIR:
db07349dSChristoph Hellwig		if (pip && (pip->i_diflags & XFS_DIFLAG_ANY))
8a569d71SDarrick J. Wong			xfs_inode_inherit_flags(ip, pip);
3e09ab8fSChristoph Hellwig		if (pip && (pip->i_diflags2 & XFS_DIFLAG2_ANY))
8a569d71SDarrick J. Wong			xfs_inode_inherit_flags2(ip, pip);
53004ee7SGustavo A. R. Silva		fallthrough;
1da177e4SLinus Torvalds	case S_IFLNK:
f7e67b20SChristoph Hellwig		ip->i_df.if_format = XFS_DINODE_FMT_EXTENTS;
fcacbc3fSChristoph Hellwig		ip->i_df.if_bytes = 0;
6bdcf26aSChristoph Hellwig		ip->i_df.if_u1.if_root = NULL;
1da177e4SLinus Torvalds		break;
1da177e4SLinus Torvalds	default:
1da177e4SLinus Torvalds		ASSERT(0);
1da177e4SLinus Torvalds	}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
e6a688c3SDave Chinner	 * If we need to create attributes immediately after allocating the
e6a688c3SDave Chinner	 * inode, initialise an empty attribute fork right now. We use the
e6a688c3SDave Chinner	 * default fork offset for attributes here as we don't know exactly what
e6a688c3SDave Chinner	 * size or how many attributes we might be adding. We can do this
e6a688c3SDave Chinner	 * safely here because we know the data fork is completely empty and
e6a688c3SDave Chinner	 * this saves us from needing to run a separate transaction to set the
e6a688c3SDave Chinner	 * fork offset in the immediate future.
e6a688c3SDave Chinner	 */
38c26bfdSDave Chinner	if (init_xattrs && xfs_has_attr(mp)) {
7821ea30SChristoph Hellwig		ip->i_forkoff = xfs_default_attroffset(ip) >> 3;
2ed5b09bSDarrick J. Wong		xfs_ifork_init_attr(ip, XFS_DINODE_FMT_EXTENTS, 0);
e6a688c3SDave Chinner	}
e6a688c3SDave Chinner
e6a688c3SDave Chinner	/*
1da177e4SLinus Torvalds	 * Log the new values stuffed into the inode.
1da177e4SLinus Torvalds	 */
ddc3415aSChristoph Hellwig	xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL);
1da177e4SLinus Torvalds	xfs_trans_log_inode(tp, ip, flags);
1da177e4SLinus Torvalds
58c90473SDave Chinner	/* now that we have an i_mode we can setup the inode structure */
41be8bedSChristoph Hellwig	xfs_setup_inode(ip);
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	*ipp = ip;
1da177e4SLinus Torvalds	return 0;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
e546cb79SDave Chinner/*
54d7b5c1SDave Chinner * Decrement the link count on an inode & log the change.  If this causes the
54d7b5c1SDave Chinner * link count to go to zero, move the inode to AGI unlinked list so that it can
54d7b5c1SDave Chinner * be freed when the last active reference goes away via xfs_inactive().
e546cb79SDave Chinner */
0d5a75e9SEric Sandeenstatic int			/* error */
e546cb79SDave Chinnerxfs_droplink(
e546cb79SDave Chinner	xfs_trans_t *tp,
e546cb79SDave Chinner	xfs_inode_t *ip)
e546cb79SDave Chinner{
47b07e51SCheng Lin	if (VFS_I(ip)->i_nlink == 0) {
47b07e51SCheng Lin		xfs_alert(ip->i_mount,
47b07e51SCheng Lin			  "%s: Attempt to drop inode (%llu) with nlink zero.",
47b07e51SCheng Lin			  __func__, ip->i_ino);
47b07e51SCheng Lin		return -EFSCORRUPTED;
47b07e51SCheng Lin	}
47b07e51SCheng Lin
e546cb79SDave Chinner	xfs_trans_ichgtime(tp, ip, XFS_ICHGTIME_CHG);
e546cb79SDave Chinner
e546cb79SDave Chinner	drop_nlink(VFS_I(ip));
e546cb79SDave Chinner	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
e546cb79SDave Chinner
54d7b5c1SDave Chinner	if (VFS_I(ip)->i_nlink)
54d7b5c1SDave Chinner		return 0;
54d7b5c1SDave Chinner
54d7b5c1SDave Chinner	return xfs_iunlink(tp, ip);
e546cb79SDave Chinner}
e546cb79SDave Chinner
e546cb79SDave Chinner/*
e546cb79SDave Chinner * Increment the link count on an inode & log the change.
e546cb79SDave Chinner */
91083269SEric Sandeenstatic void
e546cb79SDave Chinnerxfs_bumplink(
e546cb79SDave Chinner	xfs_trans_t *tp,
e546cb79SDave Chinner	xfs_inode_t *ip)
e546cb79SDave Chinner{
e546cb79SDave Chinner	xfs_trans_ichgtime(tp, ip, XFS_ICHGTIME_CHG);
e546cb79SDave Chinner
e546cb79SDave Chinner	inc_nlink(VFS_I(ip));
e546cb79SDave Chinner	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
e546cb79SDave Chinner}
e546cb79SDave Chinner
c24b5dfaSDave Chinnerint
c24b5dfaSDave Chinnerxfs_create(
f2d40141SChristian Brauner	struct mnt_idmap	*idmap,
c24b5dfaSDave Chinner	xfs_inode_t		*dp,
c24b5dfaSDave Chinner	struct xfs_name		*name,
c24b5dfaSDave Chinner	umode_t			mode,
66f36464SChristoph Hellwig	dev_t			rdev,
e6a688c3SDave Chinner	bool			init_xattrs,
c24b5dfaSDave Chinner	xfs_inode_t		**ipp)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	int			is_dir = S_ISDIR(mode);
c24b5dfaSDave Chinner	struct xfs_mount	*mp = dp->i_mount;
c24b5dfaSDave Chinner	struct xfs_inode	*ip = NULL;
c24b5dfaSDave Chinner	struct xfs_trans	*tp = NULL;
c24b5dfaSDave Chinner	int			error;
c24b5dfaSDave Chinner	bool                    unlock_dp_on_error = false;
c24b5dfaSDave Chinner	prid_t			prid;
c24b5dfaSDave Chinner	struct xfs_dquot	*udqp = NULL;
c24b5dfaSDave Chinner	struct xfs_dquot	*gdqp = NULL;
c24b5dfaSDave Chinner	struct xfs_dquot	*pdqp = NULL;
062647a8SBrian Foster	struct xfs_trans_res	*tres;
c24b5dfaSDave Chinner	uint			resblks;
b652afd9SDave Chinner	xfs_ino_t		ino;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	trace_xfs_create(dp, name);
c24b5dfaSDave Chinner
75c8c50fSDave Chinner	if (xfs_is_shutdown(mp))
2451337dSDave Chinner		return -EIO;
c24b5dfaSDave Chinner
163467d3SZhi Yong Wu	prid = xfs_get_initial_prid(dp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
ff627196SDarrick J. Wong	 * Make sure that we have allocated dquot(s) on disk.  The uid/gid
ff627196SDarrick J. Wong	 * computation code must match what the VFS uses to assign i_[ug]id.
ff627196SDarrick J. Wong	 * INHERIT adjusts the gid computation for setgid/grpid systems.
c24b5dfaSDave Chinner	 */
ff627196SDarrick J. Wong	error = xfs_qm_vop_dqalloc(dp, mapped_fsuid(idmap, i_user_ns(VFS_I(dp))),
ff627196SDarrick J. Wong			mapped_fsgid(idmap, i_user_ns(VFS_I(dp))), prid,
c24b5dfaSDave Chinner			XFS_QMOPT_QUOTALL | XFS_QMOPT_INHERIT,
c24b5dfaSDave Chinner			&udqp, &gdqp, &pdqp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		return error;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (is_dir) {
c24b5dfaSDave Chinner		resblks = XFS_MKDIR_SPACE_RES(mp, name->len);
062647a8SBrian Foster		tres = &M_RES(mp)->tr_mkdir;
c24b5dfaSDave Chinner	} else {
c24b5dfaSDave Chinner		resblks = XFS_CREATE_SPACE_RES(mp, name->len);
062647a8SBrian Foster		tres = &M_RES(mp)->tr_create;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Initially assume that the file does not exist and
c24b5dfaSDave Chinner	 * reserve the resources for that case.  If that is not
c24b5dfaSDave Chinner	 * the case we'll drop the one we have and get a more
c24b5dfaSDave Chinner	 * appropriate transaction later.
c24b5dfaSDave Chinner	 */
f2f7b9ffSDarrick J. Wong	error = xfs_trans_alloc_icreate(mp, tres, udqp, gdqp, pdqp, resblks,
f2f7b9ffSDarrick J. Wong			&tp);
2451337dSDave Chinner	if (error == -ENOSPC) {
c24b5dfaSDave Chinner		/* flush outstanding delalloc blocks and retry */
c24b5dfaSDave Chinner		xfs_flush_inodes(mp);
f2f7b9ffSDarrick J. Wong		error = xfs_trans_alloc_icreate(mp, tres, udqp, gdqp, pdqp,
f2f7b9ffSDarrick J. Wong				resblks, &tp);
c24b5dfaSDave Chinner	}
4906e215SChristoph Hellwig	if (error)
f2f7b9ffSDarrick J. Wong		goto out_release_dquots;
c24b5dfaSDave Chinner
65523218SChristoph Hellwig	xfs_ilock(dp, XFS_ILOCK_EXCL | XFS_ILOCK_PARENT);
c24b5dfaSDave Chinner	unlock_dp_on_error = true;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * A newly created regular or special file just has one directory
c24b5dfaSDave Chinner	 * entry pointing to them, but a directory also the "." entry
c24b5dfaSDave Chinner	 * pointing to itself.
c24b5dfaSDave Chinner	 */
b652afd9SDave Chinner	error = xfs_dialloc(&tp, dp->i_ino, mode, &ino);
b652afd9SDave Chinner	if (!error)
f2d40141SChristian Brauner		error = xfs_init_new_inode(idmap, tp, dp, ino, mode,
b652afd9SDave Chinner				is_dir ? 2 : 1, rdev, prid, init_xattrs, &ip);
d6077aa3SJan Kara	if (error)
c24b5dfaSDave Chinner		goto out_trans_cancel;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Now we join the directory inode to the transaction.  We do not do it
b652afd9SDave Chinner	 * earlier because xfs_dialloc might commit the previous transaction
c24b5dfaSDave Chinner	 * (and release all the locks).  An error from here on will result in
c24b5dfaSDave Chinner	 * the transaction cancel unlocking dp so don't do it explicitly in the
c24b5dfaSDave Chinner	 * error path.
c24b5dfaSDave Chinner	 */
65523218SChristoph Hellwig	xfs_trans_ijoin(tp, dp, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner	unlock_dp_on_error = false;
c24b5dfaSDave Chinner
381eee69SBrian Foster	error = xfs_dir_createname(tp, dp, name, ip->i_ino,
63337b63SKaixu Xia					resblks - XFS_IALLOC_SPACE_RES(mp));
c24b5dfaSDave Chinner	if (error) {
2451337dSDave Chinner		ASSERT(error != -ENOSPC);
4906e215SChristoph Hellwig		goto out_trans_cancel;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner	xfs_trans_ichgtime(tp, dp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
c24b5dfaSDave Chinner	xfs_trans_log_inode(tp, dp, XFS_ILOG_CORE);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (is_dir) {
c24b5dfaSDave Chinner		error = xfs_dir_init(tp, ip, dp);
c24b5dfaSDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
c24b5dfaSDave Chinner
91083269SEric Sandeen		xfs_bumplink(tp, dp);
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If this is a synchronous mount, make sure that the
c24b5dfaSDave Chinner	 * create transaction goes to disk before returning to
c24b5dfaSDave Chinner	 * the user.
c24b5dfaSDave Chinner	 */
0560f31aSDave Chinner	if (xfs_has_wsync(mp) || xfs_has_dirsync(mp))
c24b5dfaSDave Chinner		xfs_trans_set_sync(tp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Attach the dquot(s) to the inodes and modify them incore.
c24b5dfaSDave Chinner	 * These ids of the inode couldn't have changed since the new
c24b5dfaSDave Chinner	 * inode has been locked ever since it was created.
c24b5dfaSDave Chinner	 */
c24b5dfaSDave Chinner	xfs_qm_vop_create_dqattach(tp, ip, udqp, gdqp, pdqp);
c24b5dfaSDave Chinner
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto out_release_inode;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	xfs_qm_dqrele(udqp);
c24b5dfaSDave Chinner	xfs_qm_dqrele(gdqp);
c24b5dfaSDave Chinner	xfs_qm_dqrele(pdqp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	*ipp = ip;
c24b5dfaSDave Chinner	return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner out_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
c24b5dfaSDave Chinner out_release_inode:
c24b5dfaSDave Chinner	/*
58c90473SDave Chinner	 * Wait until after the current transaction is aborted to finish the
58c90473SDave Chinner	 * setup of the inode and release the inode.  This prevents recursive
58c90473SDave Chinner	 * transactions and deadlocks from xfs_inactive.
c24b5dfaSDave Chinner	 */
58c90473SDave Chinner	if (ip) {
58c90473SDave Chinner		xfs_finish_inode_setup(ip);
44a8736bSDarrick J. Wong		xfs_irele(ip);
58c90473SDave Chinner	}
f2f7b9ffSDarrick J. Wong out_release_dquots:
c24b5dfaSDave Chinner	xfs_qm_dqrele(udqp);
c24b5dfaSDave Chinner	xfs_qm_dqrele(gdqp);
c24b5dfaSDave Chinner	xfs_qm_dqrele(pdqp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	if (unlock_dp_on_error)
65523218SChristoph Hellwig		xfs_iunlock(dp, XFS_ILOCK_EXCL);
c24b5dfaSDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinnerint
99b6436bSZhi Yong Wuxfs_create_tmpfile(
f2d40141SChristian Brauner	struct mnt_idmap	*idmap,
99b6436bSZhi Yong Wu	struct xfs_inode	*dp,
330033d6SBrian Foster	umode_t			mode,
330033d6SBrian Foster	struct xfs_inode	**ipp)
99b6436bSZhi Yong Wu{
99b6436bSZhi Yong Wu	struct xfs_mount	*mp = dp->i_mount;
99b6436bSZhi Yong Wu	struct xfs_inode	*ip = NULL;
99b6436bSZhi Yong Wu	struct xfs_trans	*tp = NULL;
99b6436bSZhi Yong Wu	int			error;
99b6436bSZhi Yong Wu	prid_t                  prid;
99b6436bSZhi Yong Wu	struct xfs_dquot	*udqp = NULL;
99b6436bSZhi Yong Wu	struct xfs_dquot	*gdqp = NULL;
99b6436bSZhi Yong Wu	struct xfs_dquot	*pdqp = NULL;
99b6436bSZhi Yong Wu	struct xfs_trans_res	*tres;
99b6436bSZhi Yong Wu	uint			resblks;
b652afd9SDave Chinner	xfs_ino_t		ino;
99b6436bSZhi Yong Wu
75c8c50fSDave Chinner	if (xfs_is_shutdown(mp))
2451337dSDave Chinner		return -EIO;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	prid = xfs_get_initial_prid(dp);
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	/*
ff627196SDarrick J. Wong	 * Make sure that we have allocated dquot(s) on disk.  The uid/gid
ff627196SDarrick J. Wong	 * computation code must match what the VFS uses to assign i_[ug]id.
ff627196SDarrick J. Wong	 * INHERIT adjusts the gid computation for setgid/grpid systems.
99b6436bSZhi Yong Wu	 */
ff627196SDarrick J. Wong	error = xfs_qm_vop_dqalloc(dp, mapped_fsuid(idmap, i_user_ns(VFS_I(dp))),
ff627196SDarrick J. Wong			mapped_fsgid(idmap, i_user_ns(VFS_I(dp))), prid,
99b6436bSZhi Yong Wu			XFS_QMOPT_QUOTALL | XFS_QMOPT_INHERIT,
99b6436bSZhi Yong Wu			&udqp, &gdqp, &pdqp);
99b6436bSZhi Yong Wu	if (error)
99b6436bSZhi Yong Wu		return error;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	resblks = XFS_IALLOC_SPACE_RES(mp);
99b6436bSZhi Yong Wu	tres = &M_RES(mp)->tr_create_tmpfile;
253f4911SChristoph Hellwig
f2f7b9ffSDarrick J. Wong	error = xfs_trans_alloc_icreate(mp, tres, udqp, gdqp, pdqp, resblks,
f2f7b9ffSDarrick J. Wong			&tp);
4906e215SChristoph Hellwig	if (error)
f2f7b9ffSDarrick J. Wong		goto out_release_dquots;
99b6436bSZhi Yong Wu
b652afd9SDave Chinner	error = xfs_dialloc(&tp, dp->i_ino, mode, &ino);
b652afd9SDave Chinner	if (!error)
f2d40141SChristian Brauner		error = xfs_init_new_inode(idmap, tp, dp, ino, mode,
b652afd9SDave Chinner				0, 0, prid, false, &ip);
d6077aa3SJan Kara	if (error)
99b6436bSZhi Yong Wu		goto out_trans_cancel;
99b6436bSZhi Yong Wu
0560f31aSDave Chinner	if (xfs_has_wsync(mp))
99b6436bSZhi Yong Wu		xfs_trans_set_sync(tp);
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	/*
99b6436bSZhi Yong Wu	 * Attach the dquot(s) to the inodes and modify them incore.
99b6436bSZhi Yong Wu	 * These ids of the inode couldn't have changed since the new
99b6436bSZhi Yong Wu	 * inode has been locked ever since it was created.
99b6436bSZhi Yong Wu	 */
99b6436bSZhi Yong Wu	xfs_qm_vop_create_dqattach(tp, ip, udqp, gdqp, pdqp);
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	error = xfs_iunlink(tp, ip);
99b6436bSZhi Yong Wu	if (error)
4906e215SChristoph Hellwig		goto out_trans_cancel;
99b6436bSZhi Yong Wu
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
99b6436bSZhi Yong Wu	if (error)
99b6436bSZhi Yong Wu		goto out_release_inode;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	xfs_qm_dqrele(udqp);
99b6436bSZhi Yong Wu	xfs_qm_dqrele(gdqp);
99b6436bSZhi Yong Wu	xfs_qm_dqrele(pdqp);
99b6436bSZhi Yong Wu
330033d6SBrian Foster	*ipp = ip;
99b6436bSZhi Yong Wu	return 0;
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu out_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
99b6436bSZhi Yong Wu out_release_inode:
99b6436bSZhi Yong Wu	/*
58c90473SDave Chinner	 * Wait until after the current transaction is aborted to finish the
58c90473SDave Chinner	 * setup of the inode and release the inode.  This prevents recursive
58c90473SDave Chinner	 * transactions and deadlocks from xfs_inactive.
99b6436bSZhi Yong Wu	 */
58c90473SDave Chinner	if (ip) {
58c90473SDave Chinner		xfs_finish_inode_setup(ip);
44a8736bSDarrick J. Wong		xfs_irele(ip);
58c90473SDave Chinner	}
f2f7b9ffSDarrick J. Wong out_release_dquots:
99b6436bSZhi Yong Wu	xfs_qm_dqrele(udqp);
99b6436bSZhi Yong Wu	xfs_qm_dqrele(gdqp);
99b6436bSZhi Yong Wu	xfs_qm_dqrele(pdqp);
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wu	return error;
99b6436bSZhi Yong Wu}
99b6436bSZhi Yong Wu
99b6436bSZhi Yong Wuint
c24b5dfaSDave Chinnerxfs_link(
c24b5dfaSDave Chinner	xfs_inode_t		*tdp,
c24b5dfaSDave Chinner	xfs_inode_t		*sip,
c24b5dfaSDave Chinner	struct xfs_name		*target_name)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	xfs_mount_t		*mp = tdp->i_mount;
c24b5dfaSDave Chinner	xfs_trans_t		*tp;
871b9316SDarrick J. Wong	int			error, nospace_error = 0;
c24b5dfaSDave Chinner	int			resblks;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	trace_xfs_link(tdp, target_name);
c24b5dfaSDave Chinner
c19b3b05SDave Chinner	ASSERT(!S_ISDIR(VFS_I(sip)->i_mode));
c24b5dfaSDave Chinner
75c8c50fSDave Chinner	if (xfs_is_shutdown(mp))
2451337dSDave Chinner		return -EIO;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(sip);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(tdp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	resblks = XFS_LINK_SPACE_RES(mp, target_name->len);
871b9316SDarrick J. Wong	error = xfs_trans_alloc_dir(tdp, &M_RES(mp)->tr_link, sip, &resblks,
871b9316SDarrick J. Wong			&tp, &nospace_error);
4906e215SChristoph Hellwig	if (error)
253f4911SChristoph Hellwig		goto std_return;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If we are using project inheritance, we only allow hard link
c24b5dfaSDave Chinner	 * creation in our tree when the project IDs are the same; else
c24b5dfaSDave Chinner	 * the tree quota mechanism could be circumvented.
c24b5dfaSDave Chinner	 */
db07349dSChristoph Hellwig	if (unlikely((tdp->i_diflags & XFS_DIFLAG_PROJINHERIT) &&
ceaf603cSChristoph Hellwig		     tdp->i_projid != sip->i_projid)) {
9f205010SAndrey Albershteyn		/*
9f205010SAndrey Albershteyn		 * Project quota setup skips special files which can
9f205010SAndrey Albershteyn		 * leave inodes in a PROJINHERIT directory without a
9f205010SAndrey Albershteyn		 * project ID set. We need to allow links to be made
9f205010SAndrey Albershteyn		 * to these "project-less" inodes because userspace
9f205010SAndrey Albershteyn		 * expects them to succeed after project ID setup,
9f205010SAndrey Albershteyn		 * but everything else should be rejected.
9f205010SAndrey Albershteyn		 */
9f205010SAndrey Albershteyn		if (!special_file(VFS_I(sip)->i_mode) ||
9f205010SAndrey Albershteyn		    sip->i_projid != 0) {
2451337dSDave Chinner			error = -EXDEV;
c24b5dfaSDave Chinner			goto error_return;
c24b5dfaSDave Chinner		}
9f205010SAndrey Albershteyn	}
c24b5dfaSDave Chinner
94f3cad5SEric Sandeen	if (!resblks) {
94f3cad5SEric Sandeen		error = xfs_dir_canenter(tp, tdp, target_name);
c24b5dfaSDave Chinner		if (error)
c24b5dfaSDave Chinner			goto error_return;
94f3cad5SEric Sandeen	}
c24b5dfaSDave Chinner
54d7b5c1SDave Chinner	/*
54d7b5c1SDave Chinner	 * Handle initial link state of O_TMPFILE inode
54d7b5c1SDave Chinner	 */
54d7b5c1SDave Chinner	if (VFS_I(sip)->i_nlink == 0) {
f40aadb2SDave Chinner		struct xfs_perag	*pag;
f40aadb2SDave Chinner
f40aadb2SDave Chinner		pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, sip->i_ino));
f40aadb2SDave Chinner		error = xfs_iunlink_remove(tp, pag, sip);
f40aadb2SDave Chinner		xfs_perag_put(pag);
ab297431SZhi Yong Wu		if (error)
4906e215SChristoph Hellwig			goto error_return;
ab297431SZhi Yong Wu	}
ab297431SZhi Yong Wu
c24b5dfaSDave Chinner	error = xfs_dir_createname(tp, tdp, target_name, sip->i_ino,
381eee69SBrian Foster				   resblks);
c24b5dfaSDave Chinner	if (error)
4906e215SChristoph Hellwig		goto error_return;
c24b5dfaSDave Chinner	xfs_trans_ichgtime(tp, tdp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
c24b5dfaSDave Chinner	xfs_trans_log_inode(tp, tdp, XFS_ILOG_CORE);
c24b5dfaSDave Chinner
91083269SEric Sandeen	xfs_bumplink(tp, sip);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If this is a synchronous mount, make sure that the
c24b5dfaSDave Chinner	 * link transaction goes to disk before returning to
c24b5dfaSDave Chinner	 * the user.
c24b5dfaSDave Chinner	 */
0560f31aSDave Chinner	if (xfs_has_wsync(mp) || xfs_has_dirsync(mp))
c24b5dfaSDave Chinner		xfs_trans_set_sync(tp);
c24b5dfaSDave Chinner
70393313SChristoph Hellwig	return xfs_trans_commit(tp);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner error_return:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
c24b5dfaSDave Chinner std_return:
871b9316SDarrick J. Wong	if (error == -ENOSPC && nospace_error)
871b9316SDarrick J. Wong		error = nospace_error;
c24b5dfaSDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
363e59baSDarrick J. Wong/* Clear the reflink flag and the cowblocks tag if possible. */
363e59baSDarrick J. Wongstatic void
363e59baSDarrick J. Wongxfs_itruncate_clear_reflink_flags(
363e59baSDarrick J. Wong	struct xfs_inode	*ip)
363e59baSDarrick J. Wong{
363e59baSDarrick J. Wong	struct xfs_ifork	*dfork;
363e59baSDarrick J. Wong	struct xfs_ifork	*cfork;
363e59baSDarrick J. Wong
363e59baSDarrick J. Wong	if (!xfs_is_reflink_inode(ip))
363e59baSDarrick J. Wong		return;
732436efSDarrick J. Wong	dfork = xfs_ifork_ptr(ip, XFS_DATA_FORK);
732436efSDarrick J. Wong	cfork = xfs_ifork_ptr(ip, XFS_COW_FORK);
363e59baSDarrick J. Wong	if (dfork->if_bytes == 0 && cfork->if_bytes == 0)
3e09ab8fSChristoph Hellwig		ip->i_diflags2 &= ~XFS_DIFLAG2_REFLINK;
363e59baSDarrick J. Wong	if (cfork->if_bytes == 0)
363e59baSDarrick J. Wong		xfs_inode_clear_cowblocks_tag(ip);
363e59baSDarrick J. Wong}
363e59baSDarrick J. Wong
1da177e4SLinus Torvalds/*
8f04c47aSChristoph Hellwig * Free up the underlying blocks past new_size.  The new size must be smaller
8f04c47aSChristoph Hellwig * than the current size.  This routine can be used both for the attribute and
8f04c47aSChristoph Hellwig * data fork, and does not modify the inode size, which is left to the caller.
1da177e4SLinus Torvalds *
f6485057SDavid Chinner * The transaction passed to this routine must have made a permanent log
f6485057SDavid Chinner * reservation of at least XFS_ITRUNCATE_LOG_RES.  This routine may commit the
f6485057SDavid Chinner * given transaction and start new ones, so make sure everything involved in
f6485057SDavid Chinner * the transaction is tidy before calling here.  Some transaction will be
f6485057SDavid Chinner * returned to the caller to be committed.  The incoming transaction must
f6485057SDavid Chinner * already include the inode, and both inode locks must be held exclusively.
f6485057SDavid Chinner * The inode must also be "held" within the transaction.  On return the inode
f6485057SDavid Chinner * will be "held" within the returned transaction.  This routine does NOT
f6485057SDavid Chinner * require any disk space to be reserved for it within the transaction.
1da177e4SLinus Torvalds *
f6485057SDavid Chinner * If we get an error, we must return with the inode locked and linked into the
f6485057SDavid Chinner * current transaction. This keeps things simple for the higher level code,
f6485057SDavid Chinner * because it always knows that the inode is locked and held in the transaction
f6485057SDavid Chinner * that returns to it whether errors occur or not.  We don't mark the inode
f6485057SDavid Chinner * dirty on error so that transactions can be easily aborted if possible.
1da177e4SLinus Torvalds */
1da177e4SLinus Torvaldsint
4e529339SBrian Fosterxfs_itruncate_extents_flags(
8f04c47aSChristoph Hellwig	struct xfs_trans	**tpp,
8f04c47aSChristoph Hellwig	struct xfs_inode	*ip,
8f04c47aSChristoph Hellwig	int			whichfork,
13b86fc3SBrian Foster	xfs_fsize_t		new_size,
4e529339SBrian Foster	int			flags)
1da177e4SLinus Torvalds{
8f04c47aSChristoph Hellwig	struct xfs_mount	*mp = ip->i_mount;
8f04c47aSChristoph Hellwig	struct xfs_trans	*tp = *tpp;
1da177e4SLinus Torvalds	xfs_fileoff_t		first_unmap_block;
8f04c47aSChristoph Hellwig	xfs_filblks_t		unmap_len;
8f04c47aSChristoph Hellwig	int			error = 0;
1da177e4SLinus Torvalds
0b56185bSChristoph Hellwig	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL));
0b56185bSChristoph Hellwig	ASSERT(!atomic_read(&VFS_I(ip)->i_count) ||
0b56185bSChristoph Hellwig	       xfs_isilocked(ip, XFS_IOLOCK_EXCL));
ce7ae151SChristoph Hellwig	ASSERT(new_size <= XFS_ISIZE(ip));
8f04c47aSChristoph Hellwig	ASSERT(tp->t_flags & XFS_TRANS_PERM_LOG_RES);
1da177e4SLinus Torvalds	ASSERT(ip->i_itemp != NULL);
898621d5SChristoph Hellwig	ASSERT(ip->i_itemp->ili_lock_flags == 0);
1da177e4SLinus Torvalds	ASSERT(!XFS_NOT_DQATTACHED(mp, ip));
1da177e4SLinus Torvalds
673e8e59SChristoph Hellwig	trace_xfs_itruncate_extents_start(ip, new_size);
673e8e59SChristoph Hellwig
4e529339SBrian Foster	flags |= xfs_bmapi_aflag(whichfork);
13b86fc3SBrian Foster
1da177e4SLinus Torvalds	/*
1da177e4SLinus Torvalds	 * Since it is possible for space to become allocated beyond
1da177e4SLinus Torvalds	 * the end of the file (in a crash where the space is allocated
1da177e4SLinus Torvalds	 * but the inode size is not yet updated), simply remove any
1da177e4SLinus Torvalds	 * blocks which show up between the new EOF and the maximum
4bbb04abSDarrick J. Wong	 * possible file size.
4bbb04abSDarrick J. Wong	 *
4bbb04abSDarrick J. Wong	 * We have to free all the blocks to the bmbt maximum offset, even if
4bbb04abSDarrick J. Wong	 * the page cache can't scale that far.
1da177e4SLinus Torvalds	 */
8f04c47aSChristoph Hellwig	first_unmap_block = XFS_B_TO_FSB(mp, (xfs_ufsize_t)new_size);
33005fd0SDarrick J. Wong	if (!xfs_verify_fileoff(mp, first_unmap_block)) {
4bbb04abSDarrick J. Wong		WARN_ON_ONCE(first_unmap_block > XFS_MAX_FILEOFF);
8f04c47aSChristoph Hellwig		return 0;
4bbb04abSDarrick J. Wong	}
8f04c47aSChristoph Hellwig
4bbb04abSDarrick J. Wong	unmap_len = XFS_MAX_FILEOFF - first_unmap_block + 1;
4bbb04abSDarrick J. Wong	while (unmap_len > 0) {
692b6cddSDave Chinner		ASSERT(tp->t_highest_agno == NULLAGNUMBER);
4bbb04abSDarrick J. Wong		error = __xfs_bunmapi(tp, ip, first_unmap_block, &unmap_len,
4bbb04abSDarrick J. Wong				flags, XFS_ITRUNC_MAX_EXTENTS);
8f04c47aSChristoph Hellwig		if (error)
d5a2e289SBrian Foster			goto out;
1da177e4SLinus Torvalds
6dd379c7SBrian Foster		/* free the just unmapped extents */
9e28a242SBrian Foster		error = xfs_defer_finish(&tp);
8f04c47aSChristoph Hellwig		if (error)
9b1f4e98SBrian Foster			goto out;
1da177e4SLinus Torvalds	}
8f04c47aSChristoph Hellwig
4919d42aSDarrick J. Wong	if (whichfork == XFS_DATA_FORK) {
aa8968f2SDarrick J. Wong		/* Remove all pending CoW reservations. */
4919d42aSDarrick J. Wong		error = xfs_reflink_cancel_cow_blocks(ip, &tp,
4bbb04abSDarrick J. Wong				first_unmap_block, XFS_MAX_FILEOFF, true);
aa8968f2SDarrick J. Wong		if (error)
aa8968f2SDarrick J. Wong			goto out;
aa8968f2SDarrick J. Wong
363e59baSDarrick J. Wong		xfs_itruncate_clear_reflink_flags(ip);
4919d42aSDarrick J. Wong	}
aa8968f2SDarrick J. Wong
673e8e59SChristoph Hellwig	/*
673e8e59SChristoph Hellwig	 * Always re-log the inode so that our permanent transaction can keep
673e8e59SChristoph Hellwig	 * on rolling it forward in the log.
673e8e59SChristoph Hellwig	 */
673e8e59SChristoph Hellwig	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
673e8e59SChristoph Hellwig
673e8e59SChristoph Hellwig	trace_xfs_itruncate_extents_end(ip, new_size);
673e8e59SChristoph Hellwig
8f04c47aSChristoph Hellwigout:
8f04c47aSChristoph Hellwig	*tpp = tp;
8f04c47aSChristoph Hellwig	return error;
8f04c47aSChristoph Hellwig}
8f04c47aSChristoph Hellwig
c24b5dfaSDave Chinnerint
c24b5dfaSDave Chinnerxfs_release(
c24b5dfaSDave Chinner	xfs_inode_t	*ip)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	xfs_mount_t	*mp = ip->i_mount;
7d88329eSDarrick J. Wong	int		error = 0;
c24b5dfaSDave Chinner
c19b3b05SDave Chinner	if (!S_ISREG(VFS_I(ip)->i_mode) || (VFS_I(ip)->i_mode == 0))
c24b5dfaSDave Chinner		return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/* If this is a read-only mount, don't do this (would generate I/O) */
2e973b2cSDave Chinner	if (xfs_is_readonly(mp))
c24b5dfaSDave Chinner		return 0;
c24b5dfaSDave Chinner
75c8c50fSDave Chinner	if (!xfs_is_shutdown(mp)) {
c24b5dfaSDave Chinner		int truncated;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		/*
c24b5dfaSDave Chinner		 * If we previously truncated this file and removed old data
c24b5dfaSDave Chinner		 * in the process, we want to initiate "early" writeout on
c24b5dfaSDave Chinner		 * the last close.  This is an attempt to combat the notorious
c24b5dfaSDave Chinner		 * NULL files problem which is particularly noticeable from a
c24b5dfaSDave Chinner		 * truncate down, buffered (re-)write (delalloc), followed by
c24b5dfaSDave Chinner		 * a crash.  What we are effectively doing here is
c24b5dfaSDave Chinner		 * significantly reducing the time window where we'd otherwise
c24b5dfaSDave Chinner		 * be exposed to that problem.
c24b5dfaSDave Chinner		 */
c24b5dfaSDave Chinner		truncated = xfs_iflags_test_and_clear(ip, XFS_ITRUNCATED);
c24b5dfaSDave Chinner		if (truncated) {
c24b5dfaSDave Chinner			xfs_iflags_clear(ip, XFS_IDIRTY_RELEASE);
eac152b4SDave Chinner			if (ip->i_delayed_blks > 0) {
2451337dSDave Chinner				error = filemap_flush(VFS_I(ip)->i_mapping);
c24b5dfaSDave Chinner				if (error)
c24b5dfaSDave Chinner					return error;
c24b5dfaSDave Chinner			}
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
54d7b5c1SDave Chinner	if (VFS_I(ip)->i_nlink == 0)
c24b5dfaSDave Chinner		return 0;
c24b5dfaSDave Chinner
7d88329eSDarrick J. Wong	/*
7d88329eSDarrick J. Wong	 * If we can't get the iolock just skip truncating the blocks past EOF
7d88329eSDarrick J. Wong	 * because we could deadlock with the mmap_lock otherwise. We'll get
7d88329eSDarrick J. Wong	 * another chance to drop them once the last reference to the inode is
7d88329eSDarrick J. Wong	 * dropped, so we'll never leak blocks permanently.
7d88329eSDarrick J. Wong	 */
7d88329eSDarrick J. Wong	if (!xfs_ilock_nowait(ip, XFS_IOLOCK_EXCL))
7d88329eSDarrick J. Wong		return 0;
c24b5dfaSDave Chinner
2bc2d49cSChristoph Hellwig	if (xfs_can_free_eofblocks(ip)) {
c24b5dfaSDave Chinner		/*
a36b9261SBrian Foster		 * Check if the inode is being opened, written and closed
a36b9261SBrian Foster		 * frequently and we have delayed allocation blocks outstanding
a36b9261SBrian Foster		 * (e.g. streaming writes from the NFS server), truncating the
a36b9261SBrian Foster		 * blocks past EOF will cause fragmentation to occur.
a36b9261SBrian Foster		 *
a36b9261SBrian Foster		 * In this case don't do the truncation, but we have to be
a36b9261SBrian Foster		 * careful how we detect this case. Blocks beyond EOF show up as
a36b9261SBrian Foster		 * i_delayed_blks even when the inode is clean, so we need to
a36b9261SBrian Foster		 * truncate them away first before checking for a dirty release.
a36b9261SBrian Foster		 * Hence on the first dirty close we will still remove the
a36b9261SBrian Foster		 * speculative allocation, but after that we will leave it in
a36b9261SBrian Foster		 * place.
a36b9261SBrian Foster		 */
a36b9261SBrian Foster		if (xfs_iflags_test(ip, XFS_IDIRTY_RELEASE))
7d88329eSDarrick J. Wong			goto out_unlock;
7d88329eSDarrick J. Wong
a36b9261SBrian Foster		error = xfs_free_eofblocks(ip);
a36b9261SBrian Foster		if (error)
7d88329eSDarrick J. Wong			goto out_unlock;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner		/* delalloc blocks after truncation means it really is dirty */
c24b5dfaSDave Chinner		if (ip->i_delayed_blks)
c24b5dfaSDave Chinner			xfs_iflags_set(ip, XFS_IDIRTY_RELEASE);
c24b5dfaSDave Chinner	}
7d88329eSDarrick J. Wong
7d88329eSDarrick J. Wongout_unlock:
7d88329eSDarrick J. Wong	xfs_iunlock(ip, XFS_IOLOCK_EXCL);
7d88329eSDarrick J. Wong	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner/*
f7be2d7fSBrian Foster * xfs_inactive_truncate
f7be2d7fSBrian Foster *
f7be2d7fSBrian Foster * Called to perform a truncate when an inode becomes unlinked.
f7be2d7fSBrian Foster */
f7be2d7fSBrian FosterSTATIC int
f7be2d7fSBrian Fosterxfs_inactive_truncate(
f7be2d7fSBrian Foster	struct xfs_inode *ip)
f7be2d7fSBrian Foster{
f7be2d7fSBrian Foster	struct xfs_mount	*mp = ip->i_mount;
f7be2d7fSBrian Foster	struct xfs_trans	*tp;
f7be2d7fSBrian Foster	int			error;
f7be2d7fSBrian Foster
253f4911SChristoph Hellwig	error = xfs_trans_alloc(mp, &M_RES(mp)->tr_itruncate, 0, 0, 0, &tp);
f7be2d7fSBrian Foster	if (error) {
75c8c50fSDave Chinner		ASSERT(xfs_is_shutdown(mp));
f7be2d7fSBrian Foster		return error;
f7be2d7fSBrian Foster	}
f7be2d7fSBrian Foster	xfs_ilock(ip, XFS_ILOCK_EXCL);
f7be2d7fSBrian Foster	xfs_trans_ijoin(tp, ip, 0);
f7be2d7fSBrian Foster
f7be2d7fSBrian Foster	/*
f7be2d7fSBrian Foster	 * Log the inode size first to prevent stale data exposure in the event
f7be2d7fSBrian Foster	 * of a system crash before the truncate completes. See the related
69bca807SJan Kara	 * comment in xfs_vn_setattr_size() for details.
f7be2d7fSBrian Foster	 */
13d2c10bSChristoph Hellwig	ip->i_disk_size = 0;
f7be2d7fSBrian Foster	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
f7be2d7fSBrian Foster
f7be2d7fSBrian Foster	error = xfs_itruncate_extents(&tp, ip, XFS_DATA_FORK, 0);
f7be2d7fSBrian Foster	if (error)
f7be2d7fSBrian Foster		goto error_trans_cancel;
f7be2d7fSBrian Foster
daf83964SChristoph Hellwig	ASSERT(ip->i_df.if_nextents == 0);
f7be2d7fSBrian Foster
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
f7be2d7fSBrian Foster	if (error)
f7be2d7fSBrian Foster		goto error_unlock;
f7be2d7fSBrian Foster
f7be2d7fSBrian Foster	xfs_iunlock(ip, XFS_ILOCK_EXCL);
f7be2d7fSBrian Foster	return 0;
f7be2d7fSBrian Foster
f7be2d7fSBrian Fostererror_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
f7be2d7fSBrian Fostererror_unlock:
f7be2d7fSBrian Foster	xfs_iunlock(ip, XFS_ILOCK_EXCL);
f7be2d7fSBrian Foster	return error;
f7be2d7fSBrian Foster}
f7be2d7fSBrian Foster
f7be2d7fSBrian Foster/*
88877d2bSBrian Foster * xfs_inactive_ifree()
88877d2bSBrian Foster *
88877d2bSBrian Foster * Perform the inode free when an inode is unlinked.
88877d2bSBrian Foster */
88877d2bSBrian FosterSTATIC int
88877d2bSBrian Fosterxfs_inactive_ifree(
88877d2bSBrian Foster	struct xfs_inode *ip)
88877d2bSBrian Foster{
88877d2bSBrian Foster	struct xfs_mount	*mp = ip->i_mount;
88877d2bSBrian Foster	struct xfs_trans	*tp;
88877d2bSBrian Foster	int			error;
88877d2bSBrian Foster
9d43b180SBrian Foster	/*
76d771b4SChristoph Hellwig	 * We try to use a per-AG reservation for any block needed by the finobt
76d771b4SChristoph Hellwig	 * tree, but as the finobt feature predates the per-AG reservation
76d771b4SChristoph Hellwig	 * support a degraded file system might not have enough space for the
76d771b4SChristoph Hellwig	 * reservation at mount time.  In that case try to dip into the reserved
76d771b4SChristoph Hellwig	 * pool and pray.
9d43b180SBrian Foster	 *
9d43b180SBrian Foster	 * Send a warning if the reservation does happen to fail, as the inode
9d43b180SBrian Foster	 * now remains allocated and sits on the unlinked list until the fs is
9d43b180SBrian Foster	 * repaired.
9d43b180SBrian Foster	 */
e1f6ca11SDarrick J. Wong	if (unlikely(mp->m_finobt_nores)) {
253f4911SChristoph Hellwig		error = xfs_trans_alloc(mp, &M_RES(mp)->tr_ifree,
76d771b4SChristoph Hellwig				XFS_IFREE_SPACE_RES(mp), 0, XFS_TRANS_RESERVE,
76d771b4SChristoph Hellwig				&tp);
76d771b4SChristoph Hellwig	} else {
76d771b4SChristoph Hellwig		error = xfs_trans_alloc(mp, &M_RES(mp)->tr_ifree, 0, 0, 0, &tp);
76d771b4SChristoph Hellwig	}
88877d2bSBrian Foster	if (error) {
2451337dSDave Chinner		if (error == -ENOSPC) {
9d43b180SBrian Foster			xfs_warn_ratelimited(mp,
9d43b180SBrian Foster			"Failed to remove inode(s) from unlinked list. "
9d43b180SBrian Foster			"Please free space, unmount and run xfs_repair.");
9d43b180SBrian Foster		} else {
75c8c50fSDave Chinner			ASSERT(xfs_is_shutdown(mp));
9d43b180SBrian Foster		}
88877d2bSBrian Foster		return error;
88877d2bSBrian Foster	}
88877d2bSBrian Foster
96355d5aSDave Chinner	/*
96355d5aSDave Chinner	 * We do not hold the inode locked across the entire rolling transaction
96355d5aSDave Chinner	 * here. We only need to hold it for the first transaction that
96355d5aSDave Chinner	 * xfs_ifree() builds, which may mark the inode XFS_ISTALE if the
96355d5aSDave Chinner	 * underlying cluster buffer is freed. Relogging an XFS_ISTALE inode
96355d5aSDave Chinner	 * here breaks the relationship between cluster buffer invalidation and
96355d5aSDave Chinner	 * stale inode invalidation on cluster buffer item journal commit
96355d5aSDave Chinner	 * completion, and can result in leaving dirty stale inodes hanging
96355d5aSDave Chinner	 * around in memory.
96355d5aSDave Chinner	 *
96355d5aSDave Chinner	 * We have no need for serialising this inode operation against other
96355d5aSDave Chinner	 * operations - we freed the inode and hence reallocation is required
96355d5aSDave Chinner	 * and that will serialise on reallocating the space the deferops need
96355d5aSDave Chinner	 * to free. Hence we can unlock the inode on the first commit of
96355d5aSDave Chinner	 * the transaction rather than roll it right through the deferops. This
96355d5aSDave Chinner	 * avoids relogging the XFS_ISTALE inode.
96355d5aSDave Chinner	 *
96355d5aSDave Chinner	 * We check that xfs_ifree() hasn't grown an internal transaction roll
96355d5aSDave Chinner	 * by asserting that the inode is still locked when it returns.
96355d5aSDave Chinner	 */
88877d2bSBrian Foster	xfs_ilock(ip, XFS_ILOCK_EXCL);
96355d5aSDave Chinner	xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL);
88877d2bSBrian Foster
0e0417f3SBrian Foster	error = xfs_ifree(tp, ip);
96355d5aSDave Chinner	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL));
88877d2bSBrian Foster	if (error) {
88877d2bSBrian Foster		/*
88877d2bSBrian Foster		 * If we fail to free the inode, shut down.  The cancel
88877d2bSBrian Foster		 * might do that, we need to make sure.  Otherwise the
88877d2bSBrian Foster		 * inode might be lost for a long time or forever.
88877d2bSBrian Foster		 */
75c8c50fSDave Chinner		if (!xfs_is_shutdown(mp)) {
88877d2bSBrian Foster			xfs_notice(mp, "%s: xfs_ifree returned error %d",
88877d2bSBrian Foster				__func__, error);
88877d2bSBrian Foster			xfs_force_shutdown(mp, SHUTDOWN_META_IO_ERROR);
88877d2bSBrian Foster		}
4906e215SChristoph Hellwig		xfs_trans_cancel(tp);
88877d2bSBrian Foster		return error;
88877d2bSBrian Foster	}
88877d2bSBrian Foster
88877d2bSBrian Foster	/*
88877d2bSBrian Foster	 * Credit the quota account(s). The inode is gone.
88877d2bSBrian Foster	 */
88877d2bSBrian Foster	xfs_trans_mod_dquot_byino(tp, ip, XFS_TRANS_DQ_ICOUNT, -1);
88877d2bSBrian Foster
d4d12c02SDave Chinner	return xfs_trans_commit(tp);
88877d2bSBrian Foster}
88877d2bSBrian Foster
88877d2bSBrian Foster/*
62af7d54SDarrick J. Wong * Returns true if we need to update the on-disk metadata before we can free
62af7d54SDarrick J. Wong * the memory used by this inode.  Updates include freeing post-eof
62af7d54SDarrick J. Wong * preallocations; freeing COW staging extents; and marking the inode free in
62af7d54SDarrick J. Wong * the inobt if it is on the unlinked list.
62af7d54SDarrick J. Wong */
62af7d54SDarrick J. Wongbool
62af7d54SDarrick J. Wongxfs_inode_needs_inactive(
62af7d54SDarrick J. Wong	struct xfs_inode	*ip)
62af7d54SDarrick J. Wong{
62af7d54SDarrick J. Wong	struct xfs_mount	*mp = ip->i_mount;
732436efSDarrick J. Wong	struct xfs_ifork	*cow_ifp = xfs_ifork_ptr(ip, XFS_COW_FORK);
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/*
62af7d54SDarrick J. Wong	 * If the inode is already free, then there can be nothing
62af7d54SDarrick J. Wong	 * to clean up here.
62af7d54SDarrick J. Wong	 */
62af7d54SDarrick J. Wong	if (VFS_I(ip)->i_mode == 0)
62af7d54SDarrick J. Wong		return false;
62af7d54SDarrick J. Wong
76e58901SDarrick J. Wong	/*
76e58901SDarrick J. Wong	 * If this is a read-only mount, don't do this (would generate I/O)
76e58901SDarrick J. Wong	 * unless we're in log recovery and cleaning the iunlinked list.
76e58901SDarrick J. Wong	 */
76e58901SDarrick J. Wong	if (xfs_is_readonly(mp) && !xlog_recovery_needed(mp->m_log))
62af7d54SDarrick J. Wong		return false;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* If the log isn't running, push inodes straight to reclaim. */
75c8c50fSDave Chinner	if (xfs_is_shutdown(mp) || xfs_has_norecovery(mp))
62af7d54SDarrick J. Wong		return false;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* Metadata inodes require explicit resource cleanup. */
62af7d54SDarrick J. Wong	if (xfs_is_metadata_inode(ip))
62af7d54SDarrick J. Wong		return false;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* Want to clean out the cow blocks if there are any. */
62af7d54SDarrick J. Wong	if (cow_ifp && cow_ifp->if_bytes > 0)
62af7d54SDarrick J. Wong		return true;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/* Unlinked files must be freed. */
62af7d54SDarrick J. Wong	if (VFS_I(ip)->i_nlink == 0)
62af7d54SDarrick J. Wong		return true;
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong	/*
62af7d54SDarrick J. Wong	 * This file isn't being freed, so check if there are post-eof blocks
2bc2d49cSChristoph Hellwig	 * to free.
62af7d54SDarrick J. Wong	 *
62af7d54SDarrick J. Wong	 * Note: don't bother with iolock here since lockdep complains about
62af7d54SDarrick J. Wong	 * acquiring it in reclaim context. We have the only reference to the
62af7d54SDarrick J. Wong	 * inode at this point anyways.
62af7d54SDarrick J. Wong	 */
2bc2d49cSChristoph Hellwig	return xfs_can_free_eofblocks(ip);
62af7d54SDarrick J. Wong}
62af7d54SDarrick J. Wong
62af7d54SDarrick J. Wong/*
c24b5dfaSDave Chinner * xfs_inactive
c24b5dfaSDave Chinner *
c24b5dfaSDave Chinner * This is called when the vnode reference count for the vnode
c24b5dfaSDave Chinner * goes to zero.  If the file has been unlinked, then it must
c24b5dfaSDave Chinner * now be truncated.  Also, we clear all of the read-ahead state
c24b5dfaSDave Chinner * kept for the inode here since the file is now closed.
c24b5dfaSDave Chinner */
d4d12c02SDave Chinnerint
c24b5dfaSDave Chinnerxfs_inactive(
c24b5dfaSDave Chinner	xfs_inode_t	*ip)
c24b5dfaSDave Chinner{
3d3c8b52SJie Liu	struct xfs_mount	*mp;
d4d12c02SDave Chinner	int			error = 0;
c24b5dfaSDave Chinner	int			truncate = 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If the inode is already free, then there can be nothing
c24b5dfaSDave Chinner	 * to clean up here.
c24b5dfaSDave Chinner	 */
c19b3b05SDave Chinner	if (VFS_I(ip)->i_mode == 0) {
c24b5dfaSDave Chinner		ASSERT(ip->i_df.if_broot_bytes == 0);
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	mp = ip->i_mount;
17c12bcdSDarrick J. Wong	ASSERT(!xfs_iflags_test(ip, XFS_IRECOVERY));
c24b5dfaSDave Chinner
76e58901SDarrick J. Wong	/*
76e58901SDarrick J. Wong	 * If this is a read-only mount, don't do this (would generate I/O)
76e58901SDarrick J. Wong	 * unless we're in log recovery and cleaning the iunlinked list.
76e58901SDarrick J. Wong	 */
76e58901SDarrick J. Wong	if (xfs_is_readonly(mp) && !xlog_recovery_needed(mp->m_log))
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner
383e32b0SDarrick J. Wong	/* Metadata inodes require explicit resource cleanup. */
383e32b0SDarrick J. Wong	if (xfs_is_metadata_inode(ip))
3ea06d73SDarrick J. Wong		goto out;
383e32b0SDarrick J. Wong
6231848cSDarrick J. Wong	/* Try to clean out the cow blocks if there are any. */
970e92caSWentao Liang	if (xfs_inode_has_cow_data(ip)) {
970e92caSWentao Liang		error = xfs_reflink_cancel_cow_range(ip, 0, NULLFILEOFF, true);
970e92caSWentao Liang		if (error)
970e92caSWentao Liang			goto out;
970e92caSWentao Liang	}
6231848cSDarrick J. Wong
54d7b5c1SDave Chinner	if (VFS_I(ip)->i_nlink != 0) {
c24b5dfaSDave Chinner		/*
3b4683c2SBrian Foster		 * Note: don't bother with iolock here since lockdep complains
3b4683c2SBrian Foster		 * about acquiring it in reclaim context. We have the only
3b4683c2SBrian Foster		 * reference to the inode at this point anyways.
c24b5dfaSDave Chinner		 */
2bc2d49cSChristoph Hellwig		if (xfs_can_free_eofblocks(ip))
d4d12c02SDave Chinner			error = xfs_free_eofblocks(ip);
74564fb4SBrian Foster
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c19b3b05SDave Chinner	if (S_ISREG(VFS_I(ip)->i_mode) &&
13d2c10bSChristoph Hellwig	    (ip->i_disk_size != 0 || XFS_ISIZE(ip) != 0 ||
*b887d2feSOjaswin Mujoo	     xfs_inode_has_filedata(ip)))
c24b5dfaSDave Chinner		truncate = 1;
c24b5dfaSDave Chinner
49813a21SDarrick J. Wong	if (xfs_iflags_test(ip, XFS_IQUOTAUNCHECKED)) {
537c013bSDarrick J. Wong		/*
537c013bSDarrick J. Wong		 * If this inode is being inactivated during a quotacheck and
537c013bSDarrick J. Wong		 * has not yet been scanned by quotacheck, we /must/ remove
537c013bSDarrick J. Wong		 * the dquots from the inode before inactivation changes the
537c013bSDarrick J. Wong		 * block and inode counts.  Most probably this is a result of
537c013bSDarrick J. Wong		 * reloading the incore iunlinked list to purge unrecovered
537c013bSDarrick J. Wong		 * unlinked inodes.
537c013bSDarrick J. Wong		 */
49813a21SDarrick J. Wong		xfs_qm_dqdetach(ip);
49813a21SDarrick J. Wong	} else {
c14cfccaSDarrick J. Wong		error = xfs_qm_dqattach(ip);
c24b5dfaSDave Chinner		if (error)
3ea06d73SDarrick J. Wong			goto out;
49813a21SDarrick J. Wong	}
c24b5dfaSDave Chinner
c19b3b05SDave Chinner	if (S_ISLNK(VFS_I(ip)->i_mode))
36b21ddeSBrian Foster		error = xfs_inactive_symlink(ip);
f7be2d7fSBrian Foster	else if (truncate)
f7be2d7fSBrian Foster		error = xfs_inactive_truncate(ip);
36b21ddeSBrian Foster	if (error)
3ea06d73SDarrick J. Wong		goto out;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If there are attributes associated with the file then blow them away
c24b5dfaSDave Chinner	 * now.  The code calls a routine that recursively deconstructs the
6dfe5a04SDave Chinner	 * attribute fork. If also blows away the in-core attribute fork.
c24b5dfaSDave Chinner	 */
932b42c6SDarrick J. Wong	if (xfs_inode_has_attr_fork(ip)) {
c24b5dfaSDave Chinner		error = xfs_attr_inactive(ip);
c24b5dfaSDave Chinner		if (error)
3ea06d73SDarrick J. Wong			goto out;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
7821ea30SChristoph Hellwig	ASSERT(ip->i_forkoff == 0);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * Free the inode.
c24b5dfaSDave Chinner	 */
d4d12c02SDave Chinner	error = xfs_inactive_ifree(ip);
c24b5dfaSDave Chinner
3ea06d73SDarrick J. Wongout:
c24b5dfaSDave Chinner	/*
3ea06d73SDarrick J. Wong	 * We're done making metadata updates for this inode, so we can release
3ea06d73SDarrick J. Wong	 * the attached dquots.
c24b5dfaSDave Chinner	 */
c24b5dfaSDave Chinner	xfs_qm_dqdetach(ip);
d4d12c02SDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
1da177e4SLinus Torvalds/*
9b247179SDarrick J. Wong * In-Core Unlinked List Lookups
9b247179SDarrick J. Wong * =============================
9b247179SDarrick J. Wong *
9b247179SDarrick J. Wong * Every inode is supposed to be reachable from some other piece of metadata
9b247179SDarrick J. Wong * with the exception of the root directory.  Inodes with a connection to a
9b247179SDarrick J. Wong * file descriptor but not linked from anywhere in the on-disk directory tree
9b247179SDarrick J. Wong * are collectively known as unlinked inodes, though the filesystem itself
9b247179SDarrick J. Wong * maintains links to these inodes so that on-disk metadata are consistent.
9b247179SDarrick J. Wong *
9b247179SDarrick J. Wong * XFS implements a per-AG on-disk hash table of unlinked inodes.  The AGI
9b247179SDarrick J. Wong * header contains a number of buckets that point to an inode, and each inode
9b247179SDarrick J. Wong * record has a pointer to the next inode in the hash chain.  This
9b247179SDarrick J. Wong * singly-linked list causes scaling problems in the iunlink remove function
9b247179SDarrick J. Wong * because we must walk that list to find the inode that points to the inode
9b247179SDarrick J. Wong * being removed from the unlinked hash bucket list.
9b247179SDarrick J. Wong *
2fd26cc0SDave Chinner * Hence we keep an in-memory double linked list to link each inode on an
2fd26cc0SDave Chinner * unlinked list. Because there are 64 unlinked lists per AGI, keeping pointer
2fd26cc0SDave Chinner * based lists would require having 64 list heads in the perag, one for each
2fd26cc0SDave Chinner * list. This is expensive in terms of memory (think millions of AGs) and cache
2fd26cc0SDave Chinner * misses on lookups. Instead, use the fact that inodes on the unlinked list
2fd26cc0SDave Chinner * must be referenced at the VFS level to keep them on the list and hence we
2fd26cc0SDave Chinner * have an existence guarantee for inodes on the unlinked list.
9b247179SDarrick J. Wong *
2fd26cc0SDave Chinner * Given we have an existence guarantee, we can use lockless inode cache lookups
2fd26cc0SDave Chinner * to resolve aginos to xfs inodes. This means we only need 8 bytes per inode
2fd26cc0SDave Chinner * for the double linked unlinked list, and we don't need any extra locking to
2fd26cc0SDave Chinner * keep the list safe as all manipulations are done under the AGI buffer lock.
2fd26cc0SDave Chinner * Keeping the list up to date does not require memory allocation, just finding
2fd26cc0SDave Chinner * the XFS inode and updating the next/prev unlinked list aginos.
9b247179SDarrick J. Wong */
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/*
a83d5a8bSDave Chinner * Find an inode on the unlinked list. This does not take references to the
a83d5a8bSDave Chinner * inode as we have existence guarantees by holding the AGI buffer lock and that
a83d5a8bSDave Chinner * only unlinked, referenced inodes can be on the unlinked inode list.  If we
a83d5a8bSDave Chinner * don't find the inode in cache, then let the caller handle the situation.
9b247179SDarrick J. Wong */
a83d5a8bSDave Chinnerstatic struct xfs_inode *
a83d5a8bSDave Chinnerxfs_iunlink_lookup(
9b247179SDarrick J. Wong	struct xfs_perag	*pag,
9b247179SDarrick J. Wong	xfs_agino_t		agino)
9b247179SDarrick J. Wong{
a83d5a8bSDave Chinner	struct xfs_inode	*ip;
9b247179SDarrick J. Wong
a83d5a8bSDave Chinner	rcu_read_lock();
a83d5a8bSDave Chinner	ip = radix_tree_lookup(&pag->pag_ici_root, agino);
68b957f6SDarrick J. Wong	if (!ip) {
68b957f6SDarrick J. Wong		/* Caller can handle inode not being in memory. */
68b957f6SDarrick J. Wong		rcu_read_unlock();
68b957f6SDarrick J. Wong		return NULL;
68b957f6SDarrick J. Wong	}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong	/*
68b957f6SDarrick J. Wong	 * Inode in RCU freeing limbo should not happen.  Warn about this and
68b957f6SDarrick J. Wong	 * let the caller handle the failure.
9b247179SDarrick J. Wong	 */
68b957f6SDarrick J. Wong	if (WARN_ON_ONCE(!ip->i_ino)) {
a83d5a8bSDave Chinner		rcu_read_unlock();
a83d5a8bSDave Chinner		return NULL;
a83d5a8bSDave Chinner	}
a83d5a8bSDave Chinner	ASSERT(!xfs_iflags_test(ip, XFS_IRECLAIMABLE | XFS_IRECLAIM));
a83d5a8bSDave Chinner	rcu_read_unlock();
a83d5a8bSDave Chinner	return ip;
a83d5a8bSDave Chinner}
a83d5a8bSDave Chinner
68b957f6SDarrick J. Wong/*
68b957f6SDarrick J. Wong * Update the prev pointer of the next agino.  Returns -ENOLINK if the inode
68b957f6SDarrick J. Wong * is not in cache.
68b957f6SDarrick J. Wong */
9b247179SDarrick J. Wongstatic int
2fd26cc0SDave Chinnerxfs_iunlink_update_backref(
9b247179SDarrick J. Wong	struct xfs_perag	*pag,
9b247179SDarrick J. Wong	xfs_agino_t		prev_agino,
2fd26cc0SDave Chinner	xfs_agino_t		next_agino)
9b247179SDarrick J. Wong{
2fd26cc0SDave Chinner	struct xfs_inode	*ip;
9b247179SDarrick J. Wong
2fd26cc0SDave Chinner	/* No update necessary if we are at the end of the list. */
2fd26cc0SDave Chinner	if (next_agino == NULLAGINO)
9b247179SDarrick J. Wong		return 0;
9b247179SDarrick J. Wong
2fd26cc0SDave Chinner	ip = xfs_iunlink_lookup(pag, next_agino);
2fd26cc0SDave Chinner	if (!ip)
68b957f6SDarrick J. Wong		return -ENOLINK;
68b957f6SDarrick J. Wong
2fd26cc0SDave Chinner	ip->i_prev_unlinked = prev_agino;
9b247179SDarrick J. Wong	return 0;
9b247179SDarrick J. Wong}
9b247179SDarrick J. Wong
9b247179SDarrick J. Wong/*
9a4a5118SDarrick J. Wong * Point the AGI unlinked bucket at an inode and log the results.  The caller
9a4a5118SDarrick J. Wong * is responsible for validating the old value.
9a4a5118SDarrick J. Wong */
9a4a5118SDarrick J. WongSTATIC int
9a4a5118SDarrick J. Wongxfs_iunlink_update_bucket(
9a4a5118SDarrick J. Wong	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
9a4a5118SDarrick J. Wong	struct xfs_buf		*agibp,
9a4a5118SDarrick J. Wong	unsigned int		bucket_index,
9a4a5118SDarrick J. Wong	xfs_agino_t		new_agino)
9a4a5118SDarrick J. Wong{
370c782bSChristoph Hellwig	struct xfs_agi		*agi = agibp->b_addr;
9a4a5118SDarrick J. Wong	xfs_agino_t		old_value;
9a4a5118SDarrick J. Wong	int			offset;
9a4a5118SDarrick J. Wong
2d6ca832SDave Chinner	ASSERT(xfs_verify_agino_or_null(pag, new_agino));
9a4a5118SDarrick J. Wong
9a4a5118SDarrick J. Wong	old_value = be32_to_cpu(agi->agi_unlinked[bucket_index]);
f40aadb2SDave Chinner	trace_xfs_iunlink_update_bucket(tp->t_mountp, pag->pag_agno, bucket_index,
9a4a5118SDarrick J. Wong			old_value, new_agino);
9a4a5118SDarrick J. Wong
9a4a5118SDarrick J. Wong	/*
9a4a5118SDarrick J. Wong	 * We should never find the head of the list already set to the value
9a4a5118SDarrick J. Wong	 * passed in because either we're adding or removing ourselves from the
9a4a5118SDarrick J. Wong	 * head of the list.
9a4a5118SDarrick J. Wong	 */
a5155b87SDarrick J. Wong	if (old_value == new_agino) {
8d57c216SDarrick J. Wong		xfs_buf_mark_corrupt(agibp);
9a4a5118SDarrick J. Wong		return -EFSCORRUPTED;
a5155b87SDarrick J. Wong	}
9a4a5118SDarrick J. Wong
9a4a5118SDarrick J. Wong	agi->agi_unlinked[bucket_index] = cpu_to_be32(new_agino);
9a4a5118SDarrick J. Wong	offset = offsetof(struct xfs_agi, agi_unlinked) +
9a4a5118SDarrick J. Wong			(sizeof(xfs_agino_t) * bucket_index);
9a4a5118SDarrick J. Wong	xfs_trans_log_buf(tp, agibp, offset, offset + sizeof(xfs_agino_t) - 1);
9a4a5118SDarrick J. Wong	return 0;
9a4a5118SDarrick J. Wong}
9a4a5118SDarrick J. Wong
68b957f6SDarrick J. Wong/*
68b957f6SDarrick J. Wong * Load the inode @next_agino into the cache and set its prev_unlinked pointer
68b957f6SDarrick J. Wong * to @prev_agino.  Caller must hold the AGI to synchronize with other changes
68b957f6SDarrick J. Wong * to the unlinked list.
68b957f6SDarrick J. Wong */
68b957f6SDarrick J. WongSTATIC int
68b957f6SDarrick J. Wongxfs_iunlink_reload_next(
68b957f6SDarrick J. Wong	struct xfs_trans	*tp,
68b957f6SDarrick J. Wong	struct xfs_buf		*agibp,
68b957f6SDarrick J. Wong	xfs_agino_t		prev_agino,
68b957f6SDarrick J. Wong	xfs_agino_t		next_agino)
68b957f6SDarrick J. Wong{
68b957f6SDarrick J. Wong	struct xfs_perag	*pag = agibp->b_pag;
68b957f6SDarrick J. Wong	struct xfs_mount	*mp = pag->pag_mount;
68b957f6SDarrick J. Wong	struct xfs_inode	*next_ip = NULL;
68b957f6SDarrick J. Wong	xfs_ino_t		ino;
68b957f6SDarrick J. Wong	int			error;
68b957f6SDarrick J. Wong
68b957f6SDarrick J. Wong	ASSERT(next_agino != NULLAGINO);
68b957f6SDarrick J. Wong
68b957f6SDarrick J. Wong#ifdef DEBUG
68b957f6SDarrick J. Wong	rcu_read_lock();
68b957f6SDarrick J. Wong	next_ip = radix_tree_lookup(&pag->pag_ici_root, next_agino);
68b957f6SDarrick J. Wong	ASSERT(next_ip == NULL);
68b957f6SDarrick J. Wong	rcu_read_unlock();
68b957f6SDarrick J. Wong#endif
68b957f6SDarrick J. Wong
68b957f6SDarrick J. Wong	xfs_info_ratelimited(mp,
68b957f6SDarrick J. Wong "Found unrecovered unlinked inode 0x%x in AG 0x%x.  Initiating recovery.",
68b957f6SDarrick J. Wong			next_agino, pag->pag_agno);
68b957f6SDarrick J. Wong
68b957f6SDarrick J. Wong	/*
68b957f6SDarrick J. Wong	 * Use an untrusted lookup just to be cautious in case the AGI has been
68b957f6SDarrick J. Wong	 * corrupted and now points at a free inode.  That shouldn't happen,
68b957f6SDarrick J. Wong	 * but we'd rather shut down now since we're already running in a weird
68b957f6SDarrick J. Wong	 * situation.
68b957f6SDarrick J. Wong	 */
68b957f6SDarrick J. Wong	ino = XFS_AGINO_TO_INO(mp, pag->pag_agno, next_agino);
68b957f6SDarrick J. Wong	error = xfs_iget(mp, tp, ino, XFS_IGET_UNTRUSTED, 0, &next_ip);
68b957f6SDarrick J. Wong	if (error)
68b957f6SDarrick J. Wong		return error;
68b957f6SDarrick J. Wong
68b957f6SDarrick J. Wong	/* If this is not an unlinked inode, something is very wrong. */
68b957f6SDarrick J. Wong	if (VFS_I(next_ip)->i_nlink != 0) {
68b957f6SDarrick J. Wong		error = -EFSCORRUPTED;
68b957f6SDarrick J. Wong		goto rele;
68b957f6SDarrick J. Wong	}
68b957f6SDarrick J. Wong
68b957f6SDarrick J. Wong	next_ip->i_prev_unlinked = prev_agino;
68b957f6SDarrick J. Wong	trace_xfs_iunlink_reload_next(next_ip);
68b957f6SDarrick J. Wongrele:
68b957f6SDarrick J. Wong	ASSERT(!(VFS_I(next_ip)->i_state & I_DONTCACHE));
49813a21SDarrick J. Wong	if (xfs_is_quotacheck_running(mp) && next_ip)
49813a21SDarrick J. Wong		xfs_iflags_set(next_ip, XFS_IQUOTAUNCHECKED);
68b957f6SDarrick J. Wong	xfs_irele(next_ip);
68b957f6SDarrick J. Wong	return error;
68b957f6SDarrick J. Wong}
68b957f6SDarrick J. Wong
a4454cd6SDave Chinnerstatic int
a4454cd6SDave Chinnerxfs_iunlink_insert_inode(
f2fc16a3SDarrick J. Wong	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
a4454cd6SDave Chinner	struct xfs_buf		*agibp,
a4454cd6SDave Chinner	struct xfs_inode	*ip)
f2fc16a3SDarrick J. Wong{
f2fc16a3SDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
a4454cd6SDave Chinner	struct xfs_agi		*agi = agibp->b_addr;
a4454cd6SDave Chinner	xfs_agino_t		next_agino;
a4454cd6SDave Chinner	xfs_agino_t		agino = XFS_INO_TO_AGINO(mp, ip->i_ino);
a4454cd6SDave Chinner	short			bucket_index = agino % XFS_AGI_UNLINKED_BUCKETS;
f2fc16a3SDarrick J. Wong	int			error;
f2fc16a3SDarrick J. Wong
a4454cd6SDave Chinner	/*
a4454cd6SDave Chinner	 * Get the index into the agi hash table for the list this inode will
a4454cd6SDave Chinner	 * go on.  Make sure the pointer isn't garbage and that this inode
a4454cd6SDave Chinner	 * isn't already on the list.
a4454cd6SDave Chinner	 */
a4454cd6SDave Chinner	next_agino = be32_to_cpu(agi->agi_unlinked[bucket_index]);
a4454cd6SDave Chinner	if (next_agino == agino ||
a4454cd6SDave Chinner	    !xfs_verify_agino_or_null(pag, next_agino)) {
a4454cd6SDave Chinner		xfs_buf_mark_corrupt(agibp);
a4454cd6SDave Chinner		return -EFSCORRUPTED;
f2fc16a3SDarrick J. Wong	}
f2fc16a3SDarrick J. Wong
f2fc16a3SDarrick J. Wong	/*
2fd26cc0SDave Chinner	 * Update the prev pointer in the next inode to point back to this
2fd26cc0SDave Chinner	 * inode.
f2fc16a3SDarrick J. Wong	 */
2fd26cc0SDave Chinner	error = xfs_iunlink_update_backref(pag, agino, next_agino);
68b957f6SDarrick J. Wong	if (error == -ENOLINK)
68b957f6SDarrick J. Wong		error = xfs_iunlink_reload_next(tp, agibp, agino, next_agino);
2fd26cc0SDave Chinner	if (error)
2fd26cc0SDave Chinner		return error;
2fd26cc0SDave Chinner
a5155b87SDarrick J. Wong	if (next_agino != NULLAGINO) {
a4454cd6SDave Chinner		/*
a4454cd6SDave Chinner		 * There is already another inode in the bucket, so point this
a4454cd6SDave Chinner		 * inode to the current head of the list.
a4454cd6SDave Chinner		 */
062efdb0SDave Chinner		error = xfs_iunlink_log_inode(tp, ip, pag, next_agino);
a4454cd6SDave Chinner		if (error)
a4454cd6SDave Chinner			return error;
4fcc94d6SDave Chinner		ip->i_next_unlinked = next_agino;
f2fc16a3SDarrick J. Wong	}
f2fc16a3SDarrick J. Wong
a4454cd6SDave Chinner	/* Point the head of the list to point to this inode. */
f12b9668SDarrick J. Wong	ip->i_prev_unlinked = NULLAGINO;
a4454cd6SDave Chinner	return xfs_iunlink_update_bucket(tp, pag, agibp, bucket_index, agino);
f2fc16a3SDarrick J. Wong}
f2fc16a3SDarrick J. Wong
9a4a5118SDarrick J. Wong/*
c4a6bf7fSDarrick J. Wong * This is called when the inode's link count has gone to 0 or we are creating
c4a6bf7fSDarrick J. Wong * a tmpfile via O_TMPFILE.  The inode @ip must have nlink == 0.
54d7b5c1SDave Chinner *
54d7b5c1SDave Chinner * We place the on-disk inode on a list in the AGI.  It will be pulled from this
54d7b5c1SDave Chinner * list when the inode is freed.
1da177e4SLinus Torvalds */
54d7b5c1SDave ChinnerSTATIC int
1da177e4SLinus Torvaldsxfs_iunlink(
54d7b5c1SDave Chinner	struct xfs_trans	*tp,
54d7b5c1SDave Chinner	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
5837f625SDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
f40aadb2SDave Chinner	struct xfs_perag	*pag;
5837f625SDarrick J. Wong	struct xfs_buf		*agibp;
1da177e4SLinus Torvalds	int			error;
1da177e4SLinus Torvalds
c4a6bf7fSDarrick J. Wong	ASSERT(VFS_I(ip)->i_nlink == 0);
c19b3b05SDave Chinner	ASSERT(VFS_I(ip)->i_mode != 0);
4664c66cSDarrick J. Wong	trace_xfs_iunlink(ip);
1da177e4SLinus Torvalds
f40aadb2SDave Chinner	pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, ip->i_ino));
f40aadb2SDave Chinner
5837f625SDarrick J. Wong	/* Get the agi buffer first.  It ensures lock ordering on the list. */
61021debSDave Chinner	error = xfs_read_agi(pag, tp, &agibp);
859d7182SVlad Apostolov	if (error)
f40aadb2SDave Chinner		goto out;
5e1be0fbSChristoph Hellwig
a4454cd6SDave Chinner	error = xfs_iunlink_insert_inode(tp, pag, agibp, ip);
f40aadb2SDave Chinnerout:
f40aadb2SDave Chinner	xfs_perag_put(pag);
f40aadb2SDave Chinner	return error;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
a4454cd6SDave Chinnerstatic int
a4454cd6SDave Chinnerxfs_iunlink_remove_inode(
23ffa52cSDarrick J. Wong	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
a4454cd6SDave Chinner	struct xfs_buf		*agibp,
5837f625SDarrick J. Wong	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
5837f625SDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
a4454cd6SDave Chinner	struct xfs_agi		*agi = agibp->b_addr;
5837f625SDarrick J. Wong	xfs_agino_t		agino = XFS_INO_TO_AGINO(mp, ip->i_ino);
b1d2a068SDarrick J. Wong	xfs_agino_t		head_agino;
5837f625SDarrick J. Wong	short			bucket_index = agino % XFS_AGI_UNLINKED_BUCKETS;
1da177e4SLinus Torvalds	int			error;
1da177e4SLinus Torvalds
4664c66cSDarrick J. Wong	trace_xfs_iunlink_remove(ip);
4664c66cSDarrick J. Wong
1da177e4SLinus Torvalds	/*
86bfd375SDarrick J. Wong	 * Get the index into the agi hash table for the list this inode will
86bfd375SDarrick J. Wong	 * go on.  Make sure the head pointer isn't garbage.
1da177e4SLinus Torvalds	 */
b1d2a068SDarrick J. Wong	head_agino = be32_to_cpu(agi->agi_unlinked[bucket_index]);
2d6ca832SDave Chinner	if (!xfs_verify_agino(pag, head_agino)) {
d2e73665SDarrick J. Wong		XFS_CORRUPTION_ERROR(__func__, XFS_ERRLEVEL_LOW, mp,
d2e73665SDarrick J. Wong				agi, sizeof(*agi));
d2e73665SDarrick J. Wong		return -EFSCORRUPTED;
d2e73665SDarrick J. Wong	}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
b1d2a068SDarrick J. Wong	 * Set our inode's next_unlinked pointer to NULL and then return
b1d2a068SDarrick J. Wong	 * the old pointer value so that we can update whatever was previous
b1d2a068SDarrick J. Wong	 * to us in the list to point to whatever was next in the list.
1da177e4SLinus Torvalds	 */
062efdb0SDave Chinner	error = xfs_iunlink_log_inode(tp, ip, pag, NULLAGINO);
f2fc16a3SDarrick J. Wong	if (error)
1da177e4SLinus Torvalds		return error;
9a4a5118SDarrick J. Wong
9b247179SDarrick J. Wong	/*
2fd26cc0SDave Chinner	 * Update the prev pointer in the next inode to point back to previous
2fd26cc0SDave Chinner	 * inode in the chain.
9b247179SDarrick J. Wong	 */
2fd26cc0SDave Chinner	error = xfs_iunlink_update_backref(pag, ip->i_prev_unlinked,
2fd26cc0SDave Chinner			ip->i_next_unlinked);
68b957f6SDarrick J. Wong	if (error == -ENOLINK)
68b957f6SDarrick J. Wong		error = xfs_iunlink_reload_next(tp, agibp, ip->i_prev_unlinked,
68b957f6SDarrick J. Wong				ip->i_next_unlinked);
9b247179SDarrick J. Wong	if (error)
92a00544SGao Xiang		return error;
9b247179SDarrick J. Wong
92a00544SGao Xiang	if (head_agino != agino) {
a83d5a8bSDave Chinner		struct xfs_inode	*prev_ip;
f2fc16a3SDarrick J. Wong
2fd26cc0SDave Chinner		prev_ip = xfs_iunlink_lookup(pag, ip->i_prev_unlinked);
2fd26cc0SDave Chinner		if (!prev_ip)
2fd26cc0SDave Chinner			return -EFSCORRUPTED;
475ee413SChristoph Hellwig
062efdb0SDave Chinner		error = xfs_iunlink_log_inode(tp, prev_ip, pag,
5301f870SDave Chinner				ip->i_next_unlinked);
a83d5a8bSDave Chinner		prev_ip->i_next_unlinked = ip->i_next_unlinked;
2fd26cc0SDave Chinner	} else {
2fd26cc0SDave Chinner		/* Point the head of the list to the next unlinked inode. */
2fd26cc0SDave Chinner		error = xfs_iunlink_update_bucket(tp, pag, agibp, bucket_index,
2fd26cc0SDave Chinner				ip->i_next_unlinked);
1da177e4SLinus Torvalds	}
9b247179SDarrick J. Wong
a83d5a8bSDave Chinner	ip->i_next_unlinked = NULLAGINO;
f12b9668SDarrick J. Wong	ip->i_prev_unlinked = 0;
2fd26cc0SDave Chinner	return error;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
5b3eed75SDave Chinner/*
a4454cd6SDave Chinner * Pull the on-disk inode from the AGI unlinked list.
a4454cd6SDave Chinner */
a4454cd6SDave ChinnerSTATIC int
a4454cd6SDave Chinnerxfs_iunlink_remove(
a4454cd6SDave Chinner	struct xfs_trans	*tp,
a4454cd6SDave Chinner	struct xfs_perag	*pag,
a4454cd6SDave Chinner	struct xfs_inode	*ip)
a4454cd6SDave Chinner{
a4454cd6SDave Chinner	struct xfs_buf		*agibp;
a4454cd6SDave Chinner	int			error;
a4454cd6SDave Chinner
a4454cd6SDave Chinner	trace_xfs_iunlink_remove(ip);
a4454cd6SDave Chinner
a4454cd6SDave Chinner	/* Get the agi buffer first.  It ensures lock ordering on the list. */
a4454cd6SDave Chinner	error = xfs_read_agi(pag, tp, &agibp);
1da177e4SLinus Torvalds	if (error)
1baaed8fSDave Chinner		return error;
1da177e4SLinus Torvalds
a4454cd6SDave Chinner	return xfs_iunlink_remove_inode(tp, pag, agibp, ip);
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
71e3e356SDave Chinner * Look up the inode number specified and if it is not already marked XFS_ISTALE
71e3e356SDave Chinner * mark it stale. We should only find clean inodes in this lookup that aren't
71e3e356SDave Chinner * already stale.
5806165aSDave Chinner */
71e3e356SDave Chinnerstatic void
71e3e356SDave Chinnerxfs_ifree_mark_inode_stale(
f40aadb2SDave Chinner	struct xfs_perag	*pag,
5806165aSDave Chinner	struct xfs_inode	*free_ip,
d9fdd0adSBrian Foster	xfs_ino_t		inum)
5806165aSDave Chinner{
f40aadb2SDave Chinner	struct xfs_mount	*mp = pag->pag_mount;
71e3e356SDave Chinner	struct xfs_inode_log_item *iip;
5806165aSDave Chinner	struct xfs_inode	*ip;
5806165aSDave Chinner
5806165aSDave Chinnerretry:
5806165aSDave Chinner	rcu_read_lock();
5806165aSDave Chinner	ip = radix_tree_lookup(&pag->pag_ici_root, XFS_INO_TO_AGINO(mp, inum));
5806165aSDave Chinner
5806165aSDave Chinner	/* Inode not in memory, nothing to do */
71e3e356SDave Chinner	if (!ip) {
71e3e356SDave Chinner		rcu_read_unlock();
71e3e356SDave Chinner		return;
71e3e356SDave Chinner	}
5806165aSDave Chinner
5806165aSDave Chinner	/*
5806165aSDave Chinner	 * because this is an RCU protected lookup, we could find a recently
5806165aSDave Chinner	 * freed or even reallocated inode during the lookup. We need to check
5806165aSDave Chinner	 * under the i_flags_lock for a valid inode here. Skip it if it is not
5806165aSDave Chinner	 * valid, the wrong inode or stale.
5806165aSDave Chinner	 */
5806165aSDave Chinner	spin_lock(&ip->i_flags_lock);
718ecc50SDave Chinner	if (ip->i_ino != inum || __xfs_iflags_test(ip, XFS_ISTALE))
718ecc50SDave Chinner		goto out_iflags_unlock;
5806165aSDave Chinner
5806165aSDave Chinner	/*
5806165aSDave Chinner	 * Don't try to lock/unlock the current inode, but we _cannot_ skip the
5806165aSDave Chinner	 * other inodes that we did not find in the list attached to the buffer
5806165aSDave Chinner	 * and are not already marked stale. If we can't lock it, back off and
5806165aSDave Chinner	 * retry.
5806165aSDave Chinner	 */
5806165aSDave Chinner	if (ip != free_ip) {
5806165aSDave Chinner		if (!xfs_ilock_nowait(ip, XFS_ILOCK_EXCL)) {
71e3e356SDave Chinner			spin_unlock(&ip->i_flags_lock);
5806165aSDave Chinner			rcu_read_unlock();
5806165aSDave Chinner			delay(1);
5806165aSDave Chinner			goto retry;
5806165aSDave Chinner		}
5806165aSDave Chinner	}
71e3e356SDave Chinner	ip->i_flags |= XFS_ISTALE;
5806165aSDave Chinner
71e3e356SDave Chinner	/*
718ecc50SDave Chinner	 * If the inode is flushing, it is already attached to the buffer.  All
71e3e356SDave Chinner	 * we needed to do here is mark the inode stale so buffer IO completion
71e3e356SDave Chinner	 * will remove it from the AIL.
71e3e356SDave Chinner	 */
71e3e356SDave Chinner	iip = ip->i_itemp;
718ecc50SDave Chinner	if (__xfs_iflags_test(ip, XFS_IFLUSHING)) {
71e3e356SDave Chinner		ASSERT(!list_empty(&iip->ili_item.li_bio_list));
71e3e356SDave Chinner		ASSERT(iip->ili_last_fields);
71e3e356SDave Chinner		goto out_iunlock;
71e3e356SDave Chinner	}
5806165aSDave Chinner
5806165aSDave Chinner	/*
48d55e2aSDave Chinner	 * Inodes not attached to the buffer can be released immediately.
48d55e2aSDave Chinner	 * Everything else has to go through xfs_iflush_abort() on journal
48d55e2aSDave Chinner	 * commit as the flock synchronises removal of the inode from the
48d55e2aSDave Chinner	 * cluster buffer against inode reclaim.
5806165aSDave Chinner	 */
718ecc50SDave Chinner	if (!iip || list_empty(&iip->ili_item.li_bio_list))
71e3e356SDave Chinner		goto out_iunlock;
718ecc50SDave Chinner
718ecc50SDave Chinner	__xfs_iflags_set(ip, XFS_IFLUSHING);
718ecc50SDave Chinner	spin_unlock(&ip->i_flags_lock);
718ecc50SDave Chinner	rcu_read_unlock();
5806165aSDave Chinner
71e3e356SDave Chinner	/* we have a dirty inode in memory that has not yet been flushed. */
71e3e356SDave Chinner	spin_lock(&iip->ili_lock);
71e3e356SDave Chinner	iip->ili_last_fields = iip->ili_fields;
71e3e356SDave Chinner	iip->ili_fields = 0;
71e3e356SDave Chinner	iip->ili_fsync_fields = 0;
71e3e356SDave Chinner	spin_unlock(&iip->ili_lock);
71e3e356SDave Chinner	ASSERT(iip->ili_last_fields);
71e3e356SDave Chinner
718ecc50SDave Chinner	if (ip != free_ip)
718ecc50SDave Chinner		xfs_iunlock(ip, XFS_ILOCK_EXCL);
718ecc50SDave Chinner	return;
718ecc50SDave Chinner
71e3e356SDave Chinnerout_iunlock:
71e3e356SDave Chinner	if (ip != free_ip)
71e3e356SDave Chinner		xfs_iunlock(ip, XFS_ILOCK_EXCL);
718ecc50SDave Chinnerout_iflags_unlock:
718ecc50SDave Chinner	spin_unlock(&ip->i_flags_lock);
718ecc50SDave Chinner	rcu_read_unlock();
5806165aSDave Chinner}
5806165aSDave Chinner
5806165aSDave Chinner/*
1da177e4SLinus Torvalds * A big issue when freeing the inode cluster is that we _cannot_ skip any
1da177e4SLinus Torvalds * inodes that are in memory - they all must be marked stale and attached to
1da177e4SLinus Torvalds * the cluster buffer.
1da177e4SLinus Torvalds */
f40aadb2SDave Chinnerstatic int
1da177e4SLinus Torvaldsxfs_ifree_cluster(
71e3e356SDave Chinner	struct xfs_trans	*tp,
f40aadb2SDave Chinner	struct xfs_perag	*pag,
f40aadb2SDave Chinner	struct xfs_inode	*free_ip,
1da177e4SLinus Torvalds	struct xfs_icluster	*xic)
1da177e4SLinus Torvalds{
71e3e356SDave Chinner	struct xfs_mount	*mp = free_ip->i_mount;
71e3e356SDave Chinner	struct xfs_ino_geometry	*igeo = M_IGEO(mp);
71e3e356SDave Chinner	struct xfs_buf		*bp;
71e3e356SDave Chinner	xfs_daddr_t		blkno;
71e3e356SDave Chinner	xfs_ino_t		inum = xic->first_ino;
1da177e4SLinus Torvalds	int			nbufs;
1da177e4SLinus Torvalds	int			i, j;
1da177e4SLinus Torvalds	int			ioffset;
ce92464cSDarrick J. Wong	int			error;
1da177e4SLinus Torvalds
ef325959SDarrick J. Wong	nbufs = igeo->ialloc_blks / igeo->blocks_per_cluster;
1da177e4SLinus Torvalds
ef325959SDarrick J. Wong	for (j = 0; j < nbufs; j++, inum += igeo->inodes_per_cluster) {
1da177e4SLinus Torvalds		/*
1da177e4SLinus Torvalds		 * The allocation bitmap tells us which inodes of the chunk were
1da177e4SLinus Torvalds		 * physically allocated. Skip the cluster if an inode falls into
1da177e4SLinus Torvalds		 * a sparse region.
1da177e4SLinus Torvalds		 */
1da177e4SLinus Torvalds		ioffset = inum - xic->first_ino;
1da177e4SLinus Torvalds		if ((xic->alloc & XFS_INOBT_MASK(ioffset)) == 0) {
ef325959SDarrick J. Wong			ASSERT(ioffset % igeo->inodes_per_cluster == 0);
1da177e4SLinus Torvalds			continue;
1da177e4SLinus Torvalds		}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds		blkno = XFS_AGB_TO_DADDR(mp, XFS_INO_TO_AGNO(mp, inum),
1da177e4SLinus Torvalds					 XFS_INO_TO_AGBNO(mp, inum));
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds		/*
1da177e4SLinus Torvalds		 * We obtain and lock the backing buffer first in the process
718ecc50SDave Chinner		 * here to ensure dirty inodes attached to the buffer remain in
718ecc50SDave Chinner		 * the flushing state while we mark them stale.
718ecc50SDave Chinner		 *
1da177e4SLinus Torvalds		 * If we scan the in-memory inodes first, then buffer IO can
1da177e4SLinus Torvalds		 * complete before we get a lock on it, and hence we may fail
1da177e4SLinus Torvalds		 * to mark all the active inodes on the buffer stale.
1da177e4SLinus Torvalds		 */
ce92464cSDarrick J. Wong		error = xfs_trans_get_buf(tp, mp->m_ddev_targp, blkno,
ef325959SDarrick J. Wong				mp->m_bsize * igeo->blocks_per_cluster,
ce92464cSDarrick J. Wong				XBF_UNMAPPED, &bp);
71e3e356SDave Chinner		if (error)
ce92464cSDarrick J. Wong			return error;
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds		/*
1da177e4SLinus Torvalds		 * This buffer may not have been correctly initialised as we
1da177e4SLinus Torvalds		 * didn't read it from disk. That's not important because we are
1da177e4SLinus Torvalds		 * only using to mark the buffer as stale in the log, and to
740a427eSDave Chinner		 * attach stale cached inodes on it.
740a427eSDave Chinner		 *
740a427eSDave Chinner		 * For the inode that triggered the cluster freeing, this
740a427eSDave Chinner		 * attachment may occur in xfs_inode_item_precommit() after we
740a427eSDave Chinner		 * have marked this buffer stale.  If this buffer was not in
740a427eSDave Chinner		 * memory before xfs_ifree_cluster() started, it will not be
740a427eSDave Chinner		 * marked XBF_DONE and this will cause problems later in
740a427eSDave Chinner		 * xfs_inode_item_precommit() when we trip over a (stale, !done)
740a427eSDave Chinner		 * buffer to attached to the transaction.
740a427eSDave Chinner		 *
740a427eSDave Chinner		 * Hence we have to mark the buffer as XFS_DONE here. This is
740a427eSDave Chinner		 * safe because we are also marking the buffer as XBF_STALE and
740a427eSDave Chinner		 * XFS_BLI_STALE. That means it will never be dispatched for
740a427eSDave Chinner		 * IO and it won't be unlocked until the cluster freeing has
740a427eSDave Chinner		 * been committed to the journal and the buffer unpinned. If it
740a427eSDave Chinner		 * is written, we want to know about it, and we want it to
740a427eSDave Chinner		 * fail. We can acheive this by adding a write verifier to the
740a427eSDave Chinner		 * buffer.
1da177e4SLinus Torvalds		 */
740a427eSDave Chinner		bp->b_flags |= XBF_DONE;
1da177e4SLinus Torvalds		bp->b_ops = &xfs_inode_buf_ops;
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds		/*
71e3e356SDave Chinner		 * Now we need to set all the cached clean inodes as XFS_ISTALE,
71e3e356SDave Chinner		 * too. This requires lookups, and will skip inodes that we've
71e3e356SDave Chinner		 * already marked XFS_ISTALE.
1da177e4SLinus Torvalds		 */
71e3e356SDave Chinner		for (i = 0; i < igeo->inodes_per_cluster; i++)
f40aadb2SDave Chinner			xfs_ifree_mark_inode_stale(pag, free_ip, inum + i);
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds		xfs_trans_stale_inode_buf(tp, bp);
1da177e4SLinus Torvalds		xfs_trans_binval(tp, bp);
1da177e4SLinus Torvalds	}
1da177e4SLinus Torvalds	return 0;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
9a5280b3SDave Chinner * This is called to return an inode to the inode free list.  The inode should
9a5280b3SDave Chinner * already be truncated to 0 length and have no pages associated with it.  This
9a5280b3SDave Chinner * routine also assumes that the inode is already a part of the transaction.
1da177e4SLinus Torvalds *
9a5280b3SDave Chinner * The on-disk copy of the inode will have been added to the list of unlinked
9a5280b3SDave Chinner * inodes in the AGI. We need to remove the inode from that list atomically with
9a5280b3SDave Chinner * respect to freeing it here.
1da177e4SLinus Torvalds */
1da177e4SLinus Torvaldsint
1da177e4SLinus Torvaldsxfs_ifree(
1da177e4SLinus Torvalds	struct xfs_trans	*tp,
1da177e4SLinus Torvalds	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
f40aadb2SDave Chinner	struct xfs_mount	*mp = ip->i_mount;
f40aadb2SDave Chinner	struct xfs_perag	*pag;
1da177e4SLinus Torvalds	struct xfs_icluster	xic = { 0 };
1319ebefSDave Chinner	struct xfs_inode_log_item *iip = ip->i_itemp;
f40aadb2SDave Chinner	int			error;
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL));
1da177e4SLinus Torvalds	ASSERT(VFS_I(ip)->i_nlink == 0);
daf83964SChristoph Hellwig	ASSERT(ip->i_df.if_nextents == 0);
13d2c10bSChristoph Hellwig	ASSERT(ip->i_disk_size == 0 || !S_ISREG(VFS_I(ip)->i_mode));
6e73a545SChristoph Hellwig	ASSERT(ip->i_nblocks == 0);
1da177e4SLinus Torvalds
f40aadb2SDave Chinner	pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, ip->i_ino));
f40aadb2SDave Chinner
1da177e4SLinus Torvalds	/*
9a5280b3SDave Chinner	 * Free the inode first so that we guarantee that the AGI lock is going
9a5280b3SDave Chinner	 * to be taken before we remove the inode from the unlinked list. This
9a5280b3SDave Chinner	 * makes the AGI lock -> unlinked list modification order the same as
9a5280b3SDave Chinner	 * used in O_TMPFILE creation.
1da177e4SLinus Torvalds	 */
f40aadb2SDave Chinner	error = xfs_difree(tp, pag, ip->i_ino, &xic);
1baaed8fSDave Chinner	if (error)
6f5097e3SBrian Foster		goto out;
9a5280b3SDave Chinner
9a5280b3SDave Chinner	error = xfs_iunlink_remove(tp, pag, ip);
9a5280b3SDave Chinner	if (error)
f40aadb2SDave Chinner		goto out;
1baaed8fSDave Chinner
b2c20045SChristoph Hellwig	/*
b2c20045SChristoph Hellwig	 * Free any local-format data sitting around before we reset the
b2c20045SChristoph Hellwig	 * data fork to extents format.  Note that the attr fork data has
b2c20045SChristoph Hellwig	 * already been freed by xfs_attr_inactive.
b2c20045SChristoph Hellwig	 */
f7e67b20SChristoph Hellwig	if (ip->i_df.if_format == XFS_DINODE_FMT_LOCAL) {
b2c20045SChristoph Hellwig		kmem_free(ip->i_df.if_u1.if_data);
b2c20045SChristoph Hellwig		ip->i_df.if_u1.if_data = NULL;
b2c20045SChristoph Hellwig		ip->i_df.if_bytes = 0;
b2c20045SChristoph Hellwig	}
98c4f78dSDarrick J. Wong
c19b3b05SDave Chinner	VFS_I(ip)->i_mode = 0;		/* mark incore inode as free */
db07349dSChristoph Hellwig	ip->i_diflags = 0;
f40aadb2SDave Chinner	ip->i_diflags2 = mp->m_ino_geo.new_diflags2;
7821ea30SChristoph Hellwig	ip->i_forkoff = 0;		/* mark the attr fork not in use */
f7e67b20SChristoph Hellwig	ip->i_df.if_format = XFS_DINODE_FMT_EXTENTS;
9b3beb02SChristoph Hellwig	if (xfs_iflags_test(ip, XFS_IPRESERVE_DM_FIELDS))
9b3beb02SChristoph Hellwig		xfs_iflags_clear(ip, XFS_IPRESERVE_DM_FIELDS);
dc1baa71SEric Sandeen
dc1baa71SEric Sandeen	/* Don't attempt to replay owner changes for a deleted inode */
1319ebefSDave Chinner	spin_lock(&iip->ili_lock);
1319ebefSDave Chinner	iip->ili_fields &= ~(XFS_ILOG_AOWNER | XFS_ILOG_DOWNER);
1319ebefSDave Chinner	spin_unlock(&iip->ili_lock);
dc1baa71SEric Sandeen
1da177e4SLinus Torvalds	/*
1da177e4SLinus Torvalds	 * Bump the generation count so no one will be confused
1da177e4SLinus Torvalds	 * by reincarnations of this inode.
1da177e4SLinus Torvalds	 */
9e9a2674SDave Chinner	VFS_I(ip)->i_generation++;
1da177e4SLinus Torvalds	xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
1da177e4SLinus Torvalds
09b56604SBrian Foster	if (xic.deleted)
f40aadb2SDave Chinner		error = xfs_ifree_cluster(tp, pag, ip, &xic);
f40aadb2SDave Chinnerout:
f40aadb2SDave Chinner	xfs_perag_put(pag);
2a30f36dSChandra Seetharaman	return error;
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds/*
60ec6783SChristoph Hellwig * This is called to unpin an inode.  The caller must have the inode locked
60ec6783SChristoph Hellwig * in at least shared mode so that the buffer cannot be subsequently pinned
60ec6783SChristoph Hellwig * once someone is waiting for it to be unpinned.
1da177e4SLinus Torvalds */
60ec6783SChristoph Hellwigstatic void
f392e631SChristoph Hellwigxfs_iunpin(
60ec6783SChristoph Hellwig	struct xfs_inode	*ip)
a3f74ffbSDavid Chinner{
579aa9caSChristoph Hellwig	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL|XFS_ILOCK_SHARED));
a3f74ffbSDavid Chinner
4aaf15d1SDave Chinner	trace_xfs_inode_unpin_nowait(ip, _RET_IP_);
4aaf15d1SDave Chinner
a3f74ffbSDavid Chinner	/* Give the log a push to start the unpinning I/O */
5f9b4b0dSDave Chinner	xfs_log_force_seq(ip->i_mount, ip->i_itemp->ili_commit_seq, 0, NULL);
a14a348bSChristoph Hellwig
a3f74ffbSDavid Chinner}
a3f74ffbSDavid Chinner
f392e631SChristoph Hellwigstatic void
f392e631SChristoph Hellwig__xfs_iunpin_wait(
f392e631SChristoph Hellwig	struct xfs_inode	*ip)
f392e631SChristoph Hellwig{
f392e631SChristoph Hellwig	wait_queue_head_t *wq = bit_waitqueue(&ip->i_flags, __XFS_IPINNED_BIT);
f392e631SChristoph Hellwig	DEFINE_WAIT_BIT(wait, &ip->i_flags, __XFS_IPINNED_BIT);
f392e631SChristoph Hellwig
f392e631SChristoph Hellwig	xfs_iunpin(ip);
f392e631SChristoph Hellwig
f392e631SChristoph Hellwig	do {
21417136SIngo Molnar		prepare_to_wait(wq, &wait.wq_entry, TASK_UNINTERRUPTIBLE);
f392e631SChristoph Hellwig		if (xfs_ipincount(ip))
f392e631SChristoph Hellwig			io_schedule();
f392e631SChristoph Hellwig	} while (xfs_ipincount(ip));
21417136SIngo Molnar	finish_wait(wq, &wait.wq_entry);
f392e631SChristoph Hellwig}
f392e631SChristoph Hellwig
777df5afSDave Chinnervoid
1da177e4SLinus Torvaldsxfs_iunpin_wait(
60ec6783SChristoph Hellwig	struct xfs_inode	*ip)
1da177e4SLinus Torvalds{
f392e631SChristoph Hellwig	if (xfs_ipincount(ip))
f392e631SChristoph Hellwig		__xfs_iunpin_wait(ip);
1da177e4SLinus Torvalds}
1da177e4SLinus Torvalds
27320369SDave Chinner/*
27320369SDave Chinner * Removing an inode from the namespace involves removing the directory entry
27320369SDave Chinner * and dropping the link count on the inode. Removing the directory entry can
27320369SDave Chinner * result in locking an AGF (directory blocks were freed) and removing a link
27320369SDave Chinner * count can result in placing the inode on an unlinked list which results in
27320369SDave Chinner * locking an AGI.
27320369SDave Chinner *
27320369SDave Chinner * The big problem here is that we have an ordering constraint on AGF and AGI
27320369SDave Chinner * locking - inode allocation locks the AGI, then can allocate a new extent for
27320369SDave Chinner * new inodes, locking the AGF after the AGI. Similarly, freeing the inode
27320369SDave Chinner * removes the inode from the unlinked list, requiring that we lock the AGI
27320369SDave Chinner * first, and then freeing the inode can result in an inode chunk being freed
27320369SDave Chinner * and hence freeing disk space requiring that we lock an AGF.
27320369SDave Chinner *
27320369SDave Chinner * Hence the ordering that is imposed by other parts of the code is AGI before
27320369SDave Chinner * AGF. This means we cannot remove the directory entry before we drop the inode
27320369SDave Chinner * reference count and put it on the unlinked list as this results in a lock
27320369SDave Chinner * order of AGF then AGI, and this can deadlock against inode allocation and
27320369SDave Chinner * freeing. Therefore we must drop the link counts before we remove the
27320369SDave Chinner * directory entry.
27320369SDave Chinner *
27320369SDave Chinner * This is still safe from a transactional point of view - it is not until we
310a75a3SDarrick J. Wong * get to xfs_defer_finish() that we have the possibility of multiple
27320369SDave Chinner * transactions in this operation. Hence as long as we remove the directory
27320369SDave Chinner * entry and drop the link count in the first transaction of the remove
27320369SDave Chinner * operation, there are no transactional constraints on the ordering here.
27320369SDave Chinner */
c24b5dfaSDave Chinnerint
c24b5dfaSDave Chinnerxfs_remove(
c24b5dfaSDave Chinner	xfs_inode_t             *dp,
c24b5dfaSDave Chinner	struct xfs_name		*name,
c24b5dfaSDave Chinner	xfs_inode_t		*ip)
c24b5dfaSDave Chinner{
c24b5dfaSDave Chinner	xfs_mount_t		*mp = dp->i_mount;
c24b5dfaSDave Chinner	xfs_trans_t             *tp = NULL;
c19b3b05SDave Chinner	int			is_dir = S_ISDIR(VFS_I(ip)->i_mode);
871b9316SDarrick J. Wong	int			dontcare;
c24b5dfaSDave Chinner	int                     error = 0;
c24b5dfaSDave Chinner	uint			resblks;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	trace_xfs_remove(dp, name);
c24b5dfaSDave Chinner
75c8c50fSDave Chinner	if (xfs_is_shutdown(mp))
2451337dSDave Chinner		return -EIO;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(dp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
c14cfccaSDarrick J. Wong	error = xfs_qm_dqattach(ip);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
871b9316SDarrick J. Wong	 * We try to get the real space reservation first, allowing for
871b9316SDarrick J. Wong	 * directory btree deletion(s) implying possible bmap insert(s).  If we
871b9316SDarrick J. Wong	 * can't get the space reservation then we use 0 instead, and avoid the
871b9316SDarrick J. Wong	 * bmap btree insert(s) in the directory code by, if the bmap insert
871b9316SDarrick J. Wong	 * tries to happen, instead trimming the LAST block from the directory.
871b9316SDarrick J. Wong	 *
871b9316SDarrick J. Wong	 * Ignore EDQUOT and ENOSPC being returned via nospace_error because
871b9316SDarrick J. Wong	 * the directory code can handle a reservationless update and we don't
871b9316SDarrick J. Wong	 * want to prevent a user from trying to free space by deleting things.
c24b5dfaSDave Chinner	 */
c24b5dfaSDave Chinner	resblks = XFS_REMOVE_SPACE_RES(mp);
871b9316SDarrick J. Wong	error = xfs_trans_alloc_dir(dp, &M_RES(mp)->tr_remove, ip, &resblks,
871b9316SDarrick J. Wong			&tp, &dontcare);
c24b5dfaSDave Chinner	if (error) {
2451337dSDave Chinner		ASSERT(error != -ENOSPC);
253f4911SChristoph Hellwig		goto std_return;
c24b5dfaSDave Chinner	}
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If we're removing a directory perform some additional validation.
c24b5dfaSDave Chinner	 */
c24b5dfaSDave Chinner	if (is_dir) {
54d7b5c1SDave Chinner		ASSERT(VFS_I(ip)->i_nlink >= 2);
54d7b5c1SDave Chinner		if (VFS_I(ip)->i_nlink != 2) {
2451337dSDave Chinner			error = -ENOTEMPTY;
c24b5dfaSDave Chinner			goto out_trans_cancel;
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner		if (!xfs_dir_isempty(ip)) {
2451337dSDave Chinner			error = -ENOTEMPTY;
c24b5dfaSDave Chinner			goto out_trans_cancel;
c24b5dfaSDave Chinner		}
c24b5dfaSDave Chinner
27320369SDave Chinner		/* Drop the link from ip's "..".  */
c24b5dfaSDave Chinner		error = xfs_droplink(tp, dp);
c24b5dfaSDave Chinner		if (error)
27320369SDave Chinner			goto out_trans_cancel;
c24b5dfaSDave Chinner
27320369SDave Chinner		/* Drop the "." link from ip to self.  */
c24b5dfaSDave Chinner		error = xfs_droplink(tp, ip);
c24b5dfaSDave Chinner		if (error)
27320369SDave Chinner			goto out_trans_cancel;
5838d035SDarrick J. Wong
5838d035SDarrick J. Wong		/*
5838d035SDarrick J. Wong		 * Point the unlinked child directory's ".." entry to the root
5838d035SDarrick J. Wong		 * directory to eliminate back-references to inodes that may
5838d035SDarrick J. Wong		 * get freed before the child directory is closed.  If the fs
5838d035SDarrick J. Wong		 * gets shrunk, this can lead to dirent inode validation errors.
5838d035SDarrick J. Wong		 */
5838d035SDarrick J. Wong		if (dp->i_ino != tp->t_mountp->m_sb.sb_rootino) {
5838d035SDarrick J. Wong			error = xfs_dir_replace(tp, ip, &xfs_name_dotdot,
5838d035SDarrick J. Wong					tp->t_mountp->m_sb.sb_rootino, 0);
5838d035SDarrick J. Wong			if (error)
2653d533SDarrick J. Wong				goto out_trans_cancel;
5838d035SDarrick J. Wong		}
c24b5dfaSDave Chinner	} else {
c24b5dfaSDave Chinner		/*
c24b5dfaSDave Chinner		 * When removing a non-directory we need to log the parent
c24b5dfaSDave Chinner		 * inode here.  For a directory this is done implicitly
c24b5dfaSDave Chinner		 * by the xfs_droplink call for the ".." entry.
c24b5dfaSDave Chinner		 */
c24b5dfaSDave Chinner		xfs_trans_log_inode(tp, dp, XFS_ILOG_CORE);
c24b5dfaSDave Chinner	}
27320369SDave Chinner	xfs_trans_ichgtime(tp, dp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
c24b5dfaSDave Chinner
27320369SDave Chinner	/* Drop the link from dp to ip. */
c24b5dfaSDave Chinner	error = xfs_droplink(tp, ip);
c24b5dfaSDave Chinner	if (error)
27320369SDave Chinner		goto out_trans_cancel;
c24b5dfaSDave Chinner
381eee69SBrian Foster	error = xfs_dir_removename(tp, dp, name, ip->i_ino, resblks);
27320369SDave Chinner	if (error) {
2451337dSDave Chinner		ASSERT(error != -ENOENT);
c8eac49eSBrian Foster		goto out_trans_cancel;
27320369SDave Chinner	}
27320369SDave Chinner
c24b5dfaSDave Chinner	/*
c24b5dfaSDave Chinner	 * If this is a synchronous mount, make sure that the
c24b5dfaSDave Chinner	 * remove transaction goes to disk before returning to
c24b5dfaSDave Chinner	 * the user.
c24b5dfaSDave Chinner	 */
0560f31aSDave Chinner	if (xfs_has_wsync(mp) || xfs_has_dirsync(mp))
c24b5dfaSDave Chinner		xfs_trans_set_sync(tp);
c24b5dfaSDave Chinner
70393313SChristoph Hellwig	error = xfs_trans_commit(tp);
c24b5dfaSDave Chinner	if (error)
c24b5dfaSDave Chinner		goto std_return;
c24b5dfaSDave Chinner
2cd2ef6aSChristoph Hellwig	if (is_dir && xfs_inode_is_filestream(ip))
c24b5dfaSDave Chinner		xfs_filestream_deassociate(ip);
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner	return 0;
c24b5dfaSDave Chinner
c24b5dfaSDave Chinner out_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
c24b5dfaSDave Chinner std_return:
c24b5dfaSDave Chinner	return error;
c24b5dfaSDave Chinner}
c24b5dfaSDave Chinner
f6bba201SDave Chinner/*
f6bba201SDave Chinner * Enter all inodes for a rename transaction into a sorted array.
f6bba201SDave Chinner */
95afcf5cSDave Chinner#define __XFS_SORT_INODES	5
f6bba201SDave ChinnerSTATIC void
f6bba201SDave Chinnerxfs_sort_for_rename(
95afcf5cSDave Chinner	struct xfs_inode	*dp1,	/* in: old (source) directory inode */
95afcf5cSDave Chinner	struct xfs_inode	*dp2,	/* in: new (target) directory inode */
95afcf5cSDave Chinner	struct xfs_inode	*ip1,	/* in: inode of old entry */
95afcf5cSDave Chinner	struct xfs_inode	*ip2,	/* in: inode of new entry */
95afcf5cSDave Chinner	struct xfs_inode	*wip,	/* in: whiteout inode */
95afcf5cSDave Chinner	struct xfs_inode	**i_tab,/* out: sorted array of inodes */
95afcf5cSDave Chinner	int			*num_inodes)  /* in/out: inodes in array */
f6bba201SDave Chinner{
f6bba201SDave Chinner	int			i, j;
f6bba201SDave Chinner
95afcf5cSDave Chinner	ASSERT(*num_inodes == __XFS_SORT_INODES);
95afcf5cSDave Chinner	memset(i_tab, 0, *num_inodes * sizeof(struct xfs_inode *));
95afcf5cSDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * i_tab contains a list of pointers to inodes.  We initialize
f6bba201SDave Chinner	 * the table here & we'll sort it.  We will then use it to
f6bba201SDave Chinner	 * order the acquisition of the inode locks.
f6bba201SDave Chinner	 *
f6bba201SDave Chinner	 * Note that the table may contain duplicates.  e.g., dp1 == dp2.
f6bba201SDave Chinner	 */
95afcf5cSDave Chinner	i = 0;
95afcf5cSDave Chinner	i_tab[i++] = dp1;
95afcf5cSDave Chinner	i_tab[i++] = dp2;
95afcf5cSDave Chinner	i_tab[i++] = ip1;
95afcf5cSDave Chinner	if (ip2)
95afcf5cSDave Chinner		i_tab[i++] = ip2;
95afcf5cSDave Chinner	if (wip)
95afcf5cSDave Chinner		i_tab[i++] = wip;
95afcf5cSDave Chinner	*num_inodes = i;
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Sort the elements via bubble sort.  (Remember, there are at
95afcf5cSDave Chinner	 * most 5 elements to sort, so this is adequate.)
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	for (i = 0; i < *num_inodes; i++) {
f6bba201SDave Chinner		for (j = 1; j < *num_inodes; j++) {
f6bba201SDave Chinner			if (i_tab[j]->i_ino < i_tab[j-1]->i_ino) {
95afcf5cSDave Chinner				struct xfs_inode *temp = i_tab[j];
f6bba201SDave Chinner				i_tab[j] = i_tab[j-1];
f6bba201SDave Chinner				i_tab[j-1] = temp;
f6bba201SDave Chinner			}
f6bba201SDave Chinner		}
f6bba201SDave Chinner	}
f6bba201SDave Chinner}
f6bba201SDave Chinner
310606b0SDave Chinnerstatic int
310606b0SDave Chinnerxfs_finish_rename(
c9cfdb38SBrian Foster	struct xfs_trans	*tp)
310606b0SDave Chinner{
310606b0SDave Chinner	/*
310606b0SDave Chinner	 * If this is a synchronous mount, make sure that the rename transaction
310606b0SDave Chinner	 * goes to disk before returning to the user.
310606b0SDave Chinner	 */
0560f31aSDave Chinner	if (xfs_has_wsync(tp->t_mountp) || xfs_has_dirsync(tp->t_mountp))
310606b0SDave Chinner		xfs_trans_set_sync(tp);
310606b0SDave Chinner
70393313SChristoph Hellwig	return xfs_trans_commit(tp);
310606b0SDave Chinner}
310606b0SDave Chinner
f6bba201SDave Chinner/*
d31a1825SCarlos Maiolino * xfs_cross_rename()
d31a1825SCarlos Maiolino *
0145225eSBhaskar Chowdhury * responsible for handling RENAME_EXCHANGE flag in renameat2() syscall
d31a1825SCarlos Maiolino */
d31a1825SCarlos MaiolinoSTATIC int
d31a1825SCarlos Maiolinoxfs_cross_rename(
d31a1825SCarlos Maiolino	struct xfs_trans	*tp,
d31a1825SCarlos Maiolino	struct xfs_inode	*dp1,
d31a1825SCarlos Maiolino	struct xfs_name		*name1,
d31a1825SCarlos Maiolino	struct xfs_inode	*ip1,
d31a1825SCarlos Maiolino	struct xfs_inode	*dp2,
d31a1825SCarlos Maiolino	struct xfs_name		*name2,
d31a1825SCarlos Maiolino	struct xfs_inode	*ip2,
d31a1825SCarlos Maiolino	int			spaceres)
d31a1825SCarlos Maiolino{
d31a1825SCarlos Maiolino	int		error = 0;
d31a1825SCarlos Maiolino	int		ip1_flags = 0;
d31a1825SCarlos Maiolino	int		ip2_flags = 0;
d31a1825SCarlos Maiolino	int		dp2_flags = 0;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	/* Swap inode number for dirent in first parent */
381eee69SBrian Foster	error = xfs_dir_replace(tp, dp1, name1, ip2->i_ino, spaceres);
d31a1825SCarlos Maiolino	if (error)
eeacd321SDave Chinner		goto out_trans_abort;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	/* Swap inode number for dirent in second parent */
381eee69SBrian Foster	error = xfs_dir_replace(tp, dp2, name2, ip1->i_ino, spaceres);
d31a1825SCarlos Maiolino	if (error)
eeacd321SDave Chinner		goto out_trans_abort;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	/*
d31a1825SCarlos Maiolino	 * If we're renaming one or more directories across different parents,
d31a1825SCarlos Maiolino	 * update the respective ".." entries (and link counts) to match the new
d31a1825SCarlos Maiolino	 * parents.
d31a1825SCarlos Maiolino	 */
d31a1825SCarlos Maiolino	if (dp1 != dp2) {
d31a1825SCarlos Maiolino		dp2_flags = XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino
c19b3b05SDave Chinner		if (S_ISDIR(VFS_I(ip2)->i_mode)) {
d31a1825SCarlos Maiolino			error = xfs_dir_replace(tp, ip2, &xfs_name_dotdot,
381eee69SBrian Foster						dp1->i_ino, spaceres);
d31a1825SCarlos Maiolino			if (error)
eeacd321SDave Chinner				goto out_trans_abort;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino			/* transfer ip2 ".." reference to dp1 */
c19b3b05SDave Chinner			if (!S_ISDIR(VFS_I(ip1)->i_mode)) {
d31a1825SCarlos Maiolino				error = xfs_droplink(tp, dp2);
d31a1825SCarlos Maiolino				if (error)
eeacd321SDave Chinner					goto out_trans_abort;
91083269SEric Sandeen				xfs_bumplink(tp, dp1);
d31a1825SCarlos Maiolino			}
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino			/*
d31a1825SCarlos Maiolino			 * Although ip1 isn't changed here, userspace needs
d31a1825SCarlos Maiolino			 * to be warned about the change, so that applications
d31a1825SCarlos Maiolino			 * relying on it (like backup ones), will properly
d31a1825SCarlos Maiolino			 * notify the change
d31a1825SCarlos Maiolino			 */
d31a1825SCarlos Maiolino			ip1_flags |= XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino			ip2_flags |= XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino		}
d31a1825SCarlos Maiolino
c19b3b05SDave Chinner		if (S_ISDIR(VFS_I(ip1)->i_mode)) {
d31a1825SCarlos Maiolino			error = xfs_dir_replace(tp, ip1, &xfs_name_dotdot,
381eee69SBrian Foster						dp2->i_ino, spaceres);
d31a1825SCarlos Maiolino			if (error)
eeacd321SDave Chinner				goto out_trans_abort;
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino			/* transfer ip1 ".." reference to dp2 */
c19b3b05SDave Chinner			if (!S_ISDIR(VFS_I(ip2)->i_mode)) {
d31a1825SCarlos Maiolino				error = xfs_droplink(tp, dp1);
d31a1825SCarlos Maiolino				if (error)
eeacd321SDave Chinner					goto out_trans_abort;
91083269SEric Sandeen				xfs_bumplink(tp, dp2);
d31a1825SCarlos Maiolino			}
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino			/*
d31a1825SCarlos Maiolino			 * Although ip2 isn't changed here, userspace needs
d31a1825SCarlos Maiolino			 * to be warned about the change, so that applications
d31a1825SCarlos Maiolino			 * relying on it (like backup ones), will properly
d31a1825SCarlos Maiolino			 * notify the change
d31a1825SCarlos Maiolino			 */
d31a1825SCarlos Maiolino			ip1_flags |= XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino			ip2_flags |= XFS_ICHGTIME_CHG;
d31a1825SCarlos Maiolino		}
d31a1825SCarlos Maiolino	}
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	if (ip1_flags) {
d31a1825SCarlos Maiolino		xfs_trans_ichgtime(tp, ip1, ip1_flags);
d31a1825SCarlos Maiolino		xfs_trans_log_inode(tp, ip1, XFS_ILOG_CORE);
d31a1825SCarlos Maiolino	}
d31a1825SCarlos Maiolino	if (ip2_flags) {
d31a1825SCarlos Maiolino		xfs_trans_ichgtime(tp, ip2, ip2_flags);
d31a1825SCarlos Maiolino		xfs_trans_log_inode(tp, ip2, XFS_ILOG_CORE);
d31a1825SCarlos Maiolino	}
d31a1825SCarlos Maiolino	if (dp2_flags) {
d31a1825SCarlos Maiolino		xfs_trans_ichgtime(tp, dp2, dp2_flags);
d31a1825SCarlos Maiolino		xfs_trans_log_inode(tp, dp2, XFS_ILOG_CORE);
d31a1825SCarlos Maiolino	}
d31a1825SCarlos Maiolino	xfs_trans_ichgtime(tp, dp1, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
d31a1825SCarlos Maiolino	xfs_trans_log_inode(tp, dp1, XFS_ILOG_CORE);
c9cfdb38SBrian Foster	return xfs_finish_rename(tp);
eeacd321SDave Chinner
eeacd321SDave Chinnerout_trans_abort:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
d31a1825SCarlos Maiolino	return error;
d31a1825SCarlos Maiolino}
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino/*
7dcf5c3eSDave Chinner * xfs_rename_alloc_whiteout()
7dcf5c3eSDave Chinner *
b63da6c8SRandy Dunlap * Return a referenced, unlinked, unlocked inode that can be used as a
7dcf5c3eSDave Chinner * whiteout in a rename transaction. We use a tmpfile inode here so that if we
7dcf5c3eSDave Chinner * crash between allocating the inode and linking it into the rename transaction
7dcf5c3eSDave Chinner * recovery will free the inode and we won't leak it.
7dcf5c3eSDave Chinner */
7dcf5c3eSDave Chinnerstatic int
7dcf5c3eSDave Chinnerxfs_rename_alloc_whiteout(
f2d40141SChristian Brauner	struct mnt_idmap	*idmap,
70b589a3SEric Sandeen	struct xfs_name		*src_name,
7dcf5c3eSDave Chinner	struct xfs_inode	*dp,
7dcf5c3eSDave Chinner	struct xfs_inode	**wip)
7dcf5c3eSDave Chinner{
7dcf5c3eSDave Chinner	struct xfs_inode	*tmpfile;
70b589a3SEric Sandeen	struct qstr		name;
7dcf5c3eSDave Chinner	int			error;
7dcf5c3eSDave Chinner
f2d40141SChristian Brauner	error = xfs_create_tmpfile(idmap, dp, S_IFCHR | WHITEOUT_MODE,
f736d93dSChristoph Hellwig				   &tmpfile);
7dcf5c3eSDave Chinner	if (error)
7dcf5c3eSDave Chinner		return error;
7dcf5c3eSDave Chinner
70b589a3SEric Sandeen	name.name = src_name->name;
70b589a3SEric Sandeen	name.len = src_name->len;
70b589a3SEric Sandeen	error = xfs_inode_init_security(VFS_I(tmpfile), VFS_I(dp), &name);
70b589a3SEric Sandeen	if (error) {
70b589a3SEric Sandeen		xfs_finish_inode_setup(tmpfile);
70b589a3SEric Sandeen		xfs_irele(tmpfile);
70b589a3SEric Sandeen		return error;
70b589a3SEric Sandeen	}
70b589a3SEric Sandeen
22419ac9SBrian Foster	/*
22419ac9SBrian Foster	 * Prepare the tmpfile inode as if it were created through the VFS.
c4a6bf7fSDarrick J. Wong	 * Complete the inode setup and flag it as linkable.  nlink is already
c4a6bf7fSDarrick J. Wong	 * zero, so we can skip the drop_nlink.
22419ac9SBrian Foster	 */
2b3d1d41SChristoph Hellwig	xfs_setup_iops(tmpfile);
7dcf5c3eSDave Chinner	xfs_finish_inode_setup(tmpfile);
7dcf5c3eSDave Chinner	VFS_I(tmpfile)->i_state |= I_LINKABLE;
7dcf5c3eSDave Chinner
7dcf5c3eSDave Chinner	*wip = tmpfile;
7dcf5c3eSDave Chinner	return 0;
7dcf5c3eSDave Chinner}
7dcf5c3eSDave Chinner
7dcf5c3eSDave Chinner/*
f6bba201SDave Chinner * xfs_rename
f6bba201SDave Chinner */
f6bba201SDave Chinnerint
f6bba201SDave Chinnerxfs_rename(
f2d40141SChristian Brauner	struct mnt_idmap	*idmap,
7dcf5c3eSDave Chinner	struct xfs_inode	*src_dp,
f6bba201SDave Chinner	struct xfs_name		*src_name,
7dcf5c3eSDave Chinner	struct xfs_inode	*src_ip,
7dcf5c3eSDave Chinner	struct xfs_inode	*target_dp,
f6bba201SDave Chinner	struct xfs_name		*target_name,
7dcf5c3eSDave Chinner	struct xfs_inode	*target_ip,
d31a1825SCarlos Maiolino	unsigned int		flags)
f6bba201SDave Chinner{
7dcf5c3eSDave Chinner	struct xfs_mount	*mp = src_dp->i_mount;
7dcf5c3eSDave Chinner	struct xfs_trans	*tp;
7dcf5c3eSDave Chinner	struct xfs_inode	*wip = NULL;		/* whiteout inode */
7dcf5c3eSDave Chinner	struct xfs_inode	*inodes[__XFS_SORT_INODES];
6da1b4b1SDarrick J. Wong	int			i;
95afcf5cSDave Chinner	int			num_inodes = __XFS_SORT_INODES;
2b93681fSDave Chinner	bool			new_parent = (src_dp != target_dp);
c19b3b05SDave Chinner	bool			src_is_directory = S_ISDIR(VFS_I(src_ip)->i_mode);
f6bba201SDave Chinner	int			spaceres;
41667260SDarrick J. Wong	bool			retried = false;
41667260SDarrick J. Wong	int			error, nospace_error = 0;
f6bba201SDave Chinner
f6bba201SDave Chinner	trace_xfs_rename(src_dp, target_dp, src_name, target_name);
f6bba201SDave Chinner
eeacd321SDave Chinner	if ((flags & RENAME_EXCHANGE) && !target_ip)
eeacd321SDave Chinner		return -EINVAL;
f6bba201SDave Chinner
7dcf5c3eSDave Chinner	/*
7dcf5c3eSDave Chinner	 * If we are doing a whiteout operation, allocate the whiteout inode
7dcf5c3eSDave Chinner	 * we will be placing at the target and ensure the type is set
7dcf5c3eSDave Chinner	 * appropriately.
7dcf5c3eSDave Chinner	 */
7dcf5c3eSDave Chinner	if (flags & RENAME_WHITEOUT) {
f2d40141SChristian Brauner		error = xfs_rename_alloc_whiteout(idmap, src_name,
70b589a3SEric Sandeen						  target_dp, &wip);
7dcf5c3eSDave Chinner		if (error)
7dcf5c3eSDave Chinner			return error;
f6bba201SDave Chinner
7dcf5c3eSDave Chinner		/* setup target dirent info as whiteout */
7dcf5c3eSDave Chinner		src_name->type = XFS_DIR3_FT_CHRDEV;
7dcf5c3eSDave Chinner	}
7dcf5c3eSDave Chinner
7dcf5c3eSDave Chinner	xfs_sort_for_rename(src_dp, target_dp, src_ip, target_ip, wip,
f6bba201SDave Chinner				inodes, &num_inodes);
f6bba201SDave Chinner
41667260SDarrick J. Wongretry:
41667260SDarrick J. Wong	nospace_error = 0;
f6bba201SDave Chinner	spaceres = XFS_RENAME_SPACE_RES(mp, target_name->len);
253f4911SChristoph Hellwig	error = xfs_trans_alloc(mp, &M_RES(mp)->tr_rename, spaceres, 0, 0, &tp);
2451337dSDave Chinner	if (error == -ENOSPC) {
41667260SDarrick J. Wong		nospace_error = error;
f6bba201SDave Chinner		spaceres = 0;
253f4911SChristoph Hellwig		error = xfs_trans_alloc(mp, &M_RES(mp)->tr_rename, 0, 0, 0,
253f4911SChristoph Hellwig				&tp);
f6bba201SDave Chinner	}
445883e8SDave Chinner	if (error)
253f4911SChristoph Hellwig		goto out_release_wip;
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Attach the dquots to the inodes
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	error = xfs_qm_vop_rename_dqattach(inodes);
445883e8SDave Chinner	if (error)
445883e8SDave Chinner		goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Lock all the participating inodes. Depending upon whether
f6bba201SDave Chinner	 * the target_name exists in the target directory, and
f6bba201SDave Chinner	 * whether the target directory is the same as the source
e07ee6feSAllison Henderson	 * directory, we can lock from 2 to 5 inodes.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	xfs_lock_inodes(inodes, num_inodes, XFS_ILOCK_EXCL);
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Join all the inodes to the transaction. From this point on,
f6bba201SDave Chinner	 * we can rely on either trans_commit or trans_cancel to unlock
f6bba201SDave Chinner	 * them.
f6bba201SDave Chinner	 */
65523218SChristoph Hellwig	xfs_trans_ijoin(tp, src_dp, XFS_ILOCK_EXCL);
f6bba201SDave Chinner	if (new_parent)
65523218SChristoph Hellwig		xfs_trans_ijoin(tp, target_dp, XFS_ILOCK_EXCL);
f6bba201SDave Chinner	xfs_trans_ijoin(tp, src_ip, XFS_ILOCK_EXCL);
f6bba201SDave Chinner	if (target_ip)
f6bba201SDave Chinner		xfs_trans_ijoin(tp, target_ip, XFS_ILOCK_EXCL);
7dcf5c3eSDave Chinner	if (wip)
7dcf5c3eSDave Chinner		xfs_trans_ijoin(tp, wip, XFS_ILOCK_EXCL);
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * If we are using project inheritance, we only allow renames
f6bba201SDave Chinner	 * into our tree when the project IDs are the same; else the
f6bba201SDave Chinner	 * tree quota mechanism would be circumvented.
f6bba201SDave Chinner	 */
db07349dSChristoph Hellwig	if (unlikely((target_dp->i_diflags & XFS_DIFLAG_PROJINHERIT) &&
ceaf603cSChristoph Hellwig		     target_dp->i_projid != src_ip->i_projid)) {
2451337dSDave Chinner		error = -EXDEV;
445883e8SDave Chinner		goto out_trans_cancel;
f6bba201SDave Chinner	}
f6bba201SDave Chinner
eeacd321SDave Chinner	/* RENAME_EXCHANGE is unique from here on. */
eeacd321SDave Chinner	if (flags & RENAME_EXCHANGE)
eeacd321SDave Chinner		return xfs_cross_rename(tp, src_dp, src_name, src_ip,
d31a1825SCarlos Maiolino					target_dp, target_name, target_ip,
f16dea54SBrian Foster					spaceres);
d31a1825SCarlos Maiolino
d31a1825SCarlos Maiolino	/*
41667260SDarrick J. Wong	 * Try to reserve quota to handle an expansion of the target directory.
41667260SDarrick J. Wong	 * We'll allow the rename to continue in reservationless mode if we hit
41667260SDarrick J. Wong	 * a space usage constraint.  If we trigger reservationless mode, save
41667260SDarrick J. Wong	 * the errno if there isn't any free space in the target directory.
41667260SDarrick J. Wong	 */
41667260SDarrick J. Wong	if (spaceres != 0) {
41667260SDarrick J. Wong		error = xfs_trans_reserve_quota_nblks(tp, target_dp, spaceres,
41667260SDarrick J. Wong				0, false);
41667260SDarrick J. Wong		if (error == -EDQUOT || error == -ENOSPC) {
41667260SDarrick J. Wong			if (!retried) {
41667260SDarrick J. Wong				xfs_trans_cancel(tp);
41667260SDarrick J. Wong				xfs_blockgc_free_quota(target_dp, 0);
41667260SDarrick J. Wong				retried = true;
41667260SDarrick J. Wong				goto retry;
41667260SDarrick J. Wong			}
41667260SDarrick J. Wong
41667260SDarrick J. Wong			nospace_error = error;
41667260SDarrick J. Wong			spaceres = 0;
41667260SDarrick J. Wong			error = 0;
41667260SDarrick J. Wong		}
41667260SDarrick J. Wong		if (error)
41667260SDarrick J. Wong			goto out_trans_cancel;
41667260SDarrick J. Wong	}
41667260SDarrick J. Wong
41667260SDarrick J. Wong	/*
bc56ad8cSkaixuxia	 * Check for expected errors before we dirty the transaction
bc56ad8cSkaixuxia	 * so we can return an error without a transaction abort.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	if (target_ip == NULL) {
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * If there's no space reservation, check the entry will
f6bba201SDave Chinner		 * fit before actually inserting it.
f6bba201SDave Chinner		 */
94f3cad5SEric Sandeen		if (!spaceres) {
94f3cad5SEric Sandeen			error = xfs_dir_canenter(tp, target_dp, target_name);
f6bba201SDave Chinner			if (error)
445883e8SDave Chinner				goto out_trans_cancel;
94f3cad5SEric Sandeen		}
bc56ad8cSkaixuxia	} else {
bc56ad8cSkaixuxia		/*
bc56ad8cSkaixuxia		 * If target exists and it's a directory, check that whether
bc56ad8cSkaixuxia		 * it can be destroyed.
bc56ad8cSkaixuxia		 */
bc56ad8cSkaixuxia		if (S_ISDIR(VFS_I(target_ip)->i_mode) &&
bc56ad8cSkaixuxia		    (!xfs_dir_isempty(target_ip) ||
bc56ad8cSkaixuxia		     (VFS_I(target_ip)->i_nlink > 2))) {
bc56ad8cSkaixuxia			error = -EEXIST;
bc56ad8cSkaixuxia			goto out_trans_cancel;
bc56ad8cSkaixuxia		}
bc56ad8cSkaixuxia	}
bc56ad8cSkaixuxia
bc56ad8cSkaixuxia	/*
6da1b4b1SDarrick J. Wong	 * Lock the AGI buffers we need to handle bumping the nlink of the
6da1b4b1SDarrick J. Wong	 * whiteout inode off the unlinked list and to handle dropping the
6da1b4b1SDarrick J. Wong	 * nlink of the target inode.  Per locking order rules, do this in
6da1b4b1SDarrick J. Wong	 * increasing AG order and before directory block allocation tries to
6da1b4b1SDarrick J. Wong	 * grab AGFs because we grab AGIs before AGFs.
6da1b4b1SDarrick J. Wong	 *
6da1b4b1SDarrick J. Wong	 * The (vfs) caller must ensure that if src is a directory then
6da1b4b1SDarrick J. Wong	 * target_ip is either null or an empty directory.
6da1b4b1SDarrick J. Wong	 */
6da1b4b1SDarrick J. Wong	for (i = 0; i < num_inodes && inodes[i] != NULL; i++) {
6da1b4b1SDarrick J. Wong		if (inodes[i] == wip ||
6da1b4b1SDarrick J. Wong		    (inodes[i] == target_ip &&
6da1b4b1SDarrick J. Wong		     (VFS_I(target_ip)->i_nlink == 1 || src_is_directory))) {
61021debSDave Chinner			struct xfs_perag	*pag;
6da1b4b1SDarrick J. Wong			struct xfs_buf		*bp;
6da1b4b1SDarrick J. Wong
61021debSDave Chinner			pag = xfs_perag_get(mp,
61021debSDave Chinner					XFS_INO_TO_AGNO(mp, inodes[i]->i_ino));
61021debSDave Chinner			error = xfs_read_agi(pag, tp, &bp);
61021debSDave Chinner			xfs_perag_put(pag);
6da1b4b1SDarrick J. Wong			if (error)
6da1b4b1SDarrick J. Wong				goto out_trans_cancel;
6da1b4b1SDarrick J. Wong		}
6da1b4b1SDarrick J. Wong	}
6da1b4b1SDarrick J. Wong
6da1b4b1SDarrick J. Wong	/*
bc56ad8cSkaixuxia	 * Directory entry creation below may acquire the AGF. Remove
bc56ad8cSkaixuxia	 * the whiteout from the unlinked list first to preserve correct
bc56ad8cSkaixuxia	 * AGI/AGF locking order. This dirties the transaction so failures
bc56ad8cSkaixuxia	 * after this point will abort and log recovery will clean up the
bc56ad8cSkaixuxia	 * mess.
bc56ad8cSkaixuxia	 *
bc56ad8cSkaixuxia	 * For whiteouts, we need to bump the link count on the whiteout
bc56ad8cSkaixuxia	 * inode. After this point, we have a real link, clear the tmpfile
bc56ad8cSkaixuxia	 * state flag from the inode so it doesn't accidentally get misused
bc56ad8cSkaixuxia	 * in future.
bc56ad8cSkaixuxia	 */
bc56ad8cSkaixuxia	if (wip) {
f40aadb2SDave Chinner		struct xfs_perag	*pag;
f40aadb2SDave Chinner
bc56ad8cSkaixuxia		ASSERT(VFS_I(wip)->i_nlink == 0);
f40aadb2SDave Chinner
f40aadb2SDave Chinner		pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, wip->i_ino));
f40aadb2SDave Chinner		error = xfs_iunlink_remove(tp, pag, wip);
f40aadb2SDave Chinner		xfs_perag_put(pag);
bc56ad8cSkaixuxia		if (error)
bc56ad8cSkaixuxia			goto out_trans_cancel;
bc56ad8cSkaixuxia
bc56ad8cSkaixuxia		xfs_bumplink(tp, wip);
bc56ad8cSkaixuxia		VFS_I(wip)->i_state &= ~I_LINKABLE;
bc56ad8cSkaixuxia	}
bc56ad8cSkaixuxia
bc56ad8cSkaixuxia	/*
bc56ad8cSkaixuxia	 * Set up the target.
bc56ad8cSkaixuxia	 */
bc56ad8cSkaixuxia	if (target_ip == NULL) {
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * If target does not exist and the rename crosses
f6bba201SDave Chinner		 * directories, adjust the target directory link count
f6bba201SDave Chinner		 * to account for the ".." reference from the new entry.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_dir_createname(tp, target_dp, target_name,
381eee69SBrian Foster					   src_ip->i_ino, spaceres);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner		xfs_trans_ichgtime(tp, target_dp,
f6bba201SDave Chinner					XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
f6bba201SDave Chinner
f6bba201SDave Chinner		if (new_parent && src_is_directory) {
91083269SEric Sandeen			xfs_bumplink(tp, target_dp);
f6bba201SDave Chinner		}
f6bba201SDave Chinner	} else { /* target_ip != NULL */
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * Link the source inode under the target name.
f6bba201SDave Chinner		 * If the source inode is a directory and we are moving
f6bba201SDave Chinner		 * it across directories, its ".." entry will be
f6bba201SDave Chinner		 * inconsistent until we replace that down below.
f6bba201SDave Chinner		 *
f6bba201SDave Chinner		 * In case there is already an entry with the same
f6bba201SDave Chinner		 * name at the destination directory, remove it first.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_dir_replace(tp, target_dp, target_name,
381eee69SBrian Foster					src_ip->i_ino, spaceres);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner		xfs_trans_ichgtime(tp, target_dp,
f6bba201SDave Chinner					XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
f6bba201SDave Chinner
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * Decrement the link count on the target since the target
f6bba201SDave Chinner		 * dir no longer points to it.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_droplink(tp, target_ip);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner		if (src_is_directory) {
f6bba201SDave Chinner			/*
f6bba201SDave Chinner			 * Drop the link from the old "." entry.
f6bba201SDave Chinner			 */
f6bba201SDave Chinner			error = xfs_droplink(tp, target_ip);
f6bba201SDave Chinner			if (error)
c8eac49eSBrian Foster				goto out_trans_cancel;
f6bba201SDave Chinner		}
f6bba201SDave Chinner	} /* target_ip != NULL */
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Remove the source.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	if (new_parent && src_is_directory) {
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * Rewrite the ".." entry to point to the new
f6bba201SDave Chinner		 * directory.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_dir_replace(tp, src_ip, &xfs_name_dotdot,
381eee69SBrian Foster					target_dp->i_ino, spaceres);
2451337dSDave Chinner		ASSERT(error != -EEXIST);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner	}
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * We always want to hit the ctime on the source inode.
f6bba201SDave Chinner	 *
f6bba201SDave Chinner	 * This isn't strictly required by the standards since the source
f6bba201SDave Chinner	 * inode isn't really being changed, but old unix file systems did
f6bba201SDave Chinner	 * it and some incremental backup programs won't work without it.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	xfs_trans_ichgtime(tp, src_ip, XFS_ICHGTIME_CHG);
f6bba201SDave Chinner	xfs_trans_log_inode(tp, src_ip, XFS_ILOG_CORE);
f6bba201SDave Chinner
f6bba201SDave Chinner	/*
f6bba201SDave Chinner	 * Adjust the link count on src_dp.  This is necessary when
f6bba201SDave Chinner	 * renaming a directory, either within one parent when
f6bba201SDave Chinner	 * the target existed, or across two parent directories.
f6bba201SDave Chinner	 */
f6bba201SDave Chinner	if (src_is_directory && (new_parent || target_ip != NULL)) {
f6bba201SDave Chinner
f6bba201SDave Chinner		/*
f6bba201SDave Chinner		 * Decrement link count on src_directory since the
f6bba201SDave Chinner		 * entry that's moved no longer points to it.
f6bba201SDave Chinner		 */
f6bba201SDave Chinner		error = xfs_droplink(tp, src_dp);
f6bba201SDave Chinner		if (error)
c8eac49eSBrian Foster			goto out_trans_cancel;
f6bba201SDave Chinner	}
f6bba201SDave Chinner
7dcf5c3eSDave Chinner	/*
7dcf5c3eSDave Chinner	 * For whiteouts, we only need to update the source dirent with the
7dcf5c3eSDave Chinner	 * inode number of the whiteout inode rather than removing it
7dcf5c3eSDave Chinner	 * altogether.
7dcf5c3eSDave Chinner	 */
83a21c18SChandan Babu R	if (wip)
7dcf5c3eSDave Chinner		error = xfs_dir_replace(tp, src_dp, src_name, wip->i_ino,
381eee69SBrian Foster					spaceres);
83a21c18SChandan Babu R	else
f6bba201SDave Chinner		error = xfs_dir_removename(tp, src_dp, src_name, src_ip->i_ino,
381eee69SBrian Foster					   spaceres);
02092a2fSChandan Babu R
f6bba201SDave Chinner	if (error)
c8eac49eSBrian Foster		goto out_trans_cancel;
f6bba201SDave Chinner
f6bba201SDave Chinner	xfs_trans_ichgtime(tp, src_dp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
f6bba201SDave Chinner	xfs_trans_log_inode(tp, src_dp, XFS_ILOG_CORE);
f6bba201SDave Chinner	if (new_parent)
f6bba201SDave Chinner		xfs_trans_log_inode(tp, target_dp, XFS_ILOG_CORE);
f6bba201SDave Chinner
c9cfdb38SBrian Foster	error = xfs_finish_rename(tp);
7dcf5c3eSDave Chinner	if (wip)
44a8736bSDarrick J. Wong		xfs_irele(wip);
7dcf5c3eSDave Chinner	return error;
f6bba201SDave Chinner
445883e8SDave Chinnerout_trans_cancel:
4906e215SChristoph Hellwig	xfs_trans_cancel(tp);
253f4911SChristoph Hellwigout_release_wip:
7dcf5c3eSDave Chinner	if (wip)
44a8736bSDarrick J. Wong		xfs_irele(wip);
41667260SDarrick J. Wong	if (error == -ENOSPC && nospace_error)
41667260SDarrick J. Wong		error = nospace_error;
f6bba201SDave Chinner	return error;
f6bba201SDave Chinner}
f6bba201SDave Chinner
e6187b34SDave Chinnerstatic int
e6187b34SDave Chinnerxfs_iflush(
93848a99SChristoph Hellwig	struct xfs_inode	*ip,
93848a99SChristoph Hellwig	struct xfs_buf		*bp)
1da177e4SLinus Torvalds{
93848a99SChristoph Hellwig	struct xfs_inode_log_item *iip = ip->i_itemp;
93848a99SChristoph Hellwig	struct xfs_dinode	*dip;
93848a99SChristoph Hellwig	struct xfs_mount	*mp = ip->i_mount;
f2019299SBrian Foster	int			error;
1da177e4SLinus Torvalds
579aa9caSChristoph Hellwig	ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL|XFS_ILOCK_SHARED));
718ecc50SDave Chinner	ASSERT(xfs_iflags_test(ip, XFS_IFLUSHING));
f7e67b20SChristoph Hellwig	ASSERT(ip->i_df.if_format != XFS_DINODE_FMT_BTREE ||
daf83964SChristoph Hellwig	       ip->i_df.if_nextents > XFS_IFORK_MAXEXT(ip, XFS_DATA_FORK));
90c60e16SDave Chinner	ASSERT(iip->ili_item.li_buf == bp);
1da177e4SLinus Torvalds
88ee2df7SChristoph Hellwig	dip = xfs_buf_offset(bp, ip->i_imap.im_boffset);
1da177e4SLinus Torvalds
f2019299SBrian Foster	/*
f2019299SBrian Foster	 * We don't flush the inode if any of the following checks fail, but we
f2019299SBrian Foster	 * do still update the log item and attach to the backing buffer as if
f2019299SBrian Foster	 * the flush happened. This is a formality to facilitate predictable
f2019299SBrian Foster	 * error handling as the caller will shutdown and fail the buffer.
f2019299SBrian Foster	 */
f2019299SBrian Foster	error = -EFSCORRUPTED;
69ef921bSChristoph Hellwig	if (XFS_TEST_ERROR(dip->di_magic != cpu_to_be16(XFS_DINODE_MAGIC),
9e24cfd0SDarrick J. Wong			       mp, XFS_ERRTAG_IFLUSH_1)) {
6a19d939SDave Chinner		xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
78b0f58bSZeng Heng			"%s: Bad inode %llu magic number 0x%x, ptr "PTR_FMT,
6a19d939SDave Chinner			__func__, ip->i_ino, be16_to_cpu(dip->di_magic), dip);
f2019299SBrian Foster		goto flush_out;
1da177e4SLinus Torvalds	}
c19b3b05SDave Chinner	if (S_ISREG(VFS_I(ip)->i_mode)) {
1da177e4SLinus Torvalds		if (XFS_TEST_ERROR(
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_EXTENTS &&
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_BTREE,
9e24cfd0SDarrick J. Wong		    mp, XFS_ERRTAG_IFLUSH_3)) {
6a19d939SDave Chinner			xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
78b0f58bSZeng Heng				"%s: Bad regular inode %llu, ptr "PTR_FMT,
6a19d939SDave Chinner				__func__, ip->i_ino, ip);
f2019299SBrian Foster			goto flush_out;
1da177e4SLinus Torvalds		}
c19b3b05SDave Chinner	} else if (S_ISDIR(VFS_I(ip)->i_mode)) {
1da177e4SLinus Torvalds		if (XFS_TEST_ERROR(
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_EXTENTS &&
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_BTREE &&
f7e67b20SChristoph Hellwig		    ip->i_df.if_format != XFS_DINODE_FMT_LOCAL,
9e24cfd0SDarrick J. Wong		    mp, XFS_ERRTAG_IFLUSH_4)) {
6a19d939SDave Chinner			xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
78b0f58bSZeng Heng				"%s: Bad directory inode %llu, ptr "PTR_FMT,
6a19d939SDave Chinner				__func__, ip->i_ino, ip);
f2019299SBrian Foster			goto flush_out;
1da177e4SLinus Torvalds		}
1da177e4SLinus Torvalds	}
2ed5b09bSDarrick J. Wong	if (XFS_TEST_ERROR(ip->i_df.if_nextents + xfs_ifork_nextents(&ip->i_af) >
6e73a545SChristoph Hellwig				ip->i_nblocks, mp, XFS_ERRTAG_IFLUSH_5)) {
6a19d939SDave Chinner		xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
755c38ffSChandan Babu R			"%s: detected corrupt incore inode %llu, "
755c38ffSChandan Babu R			"total extents = %llu nblocks = %lld, ptr "PTR_FMT,
6a19d939SDave Chinner			__func__, ip->i_ino,
2ed5b09bSDarrick J. Wong			ip->i_df.if_nextents + xfs_ifork_nextents(&ip->i_af),
6e73a545SChristoph Hellwig			ip->i_nblocks, ip);
f2019299SBrian Foster		goto flush_out;
1da177e4SLinus Torvalds	}
7821ea30SChristoph Hellwig	if (XFS_TEST_ERROR(ip->i_forkoff > mp->m_sb.sb_inodesize,
9e24cfd0SDarrick J. Wong				mp, XFS_ERRTAG_IFLUSH_6)) {
6a19d939SDave Chinner		xfs_alert_tag(mp, XFS_PTAG_IFLUSH,
78b0f58bSZeng Heng			"%s: bad inode %llu, forkoff 0x%x, ptr "PTR_FMT,
7821ea30SChristoph Hellwig			__func__, ip->i_ino, ip->i_forkoff, ip);
f2019299SBrian Foster		goto flush_out;
1da177e4SLinus Torvalds	}
e60896d8SDave Chinner
1da177e4SLinus Torvalds	/*
965e0a1aSChristoph Hellwig	 * Inode item log recovery for v2 inodes are dependent on the flushiter
965e0a1aSChristoph Hellwig	 * count for correct sequencing.  We bump the flush iteration count so
965e0a1aSChristoph Hellwig	 * we can detect flushes which postdate a log record during recovery.
965e0a1aSChristoph Hellwig	 * This is redundant as we now log every change and hence this can't
965e0a1aSChristoph Hellwig	 * happen but we need to still do it to ensure backwards compatibility
965e0a1aSChristoph Hellwig	 * with old kernels that predate logging all inode changes.
1da177e4SLinus Torvalds	 */
38c26bfdSDave Chinner	if (!xfs_has_v3inodes(mp))
965e0a1aSChristoph Hellwig		ip->i_flushiter++;
1da177e4SLinus Torvalds
0f45a1b2SChristoph Hellwig	/*
0f45a1b2SChristoph Hellwig	 * If there are inline format data / attr forks attached to this inode,
0f45a1b2SChristoph Hellwig	 * make sure they are not corrupt.
0f45a1b2SChristoph Hellwig	 */
f7e67b20SChristoph Hellwig	if (ip->i_df.if_format == XFS_DINODE_FMT_LOCAL &&
0f45a1b2SChristoph Hellwig	    xfs_ifork_verify_local_data(ip))
0f45a1b2SChristoph Hellwig		goto flush_out;
932b42c6SDarrick J. Wong	if (xfs_inode_has_attr_fork(ip) &&
2ed5b09bSDarrick J. Wong	    ip->i_af.if_format == XFS_DINODE_FMT_LOCAL &&
0f45a1b2SChristoph Hellwig	    xfs_ifork_verify_local_attr(ip))
f2019299SBrian Foster		goto flush_out;
005c5db8SDarrick J. Wong
1da177e4SLinus Torvalds	/*
3987848cSDave Chinner	 * Copy the dirty parts of the inode into the on-disk inode.  We always
3987848cSDave Chinner	 * copy out the core of the inode, because if the inode is dirty at all
3987848cSDave Chinner	 * the core must be.
1da177e4SLinus Torvalds	 */
93f958f9SDave Chinner	xfs_inode_to_disk(ip, dip, iip->ili_item.li_lsn);
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/* Wrap, we never let the log put out DI_MAX_FLUSH */
38c26bfdSDave Chinner	if (!xfs_has_v3inodes(mp)) {
965e0a1aSChristoph Hellwig		if (ip->i_flushiter == DI_MAX_FLUSH)
965e0a1aSChristoph Hellwig			ip->i_flushiter = 0;
ee7b83fdSChristoph Hellwig	}
1da177e4SLinus Torvalds
005c5db8SDarrick J. Wong	xfs_iflush_fork(ip, dip, iip, XFS_DATA_FORK);
932b42c6SDarrick J. Wong	if (xfs_inode_has_attr_fork(ip))
005c5db8SDarrick J. Wong		xfs_iflush_fork(ip, dip, iip, XFS_ATTR_FORK);
1da177e4SLinus Torvalds
1da177e4SLinus Torvalds	/*
f5d8d5c4SChristoph Hellwig	 * We've recorded everything logged in the inode, so we'd like to clear
f5d8d5c4SChristoph Hellwig	 * the ili_fields bits so we don't log and flush things unnecessarily.
f5d8d5c4SChristoph Hellwig	 * However, we can't stop logging all this information until the data
f5d8d5c4SChristoph Hellwig	 * we've copied into the disk buffer is written to disk.  If we did we
f5d8d5c4SChristoph Hellwig	 * might overwrite the copy of the inode in the log with all the data
f5d8d5c4SChristoph Hellwig	 * after re-logging only part of it, and in the face of a crash we
f5d8d5c4SChristoph Hellwig	 * wouldn't have all the data we need to recover.
1da177e4SLinus Torvalds	 *
f5d8d5c4SChristoph Hellwig	 * What we do is move the bits to the ili_last_fields field.  When
f5d8d5c4SChristoph Hellwig	 * logging the inode, these bits are moved back to the ili_fields field.
664ffb8aSChristoph Hellwig	 * In the xfs_buf_inode_iodone() routine we clear ili_last_fields, since
664ffb8aSChristoph Hellwig	 * we know that the information those bits represent is permanently on
f5d8d5c4SChristoph Hellwig	 * disk.  As long as the flush completes before the inode is logged
f5d8d5c4SChristoph Hellwig	 * again, then both ili_fields and ili_last_fields will be cleared.
1da177e4SLinus Torvalds	 */
f2019299SBrian Foster	error = 0;
f2019299SBrian Fosterflush_out:
1319ebefSDave Chinner	spin_lock(&iip->ili_lock);
f5d8d5c4SChristoph Hellwig	iip->ili_last_fields = iip->ili_fields;
f5d8d5c4SChristoph Hellwig	iip->ili_fields = 0;
fc0561ceSDave Chinner	iip->ili_fsync_fields = 0;
1319ebefSDave Chinner	spin_unlock(&iip->ili_lock);
1da177e4SLinus Torvalds
1319ebefSDave Chinner	/*
1319ebefSDave Chinner	 * Store the current LSN of the inode so that we can tell whether the
664ffb8aSChristoph Hellwig	 * item has moved in the AIL from xfs_buf_inode_iodone().
1319ebefSDave Chinner	 */
7b2e2a31SDavid Chinner	xfs_trans_ail_copy_lsn(mp->m_ail, &iip->ili_flush_lsn,
7b2e2a31SDavid Chinner				&iip->ili_item.li_lsn);
1da177e4SLinus Torvalds
93848a99SChristoph Hellwig	/* generate the checksum. */
93848a99SChristoph Hellwig	xfs_dinode_calc_crc(mp, dip);
f2019299SBrian Foster	return error;
1da177e4SLinus Torvalds}
44a8736bSDarrick J. Wong
e6187b34SDave Chinner/*
e6187b34SDave Chinner * Non-blocking flush of dirty inode metadata into the backing buffer.
e6187b34SDave Chinner *
e6187b34SDave Chinner * The caller must have a reference to the inode and hold the cluster buffer
e6187b34SDave Chinner * locked. The function will walk across all the inodes on the cluster buffer it
e6187b34SDave Chinner * can find and lock without blocking, and flush them to the cluster buffer.
e6187b34SDave Chinner *
5717ea4dSDave Chinner * On successful flushing of at least one inode, the caller must write out the
5717ea4dSDave Chinner * buffer and release it. If no inodes are flushed, -EAGAIN will be returned and
5717ea4dSDave Chinner * the caller needs to release the buffer. On failure, the filesystem will be
5717ea4dSDave Chinner * shut down, the buffer will have been unlocked and released, and EFSCORRUPTED
5717ea4dSDave Chinner * will be returned.
e6187b34SDave Chinner */
e6187b34SDave Chinnerint
e6187b34SDave Chinnerxfs_iflush_cluster(
e6187b34SDave Chinner	struct xfs_buf		*bp)
e6187b34SDave Chinner{
5717ea4dSDave Chinner	struct xfs_mount	*mp = bp->b_mount;
5717ea4dSDave Chinner	struct xfs_log_item	*lip, *n;
5717ea4dSDave Chinner	struct xfs_inode	*ip;
5717ea4dSDave Chinner	struct xfs_inode_log_item *iip;
e6187b34SDave Chinner	int			clcount = 0;
5717ea4dSDave Chinner	int			error = 0;
e6187b34SDave Chinner
e6187b34SDave Chinner	/*
5717ea4dSDave Chinner	 * We must use the safe variant here as on shutdown xfs_iflush_abort()
d2d7c047SDave Chinner	 * will remove itself from the list.
e6187b34SDave Chinner	 */
5717ea4dSDave Chinner	list_for_each_entry_safe(lip, n, &bp->b_li_list, li_bio_list) {
5717ea4dSDave Chinner		iip = (struct xfs_inode_log_item *)lip;
5717ea4dSDave Chinner		ip = iip->ili_inode;
5717ea4dSDave Chinner
5717ea4dSDave Chinner		/*
5717ea4dSDave Chinner		 * Quick and dirty check to avoid locks if possible.
5717ea4dSDave Chinner		 */
718ecc50SDave Chinner		if (__xfs_iflags_test(ip, XFS_IRECLAIM | XFS_IFLUSHING))
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner		if (xfs_ipincount(ip))
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner
5717ea4dSDave Chinner		/*
5717ea4dSDave Chinner		 * The inode is still attached to the buffer, which means it is
5717ea4dSDave Chinner		 * dirty but reclaim might try to grab it. Check carefully for
5717ea4dSDave Chinner		 * that, and grab the ilock while still holding the i_flags_lock
5717ea4dSDave Chinner		 * to guarantee reclaim will not be able to reclaim this inode
5717ea4dSDave Chinner		 * once we drop the i_flags_lock.
5717ea4dSDave Chinner		 */
5717ea4dSDave Chinner		spin_lock(&ip->i_flags_lock);
5717ea4dSDave Chinner		ASSERT(!__xfs_iflags_test(ip, XFS_ISTALE));
718ecc50SDave Chinner		if (__xfs_iflags_test(ip, XFS_IRECLAIM | XFS_IFLUSHING)) {
5717ea4dSDave Chinner			spin_unlock(&ip->i_flags_lock);
e6187b34SDave Chinner			continue;
e6187b34SDave Chinner		}
e6187b34SDave Chinner
e6187b34SDave Chinner		/*
5717ea4dSDave Chinner		 * ILOCK will pin the inode against reclaim and prevent
5717ea4dSDave Chinner		 * concurrent transactions modifying the inode while we are
718ecc50SDave Chinner		 * flushing the inode. If we get the lock, set the flushing
718ecc50SDave Chinner		 * state before we drop the i_flags_lock.
e6187b34SDave Chinner		 */
5717ea4dSDave Chinner		if (!xfs_ilock_nowait(ip, XFS_ILOCK_SHARED)) {
5717ea4dSDave Chinner			spin_unlock(&ip->i_flags_lock);
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner		}
718ecc50SDave Chinner		__xfs_iflags_set(ip, XFS_IFLUSHING);
5717ea4dSDave Chinner		spin_unlock(&ip->i_flags_lock);
5717ea4dSDave Chinner
5717ea4dSDave Chinner		/*
5717ea4dSDave Chinner		 * Abort flushing this inode if we are shut down because the
5717ea4dSDave Chinner		 * inode may not currently be in the AIL. This can occur when
5717ea4dSDave Chinner		 * log I/O failure unpins the inode without inserting into the
5717ea4dSDave Chinner		 * AIL, leaving a dirty/unpinned inode attached to the buffer
5717ea4dSDave Chinner		 * that otherwise looks like it should be flushed.
5717ea4dSDave Chinner		 */
01728b44SDave Chinner		if (xlog_is_shutdown(mp->m_log)) {
5717ea4dSDave Chinner			xfs_iunpin_wait(ip);
5717ea4dSDave Chinner			xfs_iflush_abort(ip);
5717ea4dSDave Chinner			xfs_iunlock(ip, XFS_ILOCK_SHARED);
5717ea4dSDave Chinner			error = -EIO;
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner		}
5717ea4dSDave Chinner
5717ea4dSDave Chinner		/* don't block waiting on a log force to unpin dirty inodes */
5717ea4dSDave Chinner		if (xfs_ipincount(ip)) {
718ecc50SDave Chinner			xfs_iflags_clear(ip, XFS_IFLUSHING);
5717ea4dSDave Chinner			xfs_iunlock(ip, XFS_ILOCK_SHARED);
5717ea4dSDave Chinner			continue;
5717ea4dSDave Chinner		}
5717ea4dSDave Chinner
5717ea4dSDave Chinner		if (!xfs_inode_clean(ip))
5717ea4dSDave Chinner			error = xfs_iflush(ip, bp);
5717ea4dSDave Chinner		else
718ecc50SDave Chinner			xfs_iflags_clear(ip, XFS_IFLUSHING);
5717ea4dSDave Chinner		xfs_iunlock(ip, XFS_ILOCK_SHARED);
5717ea4dSDave Chinner		if (error)
e6187b34SDave Chinner			break;
e6187b34SDave Chinner		clcount++;
e6187b34SDave Chinner	}
e6187b34SDave Chinner
e6187b34SDave Chinner	if (error) {
01728b44SDave Chinner		/*
01728b44SDave Chinner		 * Shutdown first so we kill the log before we release this
01728b44SDave Chinner		 * buffer. If it is an INODE_ALLOC buffer and pins the tail
01728b44SDave Chinner		 * of the log, failing it before the _log_ is shut down can
01728b44SDave Chinner		 * result in the log tail being moved forward in the journal
01728b44SDave Chinner		 * on disk because log writes can still be taking place. Hence
01728b44SDave Chinner		 * unpinning the tail will allow the ICREATE intent to be
01728b44SDave Chinner		 * removed from the log an recovery will fail with uninitialised
01728b44SDave Chinner		 * inode cluster buffers.
01728b44SDave Chinner		 */
01728b44SDave Chinner		xfs_force_shutdown(mp, SHUTDOWN_CORRUPT_INCORE);
e6187b34SDave Chinner		bp->b_flags |= XBF_ASYNC;
e6187b34SDave Chinner		xfs_buf_ioend_fail(bp);
e6187b34SDave Chinner		return error;
e6187b34SDave Chinner	}
e6187b34SDave Chinner
5717ea4dSDave Chinner	if (!clcount)
5717ea4dSDave Chinner		return -EAGAIN;
5717ea4dSDave Chinner
5717ea4dSDave Chinner	XFS_STATS_INC(mp, xs_icluster_flushcnt);
5717ea4dSDave Chinner	XFS_STATS_ADD(mp, xs_icluster_flushinode, clcount);
5717ea4dSDave Chinner	return 0;
5717ea4dSDave Chinner
5717ea4dSDave Chinner}
5717ea4dSDave Chinner
44a8736bSDarrick J. Wong/* Release an inode. */
44a8736bSDarrick J. Wongvoid
44a8736bSDarrick J. Wongxfs_irele(
44a8736bSDarrick J. Wong	struct xfs_inode	*ip)
44a8736bSDarrick J. Wong{
44a8736bSDarrick J. Wong	trace_xfs_irele(ip, _RET_IP_);
44a8736bSDarrick J. Wong	iput(VFS_I(ip));
44a8736bSDarrick J. Wong}
54fbdd10SChristoph Hellwig
54fbdd10SChristoph Hellwig/*
54fbdd10SChristoph Hellwig * Ensure all commited transactions touching the inode are written to the log.
54fbdd10SChristoph Hellwig */
54fbdd10SChristoph Hellwigint
54fbdd10SChristoph Hellwigxfs_log_force_inode(
54fbdd10SChristoph Hellwig	struct xfs_inode	*ip)
54fbdd10SChristoph Hellwig{
5f9b4b0dSDave Chinner	xfs_csn_t		seq = 0;
54fbdd10SChristoph Hellwig
54fbdd10SChristoph Hellwig	xfs_ilock(ip, XFS_ILOCK_SHARED);
54fbdd10SChristoph Hellwig	if (xfs_ipincount(ip))
5f9b4b0dSDave Chinner		seq = ip->i_itemp->ili_commit_seq;
54fbdd10SChristoph Hellwig	xfs_iunlock(ip, XFS_ILOCK_SHARED);
54fbdd10SChristoph Hellwig
5f9b4b0dSDave Chinner	if (!seq)
54fbdd10SChristoph Hellwig		return 0;
5f9b4b0dSDave Chinner	return xfs_log_force_seq(ip->i_mount, seq, XFS_LOG_SYNC, NULL);
54fbdd10SChristoph Hellwig}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong/*
e2aaee9cSDarrick J. Wong * Grab the exclusive iolock for a data copy from src to dest, making sure to
e2aaee9cSDarrick J. Wong * abide vfs locking order (lowest pointer value goes first) and breaking the
e2aaee9cSDarrick J. Wong * layout leases before proceeding.  The loop is needed because we cannot call
e2aaee9cSDarrick J. Wong * the blocking break_layout() with the iolocks held, and therefore have to
e2aaee9cSDarrick J. Wong * back out both locks.
e2aaee9cSDarrick J. Wong */
e2aaee9cSDarrick J. Wongstatic int
e2aaee9cSDarrick J. Wongxfs_iolock_two_inodes_and_break_layout(
e2aaee9cSDarrick J. Wong	struct inode		*src,
e2aaee9cSDarrick J. Wong	struct inode		*dest)
e2aaee9cSDarrick J. Wong{
e2aaee9cSDarrick J. Wong	int			error;
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	if (src > dest)
e2aaee9cSDarrick J. Wong		swap(src, dest);
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wongretry:
e2aaee9cSDarrick J. Wong	/* Wait to break both inodes' layouts before we start locking. */
e2aaee9cSDarrick J. Wong	error = break_layout(src, true);
e2aaee9cSDarrick J. Wong	if (error)
e2aaee9cSDarrick J. Wong		return error;
e2aaee9cSDarrick J. Wong	if (src != dest) {
e2aaee9cSDarrick J. Wong		error = break_layout(dest, true);
e2aaee9cSDarrick J. Wong		if (error)
e2aaee9cSDarrick J. Wong			return error;
e2aaee9cSDarrick J. Wong	}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	/* Lock one inode and make sure nobody got in and leased it. */
e2aaee9cSDarrick J. Wong	inode_lock(src);
e2aaee9cSDarrick J. Wong	error = break_layout(src, false);
e2aaee9cSDarrick J. Wong	if (error) {
e2aaee9cSDarrick J. Wong		inode_unlock(src);
e2aaee9cSDarrick J. Wong		if (error == -EWOULDBLOCK)
e2aaee9cSDarrick J. Wong			goto retry;
e2aaee9cSDarrick J. Wong		return error;
e2aaee9cSDarrick J. Wong	}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	if (src == dest)
e2aaee9cSDarrick J. Wong		return 0;
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	/* Lock the other inode and make sure nobody got in and leased it. */
e2aaee9cSDarrick J. Wong	inode_lock_nested(dest, I_MUTEX_NONDIR2);
e2aaee9cSDarrick J. Wong	error = break_layout(dest, false);
e2aaee9cSDarrick J. Wong	if (error) {
e2aaee9cSDarrick J. Wong		inode_unlock(src);
e2aaee9cSDarrick J. Wong		inode_unlock(dest);
e2aaee9cSDarrick J. Wong		if (error == -EWOULDBLOCK)
e2aaee9cSDarrick J. Wong			goto retry;
e2aaee9cSDarrick J. Wong		return error;
e2aaee9cSDarrick J. Wong	}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	return 0;
e2aaee9cSDarrick J. Wong}
e2aaee9cSDarrick J. Wong
13f9e267SShiyang Ruanstatic int
13f9e267SShiyang Ruanxfs_mmaplock_two_inodes_and_break_dax_layout(
13f9e267SShiyang Ruan	struct xfs_inode	*ip1,
13f9e267SShiyang Ruan	struct xfs_inode	*ip2)
13f9e267SShiyang Ruan{
13f9e267SShiyang Ruan	int			error;
13f9e267SShiyang Ruan	bool			retry;
13f9e267SShiyang Ruan	struct page		*page;
13f9e267SShiyang Ruan
13f9e267SShiyang Ruan	if (ip1->i_ino > ip2->i_ino)
13f9e267SShiyang Ruan		swap(ip1, ip2);
13f9e267SShiyang Ruan
13f9e267SShiyang Ruanagain:
13f9e267SShiyang Ruan	retry = false;
13f9e267SShiyang Ruan	/* Lock the first inode */
13f9e267SShiyang Ruan	xfs_ilock(ip1, XFS_MMAPLOCK_EXCL);
13f9e267SShiyang Ruan	error = xfs_break_dax_layouts(VFS_I(ip1), &retry);
13f9e267SShiyang Ruan	if (error || retry) {
13f9e267SShiyang Ruan		xfs_iunlock(ip1, XFS_MMAPLOCK_EXCL);
13f9e267SShiyang Ruan		if (error == 0 && retry)
13f9e267SShiyang Ruan			goto again;
13f9e267SShiyang Ruan		return error;
13f9e267SShiyang Ruan	}
13f9e267SShiyang Ruan
13f9e267SShiyang Ruan	if (ip1 == ip2)
13f9e267SShiyang Ruan		return 0;
13f9e267SShiyang Ruan
13f9e267SShiyang Ruan	/* Nested lock the second inode */
13f9e267SShiyang Ruan	xfs_ilock(ip2, xfs_lock_inumorder(XFS_MMAPLOCK_EXCL, 1));
13f9e267SShiyang Ruan	/*
13f9e267SShiyang Ruan	 * We cannot use xfs_break_dax_layouts() directly here because it may
13f9e267SShiyang Ruan	 * need to unlock & lock the XFS_MMAPLOCK_EXCL which is not suitable
13f9e267SShiyang Ruan	 * for this nested lock case.
13f9e267SShiyang Ruan	 */
13f9e267SShiyang Ruan	page = dax_layout_busy_page(VFS_I(ip2)->i_mapping);
13f9e267SShiyang Ruan	if (page && page_ref_count(page) != 1) {
13f9e267SShiyang Ruan		xfs_iunlock(ip2, XFS_MMAPLOCK_EXCL);
13f9e267SShiyang Ruan		xfs_iunlock(ip1, XFS_MMAPLOCK_EXCL);
13f9e267SShiyang Ruan		goto again;
13f9e267SShiyang Ruan	}
13f9e267SShiyang Ruan
13f9e267SShiyang Ruan	return 0;
13f9e267SShiyang Ruan}
13f9e267SShiyang Ruan
e2aaee9cSDarrick J. Wong/*
e2aaee9cSDarrick J. Wong * Lock two inodes so that userspace cannot initiate I/O via file syscalls or
e2aaee9cSDarrick J. Wong * mmap activity.
e2aaee9cSDarrick J. Wong */
e2aaee9cSDarrick J. Wongint
e2aaee9cSDarrick J. Wongxfs_ilock2_io_mmap(
e2aaee9cSDarrick J. Wong	struct xfs_inode	*ip1,
e2aaee9cSDarrick J. Wong	struct xfs_inode	*ip2)
e2aaee9cSDarrick J. Wong{
e2aaee9cSDarrick J. Wong	int			ret;
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong	ret = xfs_iolock_two_inodes_and_break_layout(VFS_I(ip1), VFS_I(ip2));
e2aaee9cSDarrick J. Wong	if (ret)
e2aaee9cSDarrick J. Wong		return ret;
13f9e267SShiyang Ruan
13f9e267SShiyang Ruan	if (IS_DAX(VFS_I(ip1)) && IS_DAX(VFS_I(ip2))) {
13f9e267SShiyang Ruan		ret = xfs_mmaplock_two_inodes_and_break_dax_layout(ip1, ip2);
13f9e267SShiyang Ruan		if (ret) {
13f9e267SShiyang Ruan			inode_unlock(VFS_I(ip2));
13f9e267SShiyang Ruan			if (ip1 != ip2)
13f9e267SShiyang Ruan				inode_unlock(VFS_I(ip1));
13f9e267SShiyang Ruan			return ret;
13f9e267SShiyang Ruan		}
13f9e267SShiyang Ruan	} else
d2c292d8SJan Kara		filemap_invalidate_lock_two(VFS_I(ip1)->i_mapping,
d2c292d8SJan Kara					    VFS_I(ip2)->i_mapping);
13f9e267SShiyang Ruan
e2aaee9cSDarrick J. Wong	return 0;
e2aaee9cSDarrick J. Wong}
e2aaee9cSDarrick J. Wong
e2aaee9cSDarrick J. Wong/* Unlock both inodes to allow IO and mmap activity. */
e2aaee9cSDarrick J. Wongvoid
e2aaee9cSDarrick J. Wongxfs_iunlock2_io_mmap(
e2aaee9cSDarrick J. Wong	struct xfs_inode	*ip1,
e2aaee9cSDarrick J. Wong	struct xfs_inode	*ip2)
e2aaee9cSDarrick J. Wong{
13f9e267SShiyang Ruan	if (IS_DAX(VFS_I(ip1)) && IS_DAX(VFS_I(ip2))) {
13f9e267SShiyang Ruan		xfs_iunlock(ip2, XFS_MMAPLOCK_EXCL);
13f9e267SShiyang Ruan		if (ip1 != ip2)
13f9e267SShiyang Ruan			xfs_iunlock(ip1, XFS_MMAPLOCK_EXCL);
13f9e267SShiyang Ruan	} else
d2c292d8SJan Kara		filemap_invalidate_unlock_two(VFS_I(ip1)->i_mapping,
d2c292d8SJan Kara					      VFS_I(ip2)->i_mapping);
13f9e267SShiyang Ruan
e2aaee9cSDarrick J. Wong	inode_unlock(VFS_I(ip2));
d2c292d8SJan Kara	if (ip1 != ip2)
e2aaee9cSDarrick J. Wong		inode_unlock(VFS_I(ip1));
e2aaee9cSDarrick J. Wong}
83771c50SDarrick J. Wong
d7d84772SCatherine Hoang/* Drop the MMAPLOCK and the IOLOCK after a remap completes. */
d7d84772SCatherine Hoangvoid
d7d84772SCatherine Hoangxfs_iunlock2_remapping(
d7d84772SCatherine Hoang	struct xfs_inode	*ip1,
d7d84772SCatherine Hoang	struct xfs_inode	*ip2)
d7d84772SCatherine Hoang{
d7d84772SCatherine Hoang	xfs_iflags_clear(ip1, XFS_IREMAPPING);
d7d84772SCatherine Hoang
d7d84772SCatherine Hoang	if (ip1 != ip2)
d7d84772SCatherine Hoang		xfs_iunlock(ip1, XFS_MMAPLOCK_SHARED);
d7d84772SCatherine Hoang	xfs_iunlock(ip2, XFS_MMAPLOCK_EXCL);
d7d84772SCatherine Hoang
d7d84772SCatherine Hoang	if (ip1 != ip2)
d7d84772SCatherine Hoang		inode_unlock_shared(VFS_I(ip1));
d7d84772SCatherine Hoang	inode_unlock(VFS_I(ip2));
d7d84772SCatherine Hoang}
d7d84772SCatherine Hoang
83771c50SDarrick J. Wong/*
83771c50SDarrick J. Wong * Reload the incore inode list for this inode.  Caller should ensure that
83771c50SDarrick J. Wong * the link count cannot change, either by taking ILOCK_SHARED or otherwise
83771c50SDarrick J. Wong * preventing other threads from executing.
83771c50SDarrick J. Wong */
83771c50SDarrick J. Wongint
83771c50SDarrick J. Wongxfs_inode_reload_unlinked_bucket(
83771c50SDarrick J. Wong	struct xfs_trans	*tp,
83771c50SDarrick J. Wong	struct xfs_inode	*ip)
83771c50SDarrick J. Wong{
83771c50SDarrick J. Wong	struct xfs_mount	*mp = tp->t_mountp;
83771c50SDarrick J. Wong	struct xfs_buf		*agibp;
83771c50SDarrick J. Wong	struct xfs_agi		*agi;
83771c50SDarrick J. Wong	struct xfs_perag	*pag;
83771c50SDarrick J. Wong	xfs_agnumber_t		agno = XFS_INO_TO_AGNO(mp, ip->i_ino);
83771c50SDarrick J. Wong	xfs_agino_t		agino = XFS_INO_TO_AGINO(mp, ip->i_ino);
83771c50SDarrick J. Wong	xfs_agino_t		prev_agino, next_agino;
83771c50SDarrick J. Wong	unsigned int		bucket;
83771c50SDarrick J. Wong	bool			foundit = false;
83771c50SDarrick J. Wong	int			error;
83771c50SDarrick J. Wong
83771c50SDarrick J. Wong	/* Grab the first inode in the list */
83771c50SDarrick J. Wong	pag = xfs_perag_get(mp, agno);
83771c50SDarrick J. Wong	error = xfs_ialloc_read_agi(pag, tp, &agibp);
83771c50SDarrick J. Wong	xfs_perag_put(pag);
83771c50SDarrick J. Wong	if (error)
83771c50SDarrick J. Wong		return error;
83771c50SDarrick J. Wong
537c013bSDarrick J. Wong	/*
537c013bSDarrick J. Wong	 * We've taken ILOCK_SHARED and the AGI buffer lock to stabilize the
537c013bSDarrick J. Wong	 * incore unlinked list pointers for this inode.  Check once more to
537c013bSDarrick J. Wong	 * see if we raced with anyone else to reload the unlinked list.
537c013bSDarrick J. Wong	 */
537c013bSDarrick J. Wong	if (!xfs_inode_unlinked_incomplete(ip)) {
537c013bSDarrick J. Wong		foundit = true;
537c013bSDarrick J. Wong		goto out_agibp;
537c013bSDarrick J. Wong	}
537c013bSDarrick J. Wong
83771c50SDarrick J. Wong	bucket = agino % XFS_AGI_UNLINKED_BUCKETS;
83771c50SDarrick J. Wong	agi = agibp->b_addr;
83771c50SDarrick J. Wong
83771c50SDarrick J. Wong	trace_xfs_inode_reload_unlinked_bucket(ip);
83771c50SDarrick J. Wong
83771c50SDarrick J. Wong	xfs_info_ratelimited(mp,
83771c50SDarrick J. Wong "Found unrecovered unlinked inode 0x%x in AG 0x%x.  Initiating list recovery.",
83771c50SDarrick J. Wong			agino, agno);
83771c50SDarrick J. Wong
83771c50SDarrick J. Wong	prev_agino = NULLAGINO;
83771c50SDarrick J. Wong	next_agino = be32_to_cpu(agi->agi_unlinked[bucket]);
83771c50SDarrick J. Wong	while (next_agino != NULLAGINO) {
83771c50SDarrick J. Wong		struct xfs_inode	*next_ip = NULL;
83771c50SDarrick J. Wong
537c013bSDarrick J. Wong		/* Found this caller's inode, set its backlink. */
83771c50SDarrick J. Wong		if (next_agino == agino) {
83771c50SDarrick J. Wong			next_ip = ip;
83771c50SDarrick J. Wong			next_ip->i_prev_unlinked = prev_agino;
83771c50SDarrick J. Wong			foundit = true;
537c013bSDarrick J. Wong			goto next_inode;
83771c50SDarrick J. Wong		}
537c013bSDarrick J. Wong
537c013bSDarrick J. Wong		/* Try in-memory lookup first. */
83771c50SDarrick J. Wong		next_ip = xfs_iunlink_lookup(pag, next_agino);
537c013bSDarrick J. Wong		if (next_ip)
537c013bSDarrick J. Wong			goto next_inode;
537c013bSDarrick J. Wong
537c013bSDarrick J. Wong		/* Inode not in memory, try reloading it. */
83771c50SDarrick J. Wong		error = xfs_iunlink_reload_next(tp, agibp, prev_agino,
83771c50SDarrick J. Wong				next_agino);
83771c50SDarrick J. Wong		if (error)
83771c50SDarrick J. Wong			break;
83771c50SDarrick J. Wong
537c013bSDarrick J. Wong		/* Grab the reloaded inode. */
83771c50SDarrick J. Wong		next_ip = xfs_iunlink_lookup(pag, next_agino);
83771c50SDarrick J. Wong		if (!next_ip) {
83771c50SDarrick J. Wong			/* No incore inode at all?  We reloaded it... */
83771c50SDarrick J. Wong			ASSERT(next_ip != NULL);
83771c50SDarrick J. Wong			error = -EFSCORRUPTED;
83771c50SDarrick J. Wong			break;
83771c50SDarrick J. Wong		}
83771c50SDarrick J. Wong
537c013bSDarrick J. Wongnext_inode:
83771c50SDarrick J. Wong		prev_agino = next_agino;
83771c50SDarrick J. Wong		next_agino = next_ip->i_next_unlinked;
83771c50SDarrick J. Wong	}
83771c50SDarrick J. Wong
537c013bSDarrick J. Wongout_agibp:
83771c50SDarrick J. Wong	xfs_trans_brelse(tp, agibp);
83771c50SDarrick J. Wong	/* Should have found this inode somewhere in the iunlinked bucket. */
83771c50SDarrick J. Wong	if (!error && !foundit)
83771c50SDarrick J. Wong		error = -EFSCORRUPTED;
83771c50SDarrick J. Wong	return error;
83771c50SDarrick J. Wong}
83771c50SDarrick J. Wong
83771c50SDarrick J. Wong/* Decide if this inode is missing its unlinked list and reload it. */
83771c50SDarrick J. Wongint
83771c50SDarrick J. Wongxfs_inode_reload_unlinked(
83771c50SDarrick J. Wong	struct xfs_inode	*ip)
83771c50SDarrick J. Wong{
83771c50SDarrick J. Wong	struct xfs_trans	*tp;
83771c50SDarrick J. Wong	int			error;
83771c50SDarrick J. Wong
83771c50SDarrick J. Wong	error = xfs_trans_alloc_empty(ip->i_mount, &tp);
83771c50SDarrick J. Wong	if (error)
83771c50SDarrick J. Wong		return error;
83771c50SDarrick J. Wong
83771c50SDarrick J. Wong	xfs_ilock(ip, XFS_ILOCK_SHARED);
83771c50SDarrick J. Wong	if (xfs_inode_unlinked_incomplete(ip))
83771c50SDarrick J. Wong		error = xfs_inode_reload_unlinked_bucket(tp, ip);
83771c50SDarrick J. Wong	xfs_iunlock(ip, XFS_ILOCK_SHARED);
83771c50SDarrick J. Wong	xfs_trans_cancel(tp);
83771c50SDarrick J. Wong
83771c50SDarrick J. Wong	return error;
83771c50SDarrick J. Wong}
c070b880SDarrick J. Wong
c070b880SDarrick J. Wong/* Returns the size of fundamental allocation unit for a file, in bytes. */
c070b880SDarrick J. Wongunsigned int
c070b880SDarrick J. Wongxfs_inode_alloc_unitsize(
c070b880SDarrick J. Wong	struct xfs_inode	*ip)
c070b880SDarrick J. Wong{
c070b880SDarrick J. Wong	unsigned int		blocks = 1;
c070b880SDarrick J. Wong
c070b880SDarrick J. Wong	if (XFS_IS_REALTIME_INODE(ip))
c070b880SDarrick J. Wong		blocks = ip->i_mount->m_sb.sb_rextsize;
c070b880SDarrick J. Wong
c070b880SDarrick J. Wong	return XFS_FSB_TO_B(ip->i_mount, blocks);
c070b880SDarrick J. Wong}