xref: /linux/fs/namei.c (revision 8189ab688f5a2c745dc90b63374da02f213d6994)
1 // SPDX-License-Identifier: GPL-2.0
2 /*
3  *  linux/fs/namei.c
4  *
5  *  Copyright (C) 1991, 1992  Linus Torvalds
6  */
7 
8 /*
9  * Some corrections by tytso.
10  */
11 
12 /* [Feb 1997 T. Schoebel-Theuer] Complete rewrite of the pathname
13  * lookup logic.
14  */
15 /* [Feb-Apr 2000, AV] Rewrite to the new namespace architecture.
16  */
17 
18 #include <linux/init.h>
19 #include <linux/export.h>
20 #include <linux/slab.h>
21 #include <linux/wordpart.h>
22 #include <linux/fs.h>
23 #include <linux/filelock.h>
24 #include <linux/namei.h>
25 #include <linux/pagemap.h>
26 #include <linux/sched/mm.h>
27 #include <linux/fsnotify.h>
28 #include <linux/personality.h>
29 #include <linux/security.h>
30 #include <linux/syscalls.h>
31 #include <linux/mount.h>
32 #include <linux/audit.h>
33 #include <linux/capability.h>
34 #include <linux/file.h>
35 #include <linux/fcntl.h>
36 #include <linux/device_cgroup.h>
37 #include <linux/fs_struct.h>
38 #include <linux/posix_acl.h>
39 #include <linux/hash.h>
40 #include <linux/bitops.h>
41 #include <linux/init_task.h>
42 #include <linux/uaccess.h>
43 
44 #include <asm/runtime-const.h>
45 
46 #include "internal.h"
47 #include "mount.h"
48 
49 /* [Feb-1997 T. Schoebel-Theuer]
50  * Fundamental changes in the pathname lookup mechanisms (namei)
51  * were necessary because of omirr.  The reason is that omirr needs
52  * to know the _real_ pathname, not the user-supplied one, in case
53  * of symlinks (and also when transname replacements occur).
54  *
55  * The new code replaces the old recursive symlink resolution with
56  * an iterative one (in case of non-nested symlink chains).  It does
57  * this with calls to <fs>_follow_link().
58  * As a side effect, dir_namei(), _namei() and follow_link() are now
59  * replaced with a single function lookup_dentry() that can handle all
60  * the special cases of the former code.
61  *
62  * With the new dcache, the pathname is stored at each inode, at least as
63  * long as the refcount of the inode is positive.  As a side effect, the
64  * size of the dcache depends on the inode cache and thus is dynamic.
65  *
66  * [29-Apr-1998 C. Scott Ananian] Updated above description of symlink
67  * resolution to correspond with current state of the code.
68  *
69  * Note that the symlink resolution is not *completely* iterative.
70  * There is still a significant amount of tail- and mid- recursion in
71  * the algorithm.  Also, note that <fs>_readlink() is not used in
72  * lookup_dentry(): lookup_dentry() on the result of <fs>_readlink()
73  * may return different results than <fs>_follow_link().  Many virtual
74  * filesystems (including /proc) exhibit this behavior.
75  */
76 
77 /* [24-Feb-97 T. Schoebel-Theuer] Side effects caused by new implementation:
78  * New symlink semantics: when open() is called with flags O_CREAT | O_EXCL
79  * and the name already exists in form of a symlink, try to create the new
80  * name indicated by the symlink. The old code always complained that the
81  * name already exists, due to not following the symlink even if its target
82  * is nonexistent.  The new semantics affects also mknod() and link() when
83  * the name is a symlink pointing to a non-existent name.
84  *
85  * I don't know which semantics is the right one, since I have no access
86  * to standards. But I found by trial that HP-UX 9.0 has the full "new"
87  * semantics implemented, while SunOS 4.1.1 and Solaris (SunOS 5.4) have the
88  * "old" one. Personally, I think the new semantics is much more logical.
89  * Note that "ln old new" where "new" is a symlink pointing to a non-existing
90  * file does succeed in both HP-UX and SunOs, but not in Solaris
91  * and in the old Linux semantics.
92  */
93 
94 /* [16-Dec-97 Kevin Buhr] For security reasons, we change some symlink
95  * semantics.  See the comments in "open_namei" and "do_link" below.
96  *
97  * [10-Sep-98 Alan Modra] Another symlink change.
98  */
99 
100 /* [Feb-Apr 2000 AV] Complete rewrite. Rules for symlinks:
101  *	inside the path - always follow.
102  *	in the last component in creation/removal/renaming - never follow.
103  *	if LOOKUP_FOLLOW passed - follow.
104  *	if the pathname has trailing slashes - follow.
105  *	otherwise - don't follow.
106  * (applied in that order).
107  *
108  * [Jun 2000 AV] Inconsistent behaviour of open() in case if flags==O_CREAT
109  * restored for 2.4. This is the last surviving part of old 4.2BSD bug.
110  * During the 2.4 we need to fix the userland stuff depending on it -
111  * hopefully we will be able to get rid of that wart in 2.5. So far only
112  * XEmacs seems to be relying on it...
113  */
114 /*
115  * [Sep 2001 AV] Single-semaphore locking scheme (kudos to David Holland)
116  * implemented.  Let's see if raised priority of ->s_vfs_rename_mutex gives
117  * any extra contention...
118  */
119 
120 /* In order to reduce some races, while at the same time doing additional
121  * checking and hopefully speeding things up, we copy filenames to the
122  * kernel data space before using them..
123  *
124  * POSIX.1 2.4: an empty pathname is invalid (ENOENT).
125  * PATH_MAX includes the nul terminator --RR.
126  */
127 
128 /* SLAB cache for struct filename instances */
129 static struct kmem_cache *__names_cache __ro_after_init;
130 #define names_cache	runtime_const_ptr(__names_cache)
131 
132 /*
133  * Type of the last component on LOOKUP_PARENT
134  */
135 enum last_type {LAST_NORM, LAST_ROOT, LAST_DOT, LAST_DOTDOT};
136 
137 void __init filename_init(void)
138 {
139 	__names_cache = kmem_cache_create_usercopy("names_cache", sizeof(struct filename), 0,
140 			 SLAB_HWCACHE_ALIGN|SLAB_PANIC, offsetof(struct filename, iname),
141 			 EMBEDDED_NAME_MAX, NULL);
142 	runtime_const_init(ptr, __names_cache);
143 }
144 
145 static inline struct filename *alloc_filename(void)
146 {
147 	return kmem_cache_alloc(names_cache, GFP_KERNEL);
148 }
149 
150 static inline void free_filename(struct filename *p)
151 {
152 	kmem_cache_free(names_cache, p);
153 }
154 
155 static inline void initname(struct filename *name)
156 {
157 	name->aname = NULL;
158 	name->refcnt = 1;
159 }
160 
161 static int getname_long(struct filename *name, const char __user *filename)
162 {
163 	int len;
164 	char *p __free(kfree) = kmalloc(PATH_MAX, GFP_KERNEL);
165 	if (unlikely(!p))
166 		return -ENOMEM;
167 
168 	memcpy(p, &name->iname, EMBEDDED_NAME_MAX);
169 	len = strncpy_from_user(p + EMBEDDED_NAME_MAX,
170 				filename + EMBEDDED_NAME_MAX,
171 				PATH_MAX - EMBEDDED_NAME_MAX);
172 	if (unlikely(len < 0))
173 		return len;
174 	if (unlikely(len == PATH_MAX - EMBEDDED_NAME_MAX))
175 		return -ENAMETOOLONG;
176 	name->name = no_free_ptr(p);
177 	return 0;
178 }
179 
180 static struct filename *
181 do_getname(const char __user *filename, int flags, bool incomplete)
182 {
183 	struct filename *result;
184 	char *kname;
185 	int len;
186 
187 	result = alloc_filename();
188 	if (unlikely(!result))
189 		return ERR_PTR(-ENOMEM);
190 
191 	/*
192 	 * First, try to embed the struct filename inside the names_cache
193 	 * allocation
194 	 */
195 	kname = (char *)result->iname;
196 	result->name = kname;
197 
198 	len = strncpy_from_user(kname, filename, EMBEDDED_NAME_MAX);
199 	/*
200 	 * Handle both empty path and copy failure in one go.
201 	 */
202 	if (unlikely(len <= 0)) {
203 		/* The empty path is special. */
204 		if (!len && !(flags & LOOKUP_EMPTY))
205 			len = -ENOENT;
206 	}
207 
208 	/*
209 	 * Uh-oh. We have a name that's approaching PATH_MAX. Allocate a
210 	 * separate struct filename so we can dedicate the entire
211 	 * names_cache allocation for the pathname, and re-do the copy from
212 	 * userland.
213 	 */
214 	if (unlikely(len == EMBEDDED_NAME_MAX))
215 		len = getname_long(result, filename);
216 	if (unlikely(len < 0)) {
217 		free_filename(result);
218 		return ERR_PTR(len);
219 	}
220 
221 	initname(result);
222 	if (likely(!incomplete))
223 		audit_getname(result);
224 	return result;
225 }
226 
227 struct filename *
228 getname_flags(const char __user *filename, int flags)
229 {
230 	return do_getname(filename, flags, false);
231 }
232 
233 struct filename *getname_uflags(const char __user *filename, int uflags)
234 {
235 	int flags = (uflags & AT_EMPTY_PATH) ? LOOKUP_EMPTY : 0;
236 
237 	return getname_flags(filename, flags);
238 }
239 
240 struct filename *__getname_maybe_null(const char __user *pathname)
241 {
242 	char c;
243 
244 	/* try to save on allocations; loss on um, though */
245 	if (get_user(c, pathname))
246 		return ERR_PTR(-EFAULT);
247 	if (!c)
248 		return NULL;
249 
250 	CLASS(filename_flags, name)(pathname, LOOKUP_EMPTY);
251 	/* empty pathname translates to NULL */
252 	if (!IS_ERR(name) && !(name->name[0]))
253 		return NULL;
254 	return no_free_ptr(name);
255 }
256 
257 static struct filename *do_getname_kernel(const char *filename, bool incomplete)
258 {
259 	struct filename *result;
260 	int len = strlen(filename) + 1;
261 	char *p;
262 
263 	if (unlikely(len > PATH_MAX))
264 		return ERR_PTR(-ENAMETOOLONG);
265 
266 	result = alloc_filename();
267 	if (unlikely(!result))
268 		return ERR_PTR(-ENOMEM);
269 
270 	if (len <= EMBEDDED_NAME_MAX) {
271 		p = (char *)result->iname;
272 		memcpy(p, filename, len);
273 	} else {
274 		p = kmemdup(filename, len, GFP_KERNEL);
275 		if (unlikely(!p)) {
276 			free_filename(result);
277 			return ERR_PTR(-ENOMEM);
278 		}
279 	}
280 	result->name = p;
281 	initname(result);
282 	if (likely(!incomplete))
283 		audit_getname(result);
284 	return result;
285 }
286 
287 struct filename *getname_kernel(const char *filename)
288 {
289 	return do_getname_kernel(filename, false);
290 }
291 EXPORT_SYMBOL(getname_kernel);
292 
293 void putname(struct filename *name)
294 {
295 	int refcnt;
296 
297 	if (IS_ERR_OR_NULL(name))
298 		return;
299 
300 	refcnt = name->refcnt;
301 	if (unlikely(refcnt != 1)) {
302 		if (WARN_ON_ONCE(!refcnt))
303 			return;
304 
305 		name->refcnt--;
306 		return;
307 	}
308 
309 	if (unlikely(name->name != name->iname))
310 		kfree(name->name);
311 	free_filename(name);
312 }
313 EXPORT_SYMBOL(putname);
314 
315 static inline int __delayed_getname(struct delayed_filename *v,
316 			   const char __user *string, int flags)
317 {
318 	v->__incomplete_filename = do_getname(string, flags, true);
319 	return PTR_ERR_OR_ZERO(v->__incomplete_filename);
320 }
321 
322 int delayed_getname(struct delayed_filename *v, const char __user *string)
323 {
324 	return __delayed_getname(v, string, 0);
325 }
326 
327 int delayed_getname_uflags(struct delayed_filename *v, const char __user *string,
328 			 int uflags)
329 {
330 	int flags = (uflags & AT_EMPTY_PATH) ? LOOKUP_EMPTY : 0;
331 	return __delayed_getname(v, string, flags);
332 }
333 
334 int putname_to_delayed(struct delayed_filename *v, struct filename *name)
335 {
336 	if (likely(name->refcnt == 1)) {
337 		v->__incomplete_filename = name;
338 		return 0;
339 	}
340 	name->refcnt--;
341 	v->__incomplete_filename = do_getname_kernel(name->name, true);
342 	return PTR_ERR_OR_ZERO(v->__incomplete_filename);
343 }
344 
345 void dismiss_delayed_filename(struct delayed_filename *v)
346 {
347 	putname(no_free_ptr(v->__incomplete_filename));
348 }
349 
350 struct filename *complete_getname(struct delayed_filename *v)
351 {
352 	struct filename *res = no_free_ptr(v->__incomplete_filename);
353 	if (!IS_ERR(res))
354 		audit_getname(res);
355 	return res;
356 }
357 
358 /**
359  * check_acl - perform ACL permission checking
360  * @idmap:	idmap of the mount the inode was found from
361  * @inode:	inode to check permissions on
362  * @mask:	right to check for (%MAY_READ, %MAY_WRITE, %MAY_EXEC ...)
363  *
364  * This function performs the ACL permission checking. Since this function
365  * retrieve POSIX acls it needs to know whether it is called from a blocking or
366  * non-blocking context and thus cares about the MAY_NOT_BLOCK bit.
367  *
368  * If the inode has been found through an idmapped mount the idmap of
369  * the vfsmount must be passed through @idmap. This function will then take
370  * care to map the inode according to @idmap before checking permissions.
371  * On non-idmapped mounts or if permission checking is to be performed on the
372  * raw inode simply pass @nop_mnt_idmap.
373  */
374 static int check_acl(struct mnt_idmap *idmap,
375 		     struct inode *inode, int mask)
376 {
377 #ifdef CONFIG_FS_POSIX_ACL
378 	struct posix_acl *acl;
379 
380 	if (mask & MAY_NOT_BLOCK) {
381 		acl = get_cached_acl_rcu(inode, ACL_TYPE_ACCESS);
382 	        if (!acl)
383 	                return -EAGAIN;
384 		/* no ->get_inode_acl() calls in RCU mode... */
385 		if (is_uncached_acl(acl))
386 			return -ECHILD;
387 	        return posix_acl_permission(idmap, inode, acl, mask);
388 	}
389 
390 	acl = get_inode_acl(inode, ACL_TYPE_ACCESS);
391 	if (IS_ERR(acl))
392 		return PTR_ERR(acl);
393 	if (acl) {
394 	        int error = posix_acl_permission(idmap, inode, acl, mask);
395 	        posix_acl_release(acl);
396 	        return error;
397 	}
398 #endif
399 
400 	return -EAGAIN;
401 }
402 
403 /*
404  * Very quick optimistic "we know we have no ACL's" check.
405  *
406  * Note that this is purely for ACL_TYPE_ACCESS, and purely
407  * for the "we have cached that there are no ACLs" case.
408  *
409  * If this returns true, we know there are no ACLs. But if
410  * it returns false, we might still not have ACLs (it could
411  * be the is_uncached_acl() case).
412  */
413 static inline bool no_acl_inode(struct inode *inode)
414 {
415 #ifdef CONFIG_FS_POSIX_ACL
416 	return likely(!READ_ONCE(inode->i_acl));
417 #else
418 	return true;
419 #endif
420 }
421 
422 /**
423  * acl_permission_check - perform basic UNIX permission checking
424  * @idmap:	idmap of the mount the inode was found from
425  * @inode:	inode to check permissions on
426  * @mask:	right to check for (%MAY_READ, %MAY_WRITE, %MAY_EXEC ...)
427  *
428  * This function performs the basic UNIX permission checking. Since this
429  * function may retrieve POSIX acls it needs to know whether it is called from a
430  * blocking or non-blocking context and thus cares about the MAY_NOT_BLOCK bit.
431  *
432  * If the inode has been found through an idmapped mount the idmap of
433  * the vfsmount must be passed through @idmap. This function will then take
434  * care to map the inode according to @idmap before checking permissions.
435  * On non-idmapped mounts or if permission checking is to be performed on the
436  * raw inode simply pass @nop_mnt_idmap.
437  */
438 static int acl_permission_check(struct mnt_idmap *idmap,
439 				struct inode *inode, int mask)
440 {
441 	unsigned int mode = inode->i_mode;
442 	vfsuid_t vfsuid;
443 
444 	/*
445 	 * Common cheap case: everybody has the requested
446 	 * rights, and there are no ACLs to check. No need
447 	 * to do any owner/group checks in that case.
448 	 *
449 	 *  - 'mask&7' is the requested permission bit set
450 	 *  - multiplying by 0111 spreads them out to all of ugo
451 	 *  - '& ~mode' looks for missing inode permission bits
452 	 *  - the '!' is for "no missing permissions"
453 	 *
454 	 * After that, we just need to check that there are no
455 	 * ACL's on the inode - do the 'IS_POSIXACL()' check last
456 	 * because it will dereference the ->i_sb pointer and we
457 	 * want to avoid that if at all possible.
458 	 */
459 	if (!((mask & 7) * 0111 & ~mode)) {
460 		if (no_acl_inode(inode))
461 			return 0;
462 		if (!IS_POSIXACL(inode))
463 			return 0;
464 	}
465 
466 	/* Are we the owner? If so, ACL's don't matter */
467 	vfsuid = i_uid_into_vfsuid(idmap, inode);
468 	if (likely(vfsuid_eq_kuid(vfsuid, current_fsuid()))) {
469 		mask &= 7;
470 		mode >>= 6;
471 		return (mask & ~mode) ? -EACCES : 0;
472 	}
473 
474 	/* Do we have ACL's? */
475 	if (IS_POSIXACL(inode) && (mode & S_IRWXG)) {
476 		int error = check_acl(idmap, inode, mask);
477 		if (error != -EAGAIN)
478 			return error;
479 	}
480 
481 	/* Only RWX matters for group/other mode bits */
482 	mask &= 7;
483 
484 	/*
485 	 * Are the group permissions different from
486 	 * the other permissions in the bits we care
487 	 * about? Need to check group ownership if so.
488 	 */
489 	if (mask & (mode ^ (mode >> 3))) {
490 		vfsgid_t vfsgid = i_gid_into_vfsgid(idmap, inode);
491 		if (vfsgid_in_group_p(vfsgid))
492 			mode >>= 3;
493 	}
494 
495 	/* Bits in 'mode' clear that we require? */
496 	return (mask & ~mode) ? -EACCES : 0;
497 }
498 
499 /**
500  * generic_permission -  check for access rights on a Posix-like filesystem
501  * @idmap:	idmap of the mount the inode was found from
502  * @inode:	inode to check access rights for
503  * @mask:	right to check for (%MAY_READ, %MAY_WRITE, %MAY_EXEC,
504  *		%MAY_NOT_BLOCK ...)
505  *
506  * Used to check for read/write/execute permissions on a file.
507  * We use "fsuid" for this, letting us set arbitrary permissions
508  * for filesystem access without changing the "normal" uids which
509  * are used for other things.
510  *
511  * generic_permission is rcu-walk aware. It returns -ECHILD in case an rcu-walk
512  * request cannot be satisfied (eg. requires blocking or too much complexity).
513  * It would then be called again in ref-walk mode.
514  *
515  * If the inode has been found through an idmapped mount the idmap of
516  * the vfsmount must be passed through @idmap. This function will then take
517  * care to map the inode according to @idmap before checking permissions.
518  * On non-idmapped mounts or if permission checking is to be performed on the
519  * raw inode simply pass @nop_mnt_idmap.
520  */
521 int generic_permission(struct mnt_idmap *idmap, struct inode *inode,
522 		       int mask)
523 {
524 	int ret;
525 
526 	/*
527 	 * Do the basic permission checks.
528 	 */
529 	ret = acl_permission_check(idmap, inode, mask);
530 	if (ret != -EACCES)
531 		return ret;
532 
533 	if (S_ISDIR(inode->i_mode)) {
534 		/* DACs are overridable for directories */
535 		if (!(mask & MAY_WRITE))
536 			if (capable_wrt_inode_uidgid(idmap, inode,
537 						     CAP_DAC_READ_SEARCH))
538 				return 0;
539 		if (capable_wrt_inode_uidgid(idmap, inode,
540 					     CAP_DAC_OVERRIDE))
541 			return 0;
542 		return -EACCES;
543 	}
544 
545 	/*
546 	 * Searching includes executable on directories, else just read.
547 	 */
548 	mask &= MAY_READ | MAY_WRITE | MAY_EXEC;
549 	if (mask == MAY_READ)
550 		if (capable_wrt_inode_uidgid(idmap, inode,
551 					     CAP_DAC_READ_SEARCH))
552 			return 0;
553 	/*
554 	 * Read/write DACs are always overridable.
555 	 * Executable DACs are overridable when there is
556 	 * at least one exec bit set.
557 	 */
558 	if (!(mask & MAY_EXEC) || (inode->i_mode & S_IXUGO))
559 		if (capable_wrt_inode_uidgid(idmap, inode,
560 					     CAP_DAC_OVERRIDE))
561 			return 0;
562 
563 	return -EACCES;
564 }
565 EXPORT_SYMBOL(generic_permission);
566 
567 /**
568  * do_inode_permission - UNIX permission checking
569  * @idmap:	idmap of the mount the inode was found from
570  * @inode:	inode to check permissions on
571  * @mask:	right to check for (%MAY_READ, %MAY_WRITE, %MAY_EXEC ...)
572  *
573  * We _really_ want to just do "generic_permission()" without
574  * even looking at the inode->i_op values. So we keep a cache
575  * flag in inode->i_opflags, that says "this has not special
576  * permission function, use the fast case".
577  */
578 static inline int do_inode_permission(struct mnt_idmap *idmap,
579 				      struct inode *inode, int mask)
580 {
581 	if (unlikely(!(inode->i_opflags & IOP_FASTPERM))) {
582 		if (likely(inode->i_op->permission))
583 			return inode->i_op->permission(idmap, inode, mask);
584 
585 		/* This gets set once for the inode lifetime */
586 		spin_lock(&inode->i_lock);
587 		inode->i_opflags |= IOP_FASTPERM;
588 		spin_unlock(&inode->i_lock);
589 	}
590 	return generic_permission(idmap, inode, mask);
591 }
592 
593 /**
594  * sb_permission - Check superblock-level permissions
595  * @sb: Superblock of inode to check permission on
596  * @inode: Inode to check permission on
597  * @mask: Right to check for (%MAY_READ, %MAY_WRITE, %MAY_EXEC)
598  *
599  * Separate out file-system wide checks from inode-specific permission checks.
600  *
601  * Note: lookup_inode_permission_may_exec() does not call here. If you add
602  * MAY_EXEC checks, adjust it.
603  */
604 static int sb_permission(struct super_block *sb, struct inode *inode, int mask)
605 {
606 	if (mask & MAY_WRITE) {
607 		umode_t mode = inode->i_mode;
608 
609 		/* Nobody gets write access to a read-only fs. */
610 		if (sb_rdonly(sb) && (S_ISREG(mode) || S_ISDIR(mode) || S_ISLNK(mode)))
611 			return -EROFS;
612 	}
613 	return 0;
614 }
615 
616 /**
617  * inode_permission - Check for access rights to a given inode
618  * @idmap:	idmap of the mount the inode was found from
619  * @inode:	Inode to check permission on
620  * @mask:	Right to check for (%MAY_READ, %MAY_WRITE, %MAY_EXEC)
621  *
622  * Check for read/write/execute permissions on an inode.  We use fs[ug]id for
623  * this, letting us set arbitrary permissions for filesystem access without
624  * changing the "normal" UIDs which are used for other things.
625  *
626  * When checking for MAY_APPEND, MAY_WRITE must also be set in @mask.
627  */
628 int inode_permission(struct mnt_idmap *idmap,
629 		     struct inode *inode, int mask)
630 {
631 	int retval;
632 
633 	retval = sb_permission(inode->i_sb, inode, mask);
634 	if (unlikely(retval))
635 		return retval;
636 
637 	if (mask & MAY_WRITE) {
638 		/*
639 		 * Nobody gets write access to an immutable file.
640 		 */
641 		if (unlikely(IS_IMMUTABLE(inode)))
642 			return -EPERM;
643 
644 		/*
645 		 * Updating mtime will likely cause i_uid and i_gid to be
646 		 * written back improperly if their true value is unknown
647 		 * to the vfs.
648 		 */
649 		if (unlikely(HAS_UNMAPPED_ID(idmap, inode)))
650 			return -EACCES;
651 	}
652 
653 	retval = do_inode_permission(idmap, inode, mask);
654 	if (unlikely(retval))
655 		return retval;
656 
657 	retval = devcgroup_inode_permission(inode, mask);
658 	if (unlikely(retval))
659 		return retval;
660 
661 	return security_inode_permission(inode, mask);
662 }
663 EXPORT_SYMBOL(inode_permission);
664 
665 /*
666  * lookup_inode_permission_may_exec - Check traversal right for given inode
667  *
668  * This is a special case routine for may_lookup() making assumptions specific
669  * to path traversal. Use inode_permission() if you are doing something else.
670  *
671  * Work is shaved off compared to inode_permission() as follows:
672  * - we know for a fact there is no MAY_WRITE to worry about
673  * - it is an invariant the inode is a directory
674  *
675  * Since majority of real-world traversal happens on inodes which grant it for
676  * everyone, we check it upfront and only resort to more expensive work if it
677  * fails.
678  *
679  * Filesystems which have their own ->permission hook and consequently miss out
680  * on IOP_FASTPERM can still get the optimization if they set IOP_FASTPERM_MAY_EXEC
681  * on their directory inodes.
682  */
683 static __always_inline int lookup_inode_permission_may_exec(struct mnt_idmap *idmap,
684 	struct inode *inode, int mask)
685 {
686 	/* Lookup already checked this to return -ENOTDIR */
687 	VFS_BUG_ON_INODE(!S_ISDIR(inode->i_mode), inode);
688 	VFS_BUG_ON((mask & ~MAY_NOT_BLOCK) != 0);
689 
690 	mask |= MAY_EXEC;
691 
692 	if (unlikely(!(inode->i_opflags & (IOP_FASTPERM | IOP_FASTPERM_MAY_EXEC))))
693 		return inode_permission(idmap, inode, mask);
694 
695 	if (unlikely(((inode->i_mode & 0111) != 0111) || !no_acl_inode(inode)))
696 		return inode_permission(idmap, inode, mask);
697 
698 	return security_inode_permission(inode, mask);
699 }
700 
701 /**
702  * path_get - get a reference to a path
703  * @path: path to get the reference to
704  *
705  * Given a path increment the reference count to the dentry and the vfsmount.
706  */
707 void path_get(const struct path *path)
708 {
709 	mntget(path->mnt);
710 	dget(path->dentry);
711 }
712 EXPORT_SYMBOL(path_get);
713 
714 /**
715  * path_put - put a reference to a path
716  * @path: path to put the reference to
717  *
718  * Given a path decrement the reference count to the dentry and the vfsmount.
719  */
720 void path_put(const struct path *path)
721 {
722 	dput(path->dentry);
723 	mntput(path->mnt);
724 }
725 EXPORT_SYMBOL(path_put);
726 
727 #define EMBEDDED_LEVELS 2
728 struct nameidata {
729 	struct path	path;
730 	struct qstr	last;
731 	struct path	root;
732 	struct inode	*inode; /* path.dentry.d_inode */
733 	unsigned int	flags, state;
734 	unsigned	seq, next_seq, m_seq, r_seq;
735 	enum last_type	last_type;
736 	unsigned	depth;
737 	int		total_link_count;
738 	struct saved {
739 		struct path link;
740 		struct delayed_call done;
741 		const char *name;
742 		unsigned seq;
743 	} *stack, internal[EMBEDDED_LEVELS];
744 	struct filename	*name;
745 	const char *pathname;
746 	struct nameidata *saved;
747 	unsigned	root_seq;
748 	int		dfd;
749 	vfsuid_t	dir_vfsuid;
750 	umode_t		dir_mode;
751 } __randomize_layout;
752 
753 #define ND_ROOT_PRESET 1
754 #define ND_ROOT_GRABBED 2
755 #define ND_JUMPED 4
756 
757 static void __set_nameidata(struct nameidata *p, int dfd, struct filename *name)
758 {
759 	struct nameidata *old = current->nameidata;
760 	p->stack = p->internal;
761 	p->depth = 0;
762 	p->dfd = dfd;
763 	p->name = name;
764 	p->pathname = likely(name) ? name->name : "";
765 	p->path.mnt = NULL;
766 	p->path.dentry = NULL;
767 	p->total_link_count = old ? old->total_link_count : 0;
768 	p->saved = old;
769 	current->nameidata = p;
770 }
771 
772 static inline void set_nameidata(struct nameidata *p, int dfd, struct filename *name,
773 			  const struct path *root)
774 {
775 	__set_nameidata(p, dfd, name);
776 	p->state = 0;
777 	if (unlikely(root)) {
778 		p->state = ND_ROOT_PRESET;
779 		p->root = *root;
780 	}
781 }
782 
783 static void restore_nameidata(void)
784 {
785 	struct nameidata *now = current->nameidata, *old = now->saved;
786 
787 	current->nameidata = old;
788 	if (old)
789 		old->total_link_count = now->total_link_count;
790 	if (now->stack != now->internal)
791 		kfree(now->stack);
792 }
793 
794 static bool nd_alloc_stack(struct nameidata *nd)
795 {
796 	struct saved *p;
797 
798 	p= kmalloc_objs(struct saved, MAXSYMLINKS,
799 			nd->flags & LOOKUP_RCU ? GFP_ATOMIC : GFP_KERNEL);
800 	if (unlikely(!p))
801 		return false;
802 	memcpy(p, nd->internal, sizeof(nd->internal));
803 	nd->stack = p;
804 	return true;
805 }
806 
807 /**
808  * path_connected - Verify that a dentry is below mnt.mnt_root
809  * @mnt: The mountpoint to check.
810  * @dentry: The dentry to check.
811  *
812  * Rename can sometimes move a file or directory outside of a bind
813  * mount, path_connected allows those cases to be detected.
814  */
815 static bool path_connected(struct vfsmount *mnt, struct dentry *dentry)
816 {
817 	struct super_block *sb = mnt->mnt_sb;
818 
819 	/* Bind mounts can have disconnected paths */
820 	if (mnt->mnt_root == sb->s_root)
821 		return true;
822 
823 	return is_subdir(dentry, mnt->mnt_root);
824 }
825 
826 static void drop_links(struct nameidata *nd)
827 {
828 	int i = nd->depth;
829 	while (i--) {
830 		struct saved *last = nd->stack + i;
831 		do_delayed_call(&last->done);
832 		clear_delayed_call(&last->done);
833 	}
834 }
835 
836 static void leave_rcu(struct nameidata *nd)
837 {
838 	nd->flags &= ~LOOKUP_RCU;
839 	nd->seq = nd->next_seq = 0;
840 	rcu_read_unlock();
841 }
842 
843 static void terminate_walk(struct nameidata *nd)
844 {
845 	if (unlikely(nd->depth))
846 		drop_links(nd);
847 	if (!(nd->flags & LOOKUP_RCU)) {
848 		int i;
849 		path_put(&nd->path);
850 		for (i = 0; i < nd->depth; i++)
851 			path_put(&nd->stack[i].link);
852 		if (nd->state & ND_ROOT_GRABBED) {
853 			path_put(&nd->root);
854 			nd->state &= ~ND_ROOT_GRABBED;
855 		}
856 	} else {
857 		leave_rcu(nd);
858 	}
859 	nd->depth = 0;
860 	nd->path.mnt = NULL;
861 	nd->path.dentry = NULL;
862 }
863 
864 /* path_put is needed afterwards regardless of success or failure */
865 static bool __legitimize_path(struct path *path, unsigned seq, unsigned mseq)
866 {
867 	int res = __legitimize_mnt(path->mnt, mseq);
868 	if (unlikely(res)) {
869 		if (res > 0)
870 			path->mnt = NULL;
871 		path->dentry = NULL;
872 		return false;
873 	}
874 	if (unlikely(!lockref_get_not_dead(&path->dentry->d_lockref))) {
875 		path->dentry = NULL;
876 		return false;
877 	}
878 	return !read_seqcount_retry(&path->dentry->d_seq, seq);
879 }
880 
881 static inline bool legitimize_path(struct nameidata *nd,
882 			    struct path *path, unsigned seq)
883 {
884 	return __legitimize_path(path, seq, nd->m_seq);
885 }
886 
887 static bool legitimize_links(struct nameidata *nd)
888 {
889 	int i;
890 
891 	VFS_BUG_ON(nd->flags & LOOKUP_CACHED);
892 
893 	for (i = 0; i < nd->depth; i++) {
894 		struct saved *last = nd->stack + i;
895 		if (unlikely(!legitimize_path(nd, &last->link, last->seq))) {
896 			drop_links(nd);
897 			nd->depth = i + 1;
898 			return false;
899 		}
900 	}
901 	return true;
902 }
903 
904 static bool legitimize_root(struct nameidata *nd)
905 {
906 	/* Nothing to do if nd->root is zero or is managed by the VFS user. */
907 	if (!nd->root.mnt || (nd->state & ND_ROOT_PRESET))
908 		return true;
909 	nd->state |= ND_ROOT_GRABBED;
910 	return legitimize_path(nd, &nd->root, nd->root_seq);
911 }
912 
913 /*
914  * Path walking has 2 modes, rcu-walk and ref-walk (see
915  * Documentation/filesystems/path-lookup.txt).  In situations when we can't
916  * continue in RCU mode, we attempt to drop out of rcu-walk mode and grab
917  * normal reference counts on dentries and vfsmounts to transition to ref-walk
918  * mode.  Refcounts are grabbed at the last known good point before rcu-walk
919  * got stuck, so ref-walk may continue from there. If this is not successful
920  * (eg. a seqcount has changed), then failure is returned and it's up to caller
921  * to restart the path walk from the beginning in ref-walk mode.
922  */
923 
924 /**
925  * try_to_unlazy - try to switch to ref-walk mode.
926  * @nd: nameidata pathwalk data
927  * Returns: true on success, false on failure
928  *
929  * try_to_unlazy attempts to legitimize the current nd->path and nd->root
930  * for ref-walk mode.
931  * Must be called from rcu-walk context.
932  * Nothing should touch nameidata between try_to_unlazy() failure and
933  * terminate_walk().
934  */
935 static bool try_to_unlazy(struct nameidata *nd)
936 {
937 	struct dentry *parent = nd->path.dentry;
938 
939 	VFS_BUG_ON(!(nd->flags & LOOKUP_RCU));
940 
941 	if (unlikely(nd->flags & LOOKUP_CACHED)) {
942 		drop_links(nd);
943 		nd->depth = 0;
944 		goto out1;
945 	}
946 	if (unlikely(nd->depth && !legitimize_links(nd)))
947 		goto out1;
948 	if (unlikely(!legitimize_path(nd, &nd->path, nd->seq)))
949 		goto out;
950 	if (unlikely(!legitimize_root(nd)))
951 		goto out;
952 	leave_rcu(nd);
953 	BUG_ON(nd->inode != parent->d_inode);
954 	return true;
955 
956 out1:
957 	nd->path.mnt = NULL;
958 	nd->path.dentry = NULL;
959 out:
960 	leave_rcu(nd);
961 	return false;
962 }
963 
964 /**
965  * try_to_unlazy_next - try to switch to ref-walk mode.
966  * @nd: nameidata pathwalk data
967  * @dentry: next dentry to step into
968  * Returns: true on success, false on failure
969  *
970  * Similar to try_to_unlazy(), but here we have the next dentry already
971  * picked by rcu-walk and want to legitimize that in addition to the current
972  * nd->path and nd->root for ref-walk mode.  Must be called from rcu-walk context.
973  * Nothing should touch nameidata between try_to_unlazy_next() failure and
974  * terminate_walk().
975  */
976 static bool try_to_unlazy_next(struct nameidata *nd, struct dentry *dentry)
977 {
978 	int res;
979 
980 	VFS_BUG_ON(!(nd->flags & LOOKUP_RCU));
981 
982 	if (unlikely(nd->flags & LOOKUP_CACHED)) {
983 		drop_links(nd);
984 		nd->depth = 0;
985 		goto out2;
986 	}
987 	if (unlikely(nd->depth && !legitimize_links(nd)))
988 		goto out2;
989 	res = __legitimize_mnt(nd->path.mnt, nd->m_seq);
990 	if (unlikely(res)) {
991 		if (res > 0)
992 			goto out2;
993 		goto out1;
994 	}
995 	if (unlikely(!lockref_get_not_dead(&nd->path.dentry->d_lockref)))
996 		goto out1;
997 
998 	/*
999 	 * We need to move both the parent and the dentry from the RCU domain
1000 	 * to be properly refcounted. And the sequence number in the dentry
1001 	 * validates *both* dentry counters, since we checked the sequence
1002 	 * number of the parent after we got the child sequence number. So we
1003 	 * know the parent must still be valid if the child sequence number is
1004 	 */
1005 	if (unlikely(!lockref_get_not_dead(&dentry->d_lockref)))
1006 		goto out;
1007 	if (read_seqcount_retry(&dentry->d_seq, nd->next_seq))
1008 		goto out_dput;
1009 	/*
1010 	 * Sequence counts matched. Now make sure that the root is
1011 	 * still valid and get it if required.
1012 	 */
1013 	if (unlikely(!legitimize_root(nd)))
1014 		goto out_dput;
1015 	leave_rcu(nd);
1016 	return true;
1017 
1018 out2:
1019 	nd->path.mnt = NULL;
1020 out1:
1021 	nd->path.dentry = NULL;
1022 out:
1023 	leave_rcu(nd);
1024 	return false;
1025 out_dput:
1026 	leave_rcu(nd);
1027 	dput(dentry);
1028 	return false;
1029 }
1030 
1031 static inline int d_revalidate(struct inode *dir, const struct qstr *name,
1032 			       struct dentry *dentry, unsigned int flags)
1033 {
1034 	if (unlikely(dentry->d_flags & DCACHE_OP_REVALIDATE))
1035 		return dentry->d_op->d_revalidate(dir, name, dentry, flags);
1036 	else
1037 		return 1;
1038 }
1039 
1040 /**
1041  * complete_walk - successful completion of path walk
1042  * @nd:  pointer nameidata
1043  *
1044  * If we had been in RCU mode, drop out of it and legitimize nd->path.
1045  * Revalidate the final result, unless we'd already done that during
1046  * the path walk or the filesystem doesn't ask for it.  Return 0 on
1047  * success, -error on failure.  In case of failure caller does not
1048  * need to drop nd->path.
1049  */
1050 static int complete_walk(struct nameidata *nd)
1051 {
1052 	struct dentry *dentry = nd->path.dentry;
1053 	int status;
1054 
1055 	if (nd->flags & LOOKUP_RCU) {
1056 		/*
1057 		 * We don't want to zero nd->root for scoped-lookups or
1058 		 * externally-managed nd->root.
1059 		 */
1060 		if (likely(!(nd->state & ND_ROOT_PRESET)))
1061 			if (likely(!(nd->flags & LOOKUP_IS_SCOPED)))
1062 				nd->root.mnt = NULL;
1063 		nd->flags &= ~LOOKUP_CACHED;
1064 		if (!try_to_unlazy(nd))
1065 			return -ECHILD;
1066 	}
1067 
1068 	if (unlikely(nd->flags & LOOKUP_IS_SCOPED)) {
1069 		/*
1070 		 * While the guarantee of LOOKUP_IS_SCOPED is (roughly) "don't
1071 		 * ever step outside the root during lookup" and should already
1072 		 * be guaranteed by the rest of namei, we want to avoid a namei
1073 		 * BUG resulting in userspace being given a path that was not
1074 		 * scoped within the root at some point during the lookup.
1075 		 *
1076 		 * So, do a final sanity-check to make sure that in the
1077 		 * worst-case scenario (a complete bypass of LOOKUP_IS_SCOPED)
1078 		 * we won't silently return an fd completely outside of the
1079 		 * requested root to userspace.
1080 		 *
1081 		 * Userspace could move the path outside the root after this
1082 		 * check, but as discussed elsewhere this is not a concern (the
1083 		 * resolved file was inside the root at some point).
1084 		 */
1085 		if (!path_is_under(&nd->path, &nd->root))
1086 			return -EXDEV;
1087 	}
1088 
1089 	if (likely(!(nd->state & ND_JUMPED)))
1090 		return 0;
1091 
1092 	if (likely(!(dentry->d_flags & DCACHE_OP_WEAK_REVALIDATE)))
1093 		return 0;
1094 
1095 	status = dentry->d_op->d_weak_revalidate(dentry, nd->flags);
1096 	if (status > 0)
1097 		return 0;
1098 
1099 	if (!status)
1100 		status = -ESTALE;
1101 
1102 	return status;
1103 }
1104 
1105 static int set_root(struct nameidata *nd)
1106 {
1107 	struct fs_struct *fs = current->fs;
1108 
1109 	/*
1110 	 * Jumping to the real root in a scoped-lookup is a BUG in namei, but we
1111 	 * still have to ensure it doesn't happen because it will cause a breakout
1112 	 * from the dirfd.
1113 	 */
1114 	if (WARN_ON(nd->flags & LOOKUP_IS_SCOPED))
1115 		return -ENOTRECOVERABLE;
1116 
1117 	if (nd->flags & LOOKUP_RCU) {
1118 		unsigned seq;
1119 
1120 		do {
1121 			seq = read_seqbegin(&fs->seq);
1122 			nd->root = fs->root;
1123 			nd->root_seq = __read_seqcount_begin(&nd->root.dentry->d_seq);
1124 		} while (read_seqretry(&fs->seq, seq));
1125 	} else {
1126 		get_fs_root(fs, &nd->root);
1127 		nd->state |= ND_ROOT_GRABBED;
1128 	}
1129 	return 0;
1130 }
1131 
1132 static int nd_jump_root(struct nameidata *nd)
1133 {
1134 	if (unlikely(nd->flags & LOOKUP_BENEATH))
1135 		return -EXDEV;
1136 	if (unlikely(nd->flags & LOOKUP_NO_XDEV)) {
1137 		/* Absolute path arguments to path_init() are allowed. */
1138 		if (nd->path.mnt != NULL && nd->path.mnt != nd->root.mnt)
1139 			return -EXDEV;
1140 	}
1141 	if (!nd->root.mnt) {
1142 		int error = set_root(nd);
1143 		if (unlikely(error))
1144 			return error;
1145 	}
1146 	if (nd->flags & LOOKUP_RCU) {
1147 		struct dentry *d;
1148 		nd->path = nd->root;
1149 		d = nd->path.dentry;
1150 		nd->inode = d->d_inode;
1151 		nd->seq = nd->root_seq;
1152 		if (read_seqcount_retry(&d->d_seq, nd->seq))
1153 			return -ECHILD;
1154 	} else {
1155 		path_put(&nd->path);
1156 		nd->path = nd->root;
1157 		path_get(&nd->path);
1158 		nd->inode = nd->path.dentry->d_inode;
1159 	}
1160 	nd->state |= ND_JUMPED;
1161 	return 0;
1162 }
1163 
1164 /*
1165  * Helper to directly jump to a known parsed path from ->get_link,
1166  * caller must have taken a reference to path beforehand.
1167  */
1168 int nd_jump_link(const struct path *path)
1169 {
1170 	int error = -ELOOP;
1171 	struct nameidata *nd = current->nameidata;
1172 
1173 	if (unlikely(nd->flags & LOOKUP_NO_MAGICLINKS))
1174 		goto err;
1175 
1176 	error = -EXDEV;
1177 	if (unlikely(nd->flags & LOOKUP_NO_XDEV)) {
1178 		if (nd->path.mnt != path->mnt)
1179 			goto err;
1180 	}
1181 	/* Not currently safe for scoped-lookups. */
1182 	if (unlikely(nd->flags & LOOKUP_IS_SCOPED))
1183 		goto err;
1184 
1185 	path_put(&nd->path);
1186 	nd->path = *path;
1187 	nd->inode = nd->path.dentry->d_inode;
1188 	nd->state |= ND_JUMPED;
1189 	return 0;
1190 
1191 err:
1192 	path_put(path);
1193 	return error;
1194 }
1195 
1196 static inline void put_link(struct nameidata *nd)
1197 {
1198 	struct saved *last = nd->stack + --nd->depth;
1199 	do_delayed_call(&last->done);
1200 	if (!(nd->flags & LOOKUP_RCU))
1201 		path_put(&last->link);
1202 }
1203 
1204 static int sysctl_protected_symlinks __read_mostly;
1205 static int sysctl_protected_hardlinks __read_mostly;
1206 static int sysctl_protected_fifos __read_mostly;
1207 static int sysctl_protected_regular __read_mostly;
1208 
1209 #ifdef CONFIG_SYSCTL
1210 static const struct ctl_table namei_sysctls[] = {
1211 	{
1212 		.procname	= "protected_symlinks",
1213 		.data		= &sysctl_protected_symlinks,
1214 		.maxlen		= sizeof(int),
1215 		.mode		= 0644,
1216 		.proc_handler	= proc_dointvec_minmax,
1217 		.extra1		= SYSCTL_ZERO,
1218 		.extra2		= SYSCTL_ONE,
1219 	},
1220 	{
1221 		.procname	= "protected_hardlinks",
1222 		.data		= &sysctl_protected_hardlinks,
1223 		.maxlen		= sizeof(int),
1224 		.mode		= 0644,
1225 		.proc_handler	= proc_dointvec_minmax,
1226 		.extra1		= SYSCTL_ZERO,
1227 		.extra2		= SYSCTL_ONE,
1228 	},
1229 	{
1230 		.procname	= "protected_fifos",
1231 		.data		= &sysctl_protected_fifos,
1232 		.maxlen		= sizeof(int),
1233 		.mode		= 0644,
1234 		.proc_handler	= proc_dointvec_minmax,
1235 		.extra1		= SYSCTL_ZERO,
1236 		.extra2		= SYSCTL_TWO,
1237 	},
1238 	{
1239 		.procname	= "protected_regular",
1240 		.data		= &sysctl_protected_regular,
1241 		.maxlen		= sizeof(int),
1242 		.mode		= 0644,
1243 		.proc_handler	= proc_dointvec_minmax,
1244 		.extra1		= SYSCTL_ZERO,
1245 		.extra2		= SYSCTL_TWO,
1246 	},
1247 };
1248 
1249 static int __init init_fs_namei_sysctls(void)
1250 {
1251 	register_sysctl_init("fs", namei_sysctls);
1252 	return 0;
1253 }
1254 fs_initcall(init_fs_namei_sysctls);
1255 
1256 #endif /* CONFIG_SYSCTL */
1257 
1258 /**
1259  * may_follow_link - Check symlink following for unsafe situations
1260  * @nd: nameidata pathwalk data
1261  * @inode: Used for idmapping.
1262  *
1263  * In the case of the sysctl_protected_symlinks sysctl being enabled,
1264  * CAP_DAC_OVERRIDE needs to be specifically ignored if the symlink is
1265  * in a sticky world-writable directory. This is to protect privileged
1266  * processes from failing races against path names that may change out
1267  * from under them by way of other users creating malicious symlinks.
1268  * It will permit symlinks to be followed only when outside a sticky
1269  * world-writable directory, or when the uid of the symlink and follower
1270  * match, or when the directory owner matches the symlink's owner.
1271  *
1272  * Returns 0 if following the symlink is allowed, -ve on error.
1273  */
1274 static inline int may_follow_link(struct nameidata *nd, const struct inode *inode)
1275 {
1276 	struct mnt_idmap *idmap;
1277 	vfsuid_t vfsuid;
1278 
1279 	if (!sysctl_protected_symlinks)
1280 		return 0;
1281 
1282 	idmap = mnt_idmap(nd->path.mnt);
1283 	vfsuid = i_uid_into_vfsuid(idmap, inode);
1284 	/* Allowed if owner and follower match. */
1285 	if (vfsuid_eq_kuid(vfsuid, current_fsuid()))
1286 		return 0;
1287 
1288 	/* Allowed if parent directory not sticky and world-writable. */
1289 	if ((nd->dir_mode & (S_ISVTX|S_IWOTH)) != (S_ISVTX|S_IWOTH))
1290 		return 0;
1291 
1292 	/* Allowed if parent directory and link owner match. */
1293 	if (vfsuid_valid(nd->dir_vfsuid) && vfsuid_eq(nd->dir_vfsuid, vfsuid))
1294 		return 0;
1295 
1296 	if (nd->flags & LOOKUP_RCU)
1297 		return -ECHILD;
1298 
1299 	audit_inode(nd->name, nd->stack[0].link.dentry, 0);
1300 	audit_log_path_denied(AUDIT_ANOM_LINK, "follow_link");
1301 	return -EACCES;
1302 }
1303 
1304 /**
1305  * safe_hardlink_source - Check for safe hardlink conditions
1306  * @idmap: idmap of the mount the inode was found from
1307  * @inode: the source inode to hardlink from
1308  *
1309  * Return false if at least one of the following conditions:
1310  *    - inode is not a regular file
1311  *    - inode is setuid
1312  *    - inode is setgid and group-exec
1313  *    - access failure for read and write
1314  *
1315  * Otherwise returns true.
1316  */
1317 static bool safe_hardlink_source(struct mnt_idmap *idmap,
1318 				 struct inode *inode)
1319 {
1320 	umode_t mode = inode->i_mode;
1321 
1322 	/* Special files should not get pinned to the filesystem. */
1323 	if (!S_ISREG(mode))
1324 		return false;
1325 
1326 	/* Setuid files should not get pinned to the filesystem. */
1327 	if (mode & S_ISUID)
1328 		return false;
1329 
1330 	/* Executable setgid files should not get pinned to the filesystem. */
1331 	if ((mode & (S_ISGID | S_IXGRP)) == (S_ISGID | S_IXGRP))
1332 		return false;
1333 
1334 	/* Hardlinking to unreadable or unwritable sources is dangerous. */
1335 	if (inode_permission(idmap, inode, MAY_READ | MAY_WRITE))
1336 		return false;
1337 
1338 	return true;
1339 }
1340 
1341 /**
1342  * may_linkat - Check permissions for creating a hardlink
1343  * @idmap: idmap of the mount the inode was found from
1344  * @link:  the source to hardlink from
1345  *
1346  * Block hardlink when all of:
1347  *  - sysctl_protected_hardlinks enabled
1348  *  - fsuid does not match inode
1349  *  - hardlink source is unsafe (see safe_hardlink_source() above)
1350  *  - not CAP_FOWNER in a namespace with the inode owner uid mapped
1351  *
1352  * If the inode has been found through an idmapped mount the idmap of
1353  * the vfsmount must be passed through @idmap. This function will then take
1354  * care to map the inode according to @idmap before checking permissions.
1355  * On non-idmapped mounts or if permission checking is to be performed on the
1356  * raw inode simply pass @nop_mnt_idmap.
1357  *
1358  * Returns 0 if successful, -ve on error.
1359  */
1360 int may_linkat(struct mnt_idmap *idmap, const struct path *link)
1361 {
1362 	struct inode *inode = link->dentry->d_inode;
1363 
1364 	/* Inode writeback is not safe when the uid or gid are invalid. */
1365 	if (!vfsuid_valid(i_uid_into_vfsuid(idmap, inode)) ||
1366 	    !vfsgid_valid(i_gid_into_vfsgid(idmap, inode)))
1367 		return -EOVERFLOW;
1368 
1369 	if (!sysctl_protected_hardlinks)
1370 		return 0;
1371 
1372 	/* Source inode owner (or CAP_FOWNER) can hardlink all they like,
1373 	 * otherwise, it must be a safe source.
1374 	 */
1375 	if (safe_hardlink_source(idmap, inode) ||
1376 	    inode_owner_or_capable(idmap, inode))
1377 		return 0;
1378 
1379 	audit_log_path_denied(AUDIT_ANOM_LINK, "linkat");
1380 	return -EPERM;
1381 }
1382 
1383 /**
1384  * may_create_in_sticky - Check whether an O_CREAT open in a sticky directory
1385  *			  should be allowed, or not, on files that already
1386  *			  exist.
1387  * @idmap: idmap of the mount the inode was found from
1388  * @nd: nameidata pathwalk data
1389  * @inode: the inode of the file to open
1390  *
1391  * Block an O_CREAT open of a FIFO (or a regular file) when:
1392  *   - sysctl_protected_fifos (or sysctl_protected_regular) is enabled
1393  *   - the file already exists
1394  *   - we are in a sticky directory
1395  *   - we don't own the file
1396  *   - the owner of the directory doesn't own the file
1397  *   - the directory is world writable
1398  * If the sysctl_protected_fifos (or sysctl_protected_regular) is set to 2
1399  * the directory doesn't have to be world writable: being group writable will
1400  * be enough.
1401  *
1402  * If the inode has been found through an idmapped mount the idmap of
1403  * the vfsmount must be passed through @idmap. This function will then take
1404  * care to map the inode according to @idmap before checking permissions.
1405  * On non-idmapped mounts or if permission checking is to be performed on the
1406  * raw inode simply pass @nop_mnt_idmap.
1407  *
1408  * Returns 0 if the open is allowed, -ve on error.
1409  */
1410 static int may_create_in_sticky(struct mnt_idmap *idmap, struct nameidata *nd,
1411 				struct inode *const inode)
1412 {
1413 	umode_t dir_mode = nd->dir_mode;
1414 	vfsuid_t dir_vfsuid = nd->dir_vfsuid, i_vfsuid;
1415 
1416 	if (likely(!(dir_mode & S_ISVTX)))
1417 		return 0;
1418 
1419 	if (S_ISREG(inode->i_mode) && !sysctl_protected_regular)
1420 		return 0;
1421 
1422 	if (S_ISFIFO(inode->i_mode) && !sysctl_protected_fifos)
1423 		return 0;
1424 
1425 	i_vfsuid = i_uid_into_vfsuid(idmap, inode);
1426 
1427 	if (vfsuid_eq(i_vfsuid, dir_vfsuid))
1428 		return 0;
1429 
1430 	if (vfsuid_eq_kuid(i_vfsuid, current_fsuid()))
1431 		return 0;
1432 
1433 	if (likely(dir_mode & 0002)) {
1434 		audit_log_path_denied(AUDIT_ANOM_CREAT, "sticky_create");
1435 		return -EACCES;
1436 	}
1437 
1438 	if (dir_mode & 0020) {
1439 		if (sysctl_protected_fifos >= 2 && S_ISFIFO(inode->i_mode)) {
1440 			audit_log_path_denied(AUDIT_ANOM_CREAT,
1441 					      "sticky_create_fifo");
1442 			return -EACCES;
1443 		}
1444 
1445 		if (sysctl_protected_regular >= 2 && S_ISREG(inode->i_mode)) {
1446 			audit_log_path_denied(AUDIT_ANOM_CREAT,
1447 					      "sticky_create_regular");
1448 			return -EACCES;
1449 		}
1450 	}
1451 
1452 	return 0;
1453 }
1454 
1455 /*
1456  * follow_up - Find the mountpoint of path's vfsmount
1457  *
1458  * Given a path, find the mountpoint of its source file system.
1459  * Replace @path with the path of the mountpoint in the parent mount.
1460  * Up is towards /.
1461  *
1462  * Return 1 if we went up a level and 0 if we were already at the
1463  * root.
1464  */
1465 int follow_up(struct path *path)
1466 {
1467 	struct mount *mnt = real_mount(path->mnt);
1468 	struct mount *parent;
1469 	struct dentry *mountpoint;
1470 
1471 	read_seqlock_excl(&mount_lock);
1472 	parent = mnt->mnt_parent;
1473 	if (parent == mnt) {
1474 		read_sequnlock_excl(&mount_lock);
1475 		return 0;
1476 	}
1477 	mntget(&parent->mnt);
1478 	mountpoint = dget(mnt->mnt_mountpoint);
1479 	read_sequnlock_excl(&mount_lock);
1480 	dput(path->dentry);
1481 	path->dentry = mountpoint;
1482 	mntput(path->mnt);
1483 	path->mnt = &parent->mnt;
1484 	return 1;
1485 }
1486 EXPORT_SYMBOL(follow_up);
1487 
1488 static bool choose_mountpoint_rcu(struct mount *m, const struct path *root,
1489 				  struct path *path, unsigned *seqp)
1490 {
1491 	while (mnt_has_parent(m)) {
1492 		struct dentry *mountpoint = m->mnt_mountpoint;
1493 
1494 		m = m->mnt_parent;
1495 		if (unlikely(root->dentry == mountpoint &&
1496 			     root->mnt == &m->mnt))
1497 			break;
1498 		if (mountpoint != m->mnt.mnt_root) {
1499 			path->mnt = &m->mnt;
1500 			path->dentry = mountpoint;
1501 			*seqp = read_seqcount_begin(&mountpoint->d_seq);
1502 			return true;
1503 		}
1504 	}
1505 	return false;
1506 }
1507 
1508 static bool choose_mountpoint(struct mount *m, const struct path *root,
1509 			      struct path *path)
1510 {
1511 	bool found;
1512 
1513 	rcu_read_lock();
1514 	while (1) {
1515 		unsigned seq, mseq = read_seqbegin(&mount_lock);
1516 
1517 		found = choose_mountpoint_rcu(m, root, path, &seq);
1518 		if (unlikely(!found)) {
1519 			if (!read_seqretry(&mount_lock, mseq))
1520 				break;
1521 		} else {
1522 			if (likely(__legitimize_path(path, seq, mseq)))
1523 				break;
1524 			rcu_read_unlock();
1525 			path_put(path);
1526 			rcu_read_lock();
1527 		}
1528 	}
1529 	rcu_read_unlock();
1530 	return found;
1531 }
1532 
1533 /*
1534  * Perform an automount
1535  * - return -EISDIR to tell follow_managed() to stop and return the path we
1536  *   were called with.
1537  */
1538 static int follow_automount(struct path *path, int *count, unsigned lookup_flags)
1539 {
1540 	struct dentry *dentry = path->dentry;
1541 
1542 	/* We don't want to mount if someone's just doing a stat -
1543 	 * unless they're stat'ing a directory and appended a '/' to
1544 	 * the name.
1545 	 *
1546 	 * We do, however, want to mount if someone wants to open or
1547 	 * create a file of any type under the mountpoint, wants to
1548 	 * traverse through the mountpoint or wants to open the
1549 	 * mounted directory.  Also, autofs may mark negative dentries
1550 	 * as being automount points.  These will need the attentions
1551 	 * of the daemon to instantiate them before they can be used.
1552 	 */
1553 	if (!(lookup_flags & (LOOKUP_PARENT | LOOKUP_DIRECTORY |
1554 			   LOOKUP_OPEN | LOOKUP_CREATE | LOOKUP_AUTOMOUNT)) &&
1555 	    dentry->d_inode)
1556 		return -EISDIR;
1557 
1558 	/* No need to trigger automounts if mountpoint crossing is disabled. */
1559 	if (lookup_flags & LOOKUP_NO_XDEV)
1560 		return -EXDEV;
1561 
1562 	if (count && (*count)++ >= MAXSYMLINKS)
1563 		return -ELOOP;
1564 
1565 	return finish_automount(dentry->d_op->d_automount(path), path);
1566 }
1567 
1568 /*
1569  * mount traversal - out-of-line part.  One note on ->d_flags accesses -
1570  * dentries are pinned but not locked here, so negative dentry can go
1571  * positive right under us.  Use of smp_load_acquire() provides a barrier
1572  * sufficient for ->d_inode and ->d_flags consistency.
1573  */
1574 static int __traverse_mounts(struct path *path, unsigned flags, bool *jumped,
1575 			     int *count, unsigned lookup_flags)
1576 {
1577 	struct vfsmount *mnt = path->mnt;
1578 	bool need_mntput = false;
1579 	int ret = 0;
1580 
1581 	while (flags & DCACHE_MANAGED_DENTRY) {
1582 		/* Allow the filesystem to manage the transit without i_rwsem
1583 		 * being held. */
1584 		if (flags & DCACHE_MANAGE_TRANSIT) {
1585 			if (lookup_flags & LOOKUP_NO_XDEV) {
1586 				ret = -EXDEV;
1587 				break;
1588 			}
1589 			ret = path->dentry->d_op->d_manage(path, false);
1590 			flags = smp_load_acquire(&path->dentry->d_flags);
1591 			if (ret < 0)
1592 				break;
1593 		}
1594 
1595 		if (flags & DCACHE_MOUNTED) {	// something's mounted on it..
1596 			struct vfsmount *mounted = lookup_mnt(path);
1597 			if (mounted) {		// ... in our namespace
1598 				dput(path->dentry);
1599 				if (need_mntput)
1600 					mntput(path->mnt);
1601 				path->mnt = mounted;
1602 				path->dentry = dget(mounted->mnt_root);
1603 				// here we know it's positive
1604 				flags = path->dentry->d_flags;
1605 				need_mntput = true;
1606 				if (unlikely(lookup_flags & LOOKUP_NO_XDEV)) {
1607 					ret = -EXDEV;
1608 					break;
1609 				}
1610 				continue;
1611 			}
1612 		}
1613 
1614 		if (!(flags & DCACHE_NEED_AUTOMOUNT))
1615 			break;
1616 
1617 		// uncovered automount point
1618 		ret = follow_automount(path, count, lookup_flags);
1619 		flags = smp_load_acquire(&path->dentry->d_flags);
1620 		if (ret < 0)
1621 			break;
1622 	}
1623 
1624 	if (ret == -EISDIR)
1625 		ret = 0;
1626 	// possible if you race with several mount --move
1627 	if (need_mntput && path->mnt == mnt)
1628 		mntput(path->mnt);
1629 	if (!ret && unlikely(d_flags_negative(flags)))
1630 		ret = -ENOENT;
1631 	*jumped = need_mntput;
1632 	return ret;
1633 }
1634 
1635 static inline int traverse_mounts(struct path *path, bool *jumped,
1636 				  int *count, unsigned lookup_flags)
1637 {
1638 	unsigned flags = smp_load_acquire(&path->dentry->d_flags);
1639 
1640 	/* fastpath */
1641 	if (likely(!(flags & DCACHE_MANAGED_DENTRY))) {
1642 		*jumped = false;
1643 		if (unlikely(d_flags_negative(flags)))
1644 			return -ENOENT;
1645 		return 0;
1646 	}
1647 	return __traverse_mounts(path, flags, jumped, count, lookup_flags);
1648 }
1649 
1650 int follow_down_one(struct path *path)
1651 {
1652 	struct vfsmount *mounted;
1653 
1654 	mounted = lookup_mnt(path);
1655 	if (mounted) {
1656 		dput(path->dentry);
1657 		mntput(path->mnt);
1658 		path->mnt = mounted;
1659 		path->dentry = dget(mounted->mnt_root);
1660 		return 1;
1661 	}
1662 	return 0;
1663 }
1664 EXPORT_SYMBOL(follow_down_one);
1665 
1666 /*
1667  * Follow down to the covering mount currently visible to userspace.  At each
1668  * point, the filesystem owning that dentry may be queried as to whether the
1669  * caller is permitted to proceed or not.
1670  */
1671 int follow_down(struct path *path, unsigned int flags)
1672 {
1673 	struct vfsmount *mnt = path->mnt;
1674 	bool jumped;
1675 	int ret = traverse_mounts(path, &jumped, NULL, flags);
1676 
1677 	if (path->mnt != mnt)
1678 		mntput(mnt);
1679 	return ret;
1680 }
1681 EXPORT_SYMBOL(follow_down);
1682 
1683 /*
1684  * Try to skip to top of mountpoint pile in rcuwalk mode.  Fail if
1685  * we meet a managed dentry that would need blocking.
1686  */
1687 static bool __follow_mount_rcu(struct nameidata *nd, struct path *path)
1688 {
1689 	struct dentry *dentry = path->dentry;
1690 	unsigned int flags = dentry->d_flags;
1691 
1692 	if (unlikely(nd->flags & LOOKUP_NO_XDEV))
1693 		return false;
1694 
1695 	for (;;) {
1696 		/*
1697 		 * Don't forget we might have a non-mountpoint managed dentry
1698 		 * that wants to block transit.
1699 		 */
1700 		if (unlikely(flags & DCACHE_MANAGE_TRANSIT)) {
1701 			int res = dentry->d_op->d_manage(path, true);
1702 			if (res)
1703 				return res == -EISDIR;
1704 			flags = dentry->d_flags;
1705 		}
1706 
1707 		if (flags & DCACHE_MOUNTED) {
1708 			struct mount *mounted = __lookup_mnt(path->mnt, dentry);
1709 			if (mounted) {
1710 				path->mnt = &mounted->mnt;
1711 				dentry = path->dentry = mounted->mnt.mnt_root;
1712 				nd->state |= ND_JUMPED;
1713 				nd->next_seq = read_seqcount_begin(&dentry->d_seq);
1714 				flags = dentry->d_flags;
1715 				// makes sure that non-RCU pathwalk could reach
1716 				// this state.
1717 				if (read_seqretry(&mount_lock, nd->m_seq))
1718 					return false;
1719 				continue;
1720 			}
1721 			if (read_seqretry(&mount_lock, nd->m_seq))
1722 				return false;
1723 		}
1724 		return !(flags & DCACHE_NEED_AUTOMOUNT);
1725 	}
1726 }
1727 
1728 static inline int handle_mounts(struct nameidata *nd, struct dentry *dentry,
1729 			  struct path *path)
1730 {
1731 	bool jumped;
1732 	int ret;
1733 
1734 	path->mnt = nd->path.mnt;
1735 	path->dentry = dentry;
1736 	if (nd->flags & LOOKUP_RCU) {
1737 		unsigned int seq = nd->next_seq;
1738 		if (likely(!d_managed(dentry)))
1739 			return 0;
1740 		if (likely(__follow_mount_rcu(nd, path)))
1741 			return 0;
1742 		// *path and nd->next_seq might've been clobbered
1743 		path->mnt = nd->path.mnt;
1744 		path->dentry = dentry;
1745 		nd->next_seq = seq;
1746 		if (unlikely(!try_to_unlazy_next(nd, dentry)))
1747 			return -ECHILD;
1748 	}
1749 	ret = traverse_mounts(path, &jumped, &nd->total_link_count, nd->flags);
1750 	if (jumped)
1751 		nd->state |= ND_JUMPED;
1752 	if (unlikely(ret)) {
1753 		dput(path->dentry);
1754 		if (path->mnt != nd->path.mnt)
1755 			mntput(path->mnt);
1756 	}
1757 	return ret;
1758 }
1759 
1760 /*
1761  * This looks up the name in dcache and possibly revalidates the found dentry.
1762  * NULL is returned if the dentry does not exist in the cache.
1763  */
1764 static struct dentry *lookup_dcache(const struct qstr *name,
1765 				    struct dentry *dir,
1766 				    unsigned int flags)
1767 {
1768 	struct dentry *dentry = d_lookup(dir, name);
1769 	if (dentry) {
1770 		int error = d_revalidate(dir->d_inode, name, dentry, flags);
1771 		if (unlikely(error <= 0)) {
1772 			if (!error)
1773 				d_invalidate(dentry);
1774 			dput(dentry);
1775 			return ERR_PTR(error);
1776 		}
1777 	}
1778 	return dentry;
1779 }
1780 
1781 /*
1782  * Parent directory has inode locked exclusive.  This is one
1783  * and only case when ->lookup() gets called on non in-lookup
1784  * dentries - as the matter of fact, this only gets called
1785  * when directory is guaranteed to have no in-lookup children
1786  * at all.
1787  * Will return -ENOENT if name isn't found and LOOKUP_CREATE wasn't passed.
1788  * Will return -EEXIST if name is found and LOOKUP_EXCL was passed.
1789  */
1790 static struct dentry *lookup_one_qstr_excl(const struct qstr *name,
1791 					   struct dentry *base, unsigned int flags)
1792 {
1793 	struct dentry *dentry;
1794 	struct dentry *old;
1795 	struct inode *dir;
1796 
1797 	dentry = lookup_dcache(name, base, flags);
1798 	if (dentry)
1799 		goto found;
1800 
1801 	/* Don't create child dentry for a dead directory. */
1802 	dir = base->d_inode;
1803 	if (unlikely(IS_DEADDIR(dir)))
1804 		return ERR_PTR(-ENOENT);
1805 
1806 	dentry = d_alloc(base, name);
1807 	if (unlikely(!dentry))
1808 		return ERR_PTR(-ENOMEM);
1809 
1810 	old = dir->i_op->lookup(dir, dentry, flags);
1811 	if (unlikely(old)) {
1812 		dput(dentry);
1813 		dentry = old;
1814 	}
1815 found:
1816 	if (IS_ERR(dentry))
1817 		return dentry;
1818 	if (d_is_negative(dentry) && !(flags & LOOKUP_CREATE)) {
1819 		dput(dentry);
1820 		return ERR_PTR(-ENOENT);
1821 	}
1822 	if (d_is_positive(dentry) && (flags & LOOKUP_EXCL)) {
1823 		dput(dentry);
1824 		return ERR_PTR(-EEXIST);
1825 	}
1826 	return dentry;
1827 }
1828 
1829 /**
1830  * lookup_fast - do fast lockless (but racy) lookup of a dentry
1831  * @nd: current nameidata
1832  *
1833  * Do a fast, but racy lookup in the dcache for the given dentry, and
1834  * revalidate it. Returns a valid dentry pointer or NULL if one wasn't
1835  * found. On error, an ERR_PTR will be returned.
1836  *
1837  * If this function returns a valid dentry and the walk is no longer
1838  * lazy, the dentry will carry a reference that must later be put. If
1839  * RCU mode is still in force, then this is not the case and the dentry
1840  * must be legitimized before use. If this returns NULL, then the walk
1841  * will no longer be in RCU mode.
1842  */
1843 static struct dentry *lookup_fast(struct nameidata *nd)
1844 {
1845 	struct dentry *dentry, *parent = nd->path.dentry;
1846 	int status = 1;
1847 
1848 	/*
1849 	 * Rename seqlock is not required here because in the off chance
1850 	 * of a false negative due to a concurrent rename, the caller is
1851 	 * going to fall back to non-racy lookup.
1852 	 */
1853 	if (nd->flags & LOOKUP_RCU) {
1854 		dentry = __d_lookup_rcu(parent, &nd->last, &nd->next_seq);
1855 		if (unlikely(!dentry)) {
1856 			if (!try_to_unlazy(nd))
1857 				return ERR_PTR(-ECHILD);
1858 			return NULL;
1859 		}
1860 
1861 		/*
1862 		 * This sequence count validates that the parent had no
1863 		 * changes while we did the lookup of the dentry above.
1864 		 */
1865 		if (read_seqcount_retry(&parent->d_seq, nd->seq))
1866 			return ERR_PTR(-ECHILD);
1867 
1868 		status = d_revalidate(nd->inode, &nd->last, dentry, nd->flags);
1869 		if (likely(status > 0))
1870 			return dentry;
1871 		if (!try_to_unlazy_next(nd, dentry))
1872 			return ERR_PTR(-ECHILD);
1873 		if (status == -ECHILD)
1874 			/* we'd been told to redo it in non-rcu mode */
1875 			status = d_revalidate(nd->inode, &nd->last,
1876 					      dentry, nd->flags);
1877 	} else {
1878 		dentry = __d_lookup(parent, &nd->last);
1879 		if (unlikely(!dentry))
1880 			return NULL;
1881 		status = d_revalidate(nd->inode, &nd->last, dentry, nd->flags);
1882 	}
1883 	if (unlikely(status <= 0)) {
1884 		if (!status)
1885 			d_invalidate(dentry);
1886 		dput(dentry);
1887 		return ERR_PTR(status);
1888 	}
1889 	return dentry;
1890 }
1891 
1892 /* Fast lookup failed, do it the slow way */
1893 static struct dentry *__lookup_slow(const struct qstr *name,
1894 				    struct dentry *dir,
1895 				    unsigned int flags)
1896 {
1897 	struct dentry *dentry, *old;
1898 	struct inode *inode = dir->d_inode;
1899 
1900 	/* Don't go there if it's already dead */
1901 	if (unlikely(IS_DEADDIR(inode)))
1902 		return ERR_PTR(-ENOENT);
1903 again:
1904 	dentry = d_alloc_parallel(dir, name);
1905 	if (IS_ERR(dentry))
1906 		return dentry;
1907 	if (unlikely(!d_in_lookup(dentry))) {
1908 		int error = d_revalidate(inode, name, dentry, flags);
1909 		if (unlikely(error <= 0)) {
1910 			if (!error) {
1911 				d_invalidate(dentry);
1912 				dput(dentry);
1913 				goto again;
1914 			}
1915 			dput(dentry);
1916 			dentry = ERR_PTR(error);
1917 		}
1918 	} else {
1919 		old = inode->i_op->lookup(inode, dentry, flags);
1920 		d_lookup_done(dentry);
1921 		if (unlikely(old)) {
1922 			dput(dentry);
1923 			dentry = old;
1924 		}
1925 	}
1926 	return dentry;
1927 }
1928 
1929 static noinline struct dentry *lookup_slow(const struct qstr *name,
1930 				  struct dentry *dir,
1931 				  unsigned int flags)
1932 {
1933 	struct inode *inode = dir->d_inode;
1934 	struct dentry *res;
1935 	inode_lock_shared(inode);
1936 	res = __lookup_slow(name, dir, flags);
1937 	inode_unlock_shared(inode);
1938 	return res;
1939 }
1940 
1941 static struct dentry *lookup_slow_killable(const struct qstr *name,
1942 					   struct dentry *dir,
1943 					   unsigned int flags)
1944 {
1945 	struct inode *inode = dir->d_inode;
1946 	struct dentry *res;
1947 
1948 	if (inode_lock_shared_killable(inode))
1949 		return ERR_PTR(-EINTR);
1950 	res = __lookup_slow(name, dir, flags);
1951 	inode_unlock_shared(inode);
1952 	return res;
1953 }
1954 
1955 static inline int may_lookup(struct mnt_idmap *idmap,
1956 			     struct nameidata *restrict nd)
1957 {
1958 	int err, mask;
1959 
1960 	mask = nd->flags & LOOKUP_RCU ? MAY_NOT_BLOCK : 0;
1961 	err = lookup_inode_permission_may_exec(idmap, nd->inode, mask);
1962 	if (likely(!err))
1963 		return 0;
1964 
1965 	// If we failed, and we weren't in LOOKUP_RCU, it's final
1966 	if (!(nd->flags & LOOKUP_RCU))
1967 		return err;
1968 
1969 	// Drop out of RCU mode to make sure it wasn't transient
1970 	if (!try_to_unlazy(nd))
1971 		return -ECHILD;	// redo it all non-lazy
1972 
1973 	if (err != -ECHILD)	// hard error
1974 		return err;
1975 
1976 	return lookup_inode_permission_may_exec(idmap, nd->inode, 0);
1977 }
1978 
1979 static int reserve_stack(struct nameidata *nd, struct path *link)
1980 {
1981 	if (unlikely(nd->total_link_count++ >= MAXSYMLINKS))
1982 		return -ELOOP;
1983 
1984 	if (likely(nd->depth != EMBEDDED_LEVELS))
1985 		return 0;
1986 	if (likely(nd->stack != nd->internal))
1987 		return 0;
1988 	if (likely(nd_alloc_stack(nd)))
1989 		return 0;
1990 
1991 	if (nd->flags & LOOKUP_RCU) {
1992 		// we need to grab link before we do unlazy.  And we can't skip
1993 		// unlazy even if we fail to grab the link - cleanup needs it
1994 		bool grabbed_link = legitimize_path(nd, link, nd->next_seq);
1995 
1996 		if (!try_to_unlazy(nd) || !grabbed_link)
1997 			return -ECHILD;
1998 
1999 		if (nd_alloc_stack(nd))
2000 			return 0;
2001 	}
2002 	return -ENOMEM;
2003 }
2004 
2005 enum {WALK_TRAILING = 1, WALK_MORE = 2, WALK_NOFOLLOW = 4};
2006 
2007 static noinline const char *pick_link(struct nameidata *nd, struct path *link,
2008 		     struct inode *inode, int flags)
2009 {
2010 	struct saved *last;
2011 	const char *res;
2012 	int error;
2013 
2014 	if (nd->flags & LOOKUP_RCU) {
2015 		/* make sure that d_is_symlink from step_into_slowpath() matches the inode */
2016 		if (read_seqcount_retry(&link->dentry->d_seq, nd->next_seq))
2017 			return ERR_PTR(-ECHILD);
2018 	} else {
2019 		if (link->mnt == nd->path.mnt)
2020 			mntget(link->mnt);
2021 	}
2022 
2023 	error = reserve_stack(nd, link);
2024 	if (unlikely(error)) {
2025 		if (!(nd->flags & LOOKUP_RCU))
2026 			path_put(link);
2027 		return ERR_PTR(error);
2028 	}
2029 	last = nd->stack + nd->depth++;
2030 	last->link = *link;
2031 	clear_delayed_call(&last->done);
2032 	last->seq = nd->next_seq;
2033 
2034 	if (flags & WALK_TRAILING) {
2035 		error = may_follow_link(nd, inode);
2036 		if (unlikely(error))
2037 			return ERR_PTR(error);
2038 	}
2039 
2040 	if (unlikely(nd->flags & LOOKUP_NO_SYMLINKS) ||
2041 			unlikely(link->mnt->mnt_flags & MNT_NOSYMFOLLOW))
2042 		return ERR_PTR(-ELOOP);
2043 
2044 	if (unlikely(atime_needs_update(&last->link, inode))) {
2045 		if (nd->flags & LOOKUP_RCU) {
2046 			if (!try_to_unlazy(nd))
2047 				return ERR_PTR(-ECHILD);
2048 		}
2049 		touch_atime(&last->link);
2050 		cond_resched();
2051 	}
2052 
2053 	error = security_inode_follow_link(link->dentry, inode,
2054 					   nd->flags & LOOKUP_RCU);
2055 	if (unlikely(error))
2056 		return ERR_PTR(error);
2057 
2058 	res = READ_ONCE(inode->i_link);
2059 	if (!res) {
2060 		const char * (*get)(struct dentry *, struct inode *,
2061 				struct delayed_call *);
2062 		get = inode->i_op->get_link;
2063 		if (nd->flags & LOOKUP_RCU) {
2064 			res = get(NULL, inode, &last->done);
2065 			if (res == ERR_PTR(-ECHILD) && try_to_unlazy(nd))
2066 				res = get(link->dentry, inode, &last->done);
2067 		} else {
2068 			res = get(link->dentry, inode, &last->done);
2069 		}
2070 		if (!res)
2071 			goto all_done;
2072 		if (IS_ERR(res))
2073 			return res;
2074 	}
2075 	if (*res == '/') {
2076 		error = nd_jump_root(nd);
2077 		if (unlikely(error))
2078 			return ERR_PTR(error);
2079 		while (unlikely(*++res == '/'))
2080 			;
2081 	}
2082 	if (*res)
2083 		return res;
2084 all_done: // pure jump
2085 	put_link(nd);
2086 	return NULL;
2087 }
2088 
2089 /*
2090  * Be careful in case the dentry is unlinked or renamed. Any mounts
2091  * stacked on top of it are going away. We need to make sure that we
2092  * don't reveal the underyling dentry during refwalk. In rcuwalk we
2093  * catch this via d_seq and another lookup for the name. Give the same
2094  * guarantee in refwalk.
2095  */
2096 static bool unlink_may_reveal(struct nameidata *nd, int flags,
2097 			      struct path *path, struct dentry *dentry)
2098 {
2099 	/* ".." and LOOKUP_DOWN may land on an unhashed directory */
2100 	if (flags & WALK_NOFOLLOW)
2101 		return false;
2102 	if (nd->flags & LOOKUP_REVAL)
2103 		return false;
2104 	/* We crossed onto a mount and the name led us here while it still existed */
2105 	if (path->mnt != nd->path.mnt)
2106 		return false;
2107 	/* only a name on its way out is flagged */
2108 	if (likely(!cant_mount(dentry)))
2109 		return false;
2110 	if (!d_unlinked(dentry))
2111 		return false;
2112 	dput(no_free_ptr(path->dentry));
2113 	return true;
2114 }
2115 
2116 /*
2117  * Do we need to follow links? We _really_ want to be able
2118  * to do this check without having to look at inode->i_op,
2119  * so we keep a cache of "no, this doesn't need follow_link"
2120  * for the common case.
2121  *
2122  * NOTE: dentry must be what nd->next_seq had been sampled from.
2123  */
2124 static noinline const char *step_into_slowpath(struct nameidata *nd, int flags,
2125 		     struct dentry *dentry)
2126 {
2127 	struct path path;
2128 	struct inode *inode;
2129 	int err;
2130 
2131 	err = handle_mounts(nd, dentry, &path);
2132 	if (unlikely(err < 0))
2133 		return ERR_PTR(err);
2134 	inode = path.dentry->d_inode;
2135 	if (likely(!d_is_symlink(path.dentry)) ||
2136 	   ((flags & WALK_TRAILING) && !(nd->flags & LOOKUP_FOLLOW)) ||
2137 	   (flags & WALK_NOFOLLOW)) {
2138 		/* not a symlink or should not follow */
2139 		if (nd->flags & LOOKUP_RCU) {
2140 			if (read_seqcount_retry(&path.dentry->d_seq, nd->next_seq))
2141 				return ERR_PTR(-ECHILD);
2142 			if (unlikely(!inode))
2143 				return ERR_PTR(-ENOENT);
2144 		} else {
2145 			if (unlikely(unlink_may_reveal(nd, flags, &path, dentry)))
2146 				return ERR_PTR(-ESTALE);
2147 			dput(nd->path.dentry);
2148 			if (nd->path.mnt != path.mnt)
2149 				mntput(nd->path.mnt);
2150 		}
2151 		nd->path = path;
2152 		nd->inode = inode;
2153 		nd->seq = nd->next_seq;
2154 		return NULL;
2155 	}
2156 	return pick_link(nd, &path, inode, flags);
2157 }
2158 
2159 static __always_inline const char *step_into(struct nameidata *nd, int flags,
2160                     struct dentry *dentry)
2161 {
2162 	/*
2163 	 * In the common case we are in rcu-walk and traversing over a non-mounted on
2164 	 * directory (as opposed to e.g., a symlink).
2165 	 *
2166 	 * We can handle that and negative entries with the checks below.
2167 	 */
2168 	if (likely((nd->flags & LOOKUP_RCU) &&
2169 	    !d_managed(dentry) && !d_is_symlink(dentry))) {
2170 		struct inode *inode = dentry->d_inode;
2171 		if (read_seqcount_retry(&dentry->d_seq, nd->next_seq))
2172 			return ERR_PTR(-ECHILD);
2173 		if (unlikely(!inode))
2174 			return ERR_PTR(-ENOENT);
2175 		nd->path.dentry = dentry;
2176 		/* nd->path.mnt remains unchanged as no mount point was crossed */
2177 		nd->inode = inode;
2178 		nd->seq = nd->next_seq;
2179 		return NULL;
2180 	}
2181 	return step_into_slowpath(nd, flags, dentry);
2182 }
2183 
2184 static struct dentry *follow_dotdot_rcu(struct nameidata *nd)
2185 {
2186 	struct dentry *parent, *old;
2187 
2188 	if (path_equal(&nd->path, &nd->root))
2189 		goto in_root;
2190 	if (unlikely(nd->path.dentry == nd->path.mnt->mnt_root)) {
2191 		struct path path;
2192 		unsigned seq;
2193 		if (!choose_mountpoint_rcu(real_mount(nd->path.mnt),
2194 					   &nd->root, &path, &seq))
2195 			goto in_root;
2196 		if (unlikely(nd->flags & LOOKUP_NO_XDEV))
2197 			return ERR_PTR(-ECHILD);
2198 		nd->path = path;
2199 		nd->inode = path.dentry->d_inode;
2200 		nd->seq = seq;
2201 		// makes sure that non-RCU pathwalk could reach this state
2202 		if (read_seqretry(&mount_lock, nd->m_seq))
2203 			return ERR_PTR(-ECHILD);
2204 		/* we know that mountpoint was pinned */
2205 	}
2206 	old = nd->path.dentry;
2207 	parent = old->d_parent;
2208 	nd->next_seq = read_seqcount_begin(&parent->d_seq);
2209 	// makes sure that non-RCU pathwalk could reach this state
2210 	if (read_seqcount_retry(&old->d_seq, nd->seq))
2211 		return ERR_PTR(-ECHILD);
2212 	if (unlikely(!path_connected(nd->path.mnt, parent)))
2213 		return ERR_PTR(-ECHILD);
2214 	return parent;
2215 in_root:
2216 	if (read_seqretry(&mount_lock, nd->m_seq))
2217 		return ERR_PTR(-ECHILD);
2218 	if (unlikely(nd->flags & LOOKUP_BENEATH))
2219 		return ERR_PTR(-ECHILD);
2220 	nd->next_seq = nd->seq;
2221 	return nd->path.dentry;
2222 }
2223 
2224 static struct dentry *follow_dotdot(struct nameidata *nd)
2225 {
2226 	struct dentry *parent;
2227 
2228 	if (path_equal(&nd->path, &nd->root))
2229 		goto in_root;
2230 	if (unlikely(nd->path.dentry == nd->path.mnt->mnt_root)) {
2231 		struct path path;
2232 
2233 		if (!choose_mountpoint(real_mount(nd->path.mnt),
2234 				       &nd->root, &path))
2235 			goto in_root;
2236 		path_put(&nd->path);
2237 		nd->path = path;
2238 		nd->inode = path.dentry->d_inode;
2239 		if (unlikely(nd->flags & LOOKUP_NO_XDEV))
2240 			return ERR_PTR(-EXDEV);
2241 	}
2242 	/* rare case of legitimate dget_parent()... */
2243 	parent = dget_parent(nd->path.dentry);
2244 	if (unlikely(!path_connected(nd->path.mnt, parent))) {
2245 		dput(parent);
2246 		return ERR_PTR(-ENOENT);
2247 	}
2248 	return parent;
2249 
2250 in_root:
2251 	if (unlikely(nd->flags & LOOKUP_BENEATH))
2252 		return ERR_PTR(-EXDEV);
2253 	return dget(nd->path.dentry);
2254 }
2255 
2256 static const char *handle_dots(struct nameidata *nd, enum last_type type)
2257 {
2258 	if (type == LAST_DOTDOT) {
2259 		const char *error = NULL;
2260 		struct dentry *parent;
2261 
2262 		if (!nd->root.mnt) {
2263 			error = ERR_PTR(set_root(nd));
2264 			if (unlikely(error))
2265 				return error;
2266 		}
2267 		if (nd->flags & LOOKUP_RCU)
2268 			parent = follow_dotdot_rcu(nd);
2269 		else
2270 			parent = follow_dotdot(nd);
2271 		if (IS_ERR(parent))
2272 			return ERR_CAST(parent);
2273 		error = step_into(nd, WALK_NOFOLLOW, parent);
2274 		if (unlikely(error))
2275 			return error;
2276 
2277 		if (unlikely(nd->flags & LOOKUP_IS_SCOPED)) {
2278 			/*
2279 			 * If there was a racing rename or mount along our
2280 			 * path, then we can't be sure that ".." hasn't jumped
2281 			 * above nd->root (and so userspace should retry or use
2282 			 * some fallback).
2283 			 */
2284 			smp_rmb();
2285 			if (__read_seqcount_retry(&mount_lock.seqcount, nd->m_seq))
2286 				return ERR_PTR(-EAGAIN);
2287 			if (__read_seqcount_retry(&rename_lock.seqcount, nd->r_seq))
2288 				return ERR_PTR(-EAGAIN);
2289 		}
2290 	}
2291 	return NULL;
2292 }
2293 
2294 static __always_inline const char *walk_component(struct nameidata *nd, int flags)
2295 {
2296 	struct dentry *dentry;
2297 	/*
2298 	 * "." and ".." are special - ".." especially so because it has
2299 	 * to be able to know about the current root directory and
2300 	 * parent relationships.
2301 	 */
2302 	if (unlikely(nd->last_type != LAST_NORM)) {
2303 		if (unlikely(nd->depth) && !(flags & WALK_MORE))
2304 			put_link(nd);
2305 		return handle_dots(nd, nd->last_type);
2306 	}
2307 	dentry = lookup_fast(nd);
2308 	if (IS_ERR(dentry))
2309 		return ERR_CAST(dentry);
2310 	if (unlikely(!dentry)) {
2311 		dentry = lookup_slow(&nd->last, nd->path.dentry, nd->flags);
2312 		if (IS_ERR(dentry))
2313 			return ERR_CAST(dentry);
2314 	}
2315 	if (unlikely(nd->depth) && !(flags & WALK_MORE))
2316 		put_link(nd);
2317 	return step_into(nd, flags, dentry);
2318 }
2319 
2320 /*
2321  * We can do the critical dentry name comparison and hashing
2322  * operations one word at a time, but we are limited to:
2323  *
2324  * - Architectures with fast unaligned word accesses. We could
2325  *   do a "get_unaligned()" if this helps and is sufficiently
2326  *   fast.
2327  *
2328  * - non-CONFIG_DEBUG_PAGEALLOC configurations (so that we
2329  *   do not trap on the (extremely unlikely) case of a page
2330  *   crossing operation.
2331  *
2332  * - Furthermore, we need an efficient 64-bit compile for the
2333  *   64-bit case in order to generate the "number of bytes in
2334  *   the final mask". Again, that could be replaced with a
2335  *   efficient population count instruction or similar.
2336  */
2337 #ifdef CONFIG_DCACHE_WORD_ACCESS
2338 
2339 #include <asm/word-at-a-time.h>
2340 
2341 #ifdef HASH_MIX
2342 
2343 /* Architecture provides HASH_MIX and fold_hash() in <asm/hash.h> */
2344 
2345 #elif defined(CONFIG_64BIT)
2346 /*
2347  * Register pressure in the mixing function is an issue, particularly
2348  * on 32-bit x86, but almost any function requires one state value and
2349  * one temporary.  Instead, use a function designed for two state values
2350  * and no temporaries.
2351  *
2352  * This function cannot create a collision in only two iterations, so
2353  * we have two iterations to achieve avalanche.  In those two iterations,
2354  * we have six layers of mixing, which is enough to spread one bit's
2355  * influence out to 2^6 = 64 state bits.
2356  *
2357  * Rotate constants are scored by considering either 64 one-bit input
2358  * deltas or 64*63/2 = 2016 two-bit input deltas, and finding the
2359  * probability of that delta causing a change to each of the 128 output
2360  * bits, using a sample of random initial states.
2361  *
2362  * The Shannon entropy of the computed probabilities is then summed
2363  * to produce a score.  Ideally, any input change has a 50% chance of
2364  * toggling any given output bit.
2365  *
2366  * Mixing scores (in bits) for (12,45):
2367  * Input delta: 1-bit      2-bit
2368  * 1 round:     713.3    42542.6
2369  * 2 rounds:   2753.7   140389.8
2370  * 3 rounds:   5954.1   233458.2
2371  * 4 rounds:   7862.6   256672.2
2372  * Perfect:    8192     258048
2373  *            (64*128) (64*63/2 * 128)
2374  */
2375 #define HASH_MIX(x, y, a)	\
2376 	(	x ^= (a),	\
2377 	y ^= x,	x = rol64(x,12),\
2378 	x += y,	y = rol64(y,45),\
2379 	y *= 9			)
2380 
2381 /*
2382  * Fold two longs into one 32-bit hash value.  This must be fast, but
2383  * latency isn't quite as critical, as there is a fair bit of additional
2384  * work done before the hash value is used.
2385  */
2386 static inline unsigned int fold_hash(unsigned long x, unsigned long y)
2387 {
2388 	y ^= x * GOLDEN_RATIO_64;
2389 	y *= GOLDEN_RATIO_64;
2390 	return y >> 32;
2391 }
2392 
2393 #else	/* 32-bit case */
2394 
2395 /*
2396  * Mixing scores (in bits) for (7,20):
2397  * Input delta: 1-bit      2-bit
2398  * 1 round:     330.3     9201.6
2399  * 2 rounds:   1246.4    25475.4
2400  * 3 rounds:   1907.1    31295.1
2401  * 4 rounds:   2042.3    31718.6
2402  * Perfect:    2048      31744
2403  *            (32*64)   (32*31/2 * 64)
2404  */
2405 #define HASH_MIX(x, y, a)	\
2406 	(	x ^= (a),	\
2407 	y ^= x,	x = rol32(x, 7),\
2408 	x += y,	y = rol32(y,20),\
2409 	y *= 9			)
2410 
2411 static inline unsigned int fold_hash(unsigned long x, unsigned long y)
2412 {
2413 	/* Use arch-optimized multiply if one exists */
2414 	return __hash_32(y ^ __hash_32(x));
2415 }
2416 
2417 #endif
2418 
2419 /*
2420  * Return the hash of a string of known length.  This is carfully
2421  * designed to match hash_name(), which is the more critical function.
2422  * In particular, we must end by hashing a final word containing 0..7
2423  * payload bytes, to match the way that hash_name() iterates until it
2424  * finds the delimiter after the name.
2425  */
2426 unsigned int full_name_hash(const void *salt, const char *name, unsigned int len)
2427 {
2428 	unsigned long a, x = 0, y = (unsigned long)salt;
2429 
2430 	for (;;) {
2431 		if (!len)
2432 			goto done;
2433 		a = load_unaligned_zeropad(name);
2434 		if (len < sizeof(unsigned long))
2435 			break;
2436 		HASH_MIX(x, y, a);
2437 		name += sizeof(unsigned long);
2438 		len -= sizeof(unsigned long);
2439 	}
2440 	x ^= a & bytemask_from_count(len);
2441 done:
2442 	return fold_hash(x, y);
2443 }
2444 EXPORT_SYMBOL(full_name_hash);
2445 
2446 /* Return the "hash_len" (hash and length) of a null-terminated string */
2447 u64 hashlen_string(const void *salt, const char *name)
2448 {
2449 	unsigned long a = 0, x = 0, y = (unsigned long)salt;
2450 	unsigned long adata, mask, len;
2451 	const struct word_at_a_time constants = WORD_AT_A_TIME_CONSTANTS;
2452 
2453 	len = 0;
2454 	goto inside;
2455 
2456 	do {
2457 		HASH_MIX(x, y, a);
2458 		len += sizeof(unsigned long);
2459 inside:
2460 		a = load_unaligned_zeropad(name+len);
2461 	} while (!has_zero(a, &adata, &constants));
2462 
2463 	adata = prep_zero_mask(a, adata, &constants);
2464 	mask = create_zero_mask(adata);
2465 	x ^= a & zero_bytemask(mask);
2466 
2467 	return hashlen_create(fold_hash(x, y), len + find_zero(mask));
2468 }
2469 EXPORT_SYMBOL(hashlen_string);
2470 
2471 /*
2472  * hash_name - Calculate the length and hash of the path component
2473  * @nd: the path resolution state
2474  * @name: the pathname to read the component from
2475  * @lastword: if the component fits in a single word, LAST_WORD_IS_DOT,
2476  * LAST_WORD_IS_DOTDOT, or some other value depending on whether the
2477  * component is '.', '..', or something else. Otherwise, @lastword is 0.
2478  *
2479  * Returns: a pointer to the terminating '/' or NUL character in @name.
2480  */
2481 static inline const char *hash_name(struct nameidata *nd,
2482 				    const char *name,
2483 				    unsigned long *lastword)
2484 {
2485 	unsigned long a, b, x, y = (unsigned long)nd->path.dentry;
2486 	unsigned long adata, bdata, mask, len;
2487 	const struct word_at_a_time constants = WORD_AT_A_TIME_CONSTANTS;
2488 
2489 	/*
2490 	 * The first iteration is special, because it can result in
2491 	 * '.' and '..' and has no mixing other than the final fold.
2492 	 */
2493 	a = load_unaligned_zeropad(name);
2494 	b = a ^ REPEAT_BYTE('/');
2495 	if (has_zero(a, &adata, &constants) | has_zero(b, &bdata, &constants)) {
2496 		adata = prep_zero_mask(a, adata, &constants);
2497 		bdata = prep_zero_mask(b, bdata, &constants);
2498 		mask = create_zero_mask(adata | bdata);
2499 		a &= zero_bytemask(mask);
2500 		*lastword = a;
2501 		len = find_zero(mask);
2502 		nd->last.hash = fold_hash(a, y);
2503 		nd->last.len = len;
2504 		return name + len;
2505 	}
2506 
2507 	len = 0;
2508 	x = 0;
2509 	do {
2510 		HASH_MIX(x, y, a);
2511 		len += sizeof(unsigned long);
2512 		a = load_unaligned_zeropad(name+len);
2513 		b = a ^ REPEAT_BYTE('/');
2514 	} while (!(has_zero(a, &adata, &constants) | has_zero(b, &bdata, &constants)));
2515 
2516 	adata = prep_zero_mask(a, adata, &constants);
2517 	bdata = prep_zero_mask(b, bdata, &constants);
2518 	mask = create_zero_mask(adata | bdata);
2519 	a &= zero_bytemask(mask);
2520 	x ^= a;
2521 	len += find_zero(mask);
2522 	*lastword = 0;		// Multi-word components cannot be DOT or DOTDOT
2523 
2524 	nd->last.hash = fold_hash(x, y);
2525 	nd->last.len = len;
2526 	return name + len;
2527 }
2528 
2529 /*
2530  * Note that the 'last' word is always zero-masked, but
2531  * was loaded as a possibly big-endian word.
2532  */
2533 #ifdef __BIG_ENDIAN
2534   #define LAST_WORD_IS_DOT	(0x2eul << (BITS_PER_LONG-8))
2535   #define LAST_WORD_IS_DOTDOT	(0x2e2eul << (BITS_PER_LONG-16))
2536 #endif
2537 
2538 #else	/* !CONFIG_DCACHE_WORD_ACCESS: Slow, byte-at-a-time version */
2539 
2540 /* Return the hash of a string of known length */
2541 unsigned int full_name_hash(const void *salt, const char *name, unsigned int len)
2542 {
2543 	unsigned long hash = init_name_hash(salt);
2544 	while (len--)
2545 		hash = partial_name_hash((unsigned char)*name++, hash);
2546 	return end_name_hash(hash);
2547 }
2548 EXPORT_SYMBOL(full_name_hash);
2549 
2550 /* Return the "hash_len" (hash and length) of a null-terminated string */
2551 u64 hashlen_string(const void *salt, const char *name)
2552 {
2553 	unsigned long hash = init_name_hash(salt);
2554 	unsigned long len = 0, c;
2555 
2556 	c = (unsigned char)*name;
2557 	while (c) {
2558 		len++;
2559 		hash = partial_name_hash(c, hash);
2560 		c = (unsigned char)name[len];
2561 	}
2562 	return hashlen_create(end_name_hash(hash), len);
2563 }
2564 EXPORT_SYMBOL(hashlen_string);
2565 
2566 /*
2567  * We know there's a real path component here of at least
2568  * one character.
2569  */
2570 static inline const char *hash_name(struct nameidata *nd, const char *name, unsigned long *lastword)
2571 {
2572 	unsigned long hash = init_name_hash(nd->path.dentry);
2573 	unsigned long len = 0, c, last = 0;
2574 
2575 	c = (unsigned char)*name;
2576 	do {
2577 		last = (last << 8) + c;
2578 		len++;
2579 		hash = partial_name_hash(c, hash);
2580 		c = (unsigned char)name[len];
2581 	} while (c && c != '/');
2582 
2583 	// This is reliable for DOT or DOTDOT, since the component
2584 	// cannot contain NUL characters - top bits being zero means
2585 	// we cannot have had any other pathnames.
2586 	*lastword = last;
2587 	nd->last.hash = end_name_hash(hash);
2588 	nd->last.len = len;
2589 	return name + len;
2590 }
2591 
2592 #endif
2593 
2594 #ifndef LAST_WORD_IS_DOT
2595   #define LAST_WORD_IS_DOT	0x2e
2596   #define LAST_WORD_IS_DOTDOT	0x2e2e
2597 #endif
2598 
2599 /*
2600  * Name resolution.
2601  * This is the basic name resolution function, turning a pathname into
2602  * the final dentry. We expect 'base' to be positive and a directory.
2603  *
2604  * Returns 0 and nd will have valid dentry and mnt on success.
2605  * Returns error and drops reference to input namei data on failure.
2606  */
2607 static int link_path_walk(const char *name, struct nameidata *nd)
2608 {
2609 	int depth = 0; // depth <= nd->depth
2610 	int err;
2611 
2612 	nd->last_type = LAST_ROOT;
2613 	nd->flags |= LOOKUP_PARENT;
2614 	if (IS_ERR(name))
2615 		return PTR_ERR(name);
2616 	if (*name == '/') {
2617 		do {
2618 			name++;
2619 		} while (unlikely(*name == '/'));
2620 	}
2621 	if (unlikely(!*name)) {
2622 		nd->dir_mode = 0; // short-circuit the 'hardening' idiocy
2623 		return 0;
2624 	}
2625 
2626 	/* At this point we know we have a real path component. */
2627 	for(;;) {
2628 		struct mnt_idmap *idmap;
2629 		const char *link;
2630 		unsigned long lastword;
2631 
2632 		idmap = mnt_idmap(nd->path.mnt);
2633 		err = may_lookup(idmap, nd);
2634 		if (unlikely(err))
2635 			return err;
2636 
2637 		nd->last.name = name;
2638 		name = hash_name(nd, name, &lastword);
2639 
2640 		switch(lastword) {
2641 		case LAST_WORD_IS_DOTDOT:
2642 			nd->last_type = LAST_DOTDOT;
2643 			nd->state |= ND_JUMPED;
2644 			break;
2645 
2646 		case LAST_WORD_IS_DOT:
2647 			nd->last_type = LAST_DOT;
2648 			break;
2649 
2650 		default:
2651 			nd->last_type = LAST_NORM;
2652 			nd->state &= ~ND_JUMPED;
2653 
2654 			struct dentry *parent = nd->path.dentry;
2655 			if (unlikely(parent->d_flags & DCACHE_OP_HASH)) {
2656 				err = parent->d_op->d_hash(parent, &nd->last);
2657 				if (err < 0)
2658 					return err;
2659 			}
2660 		}
2661 
2662 		if (!*name)
2663 			goto OK;
2664 		/*
2665 		 * If it wasn't NUL, we know it was '/'. Skip that
2666 		 * slash, and continue until no more slashes.
2667 		 */
2668 		do {
2669 			name++;
2670 		} while (unlikely(*name == '/'));
2671 		if (unlikely(!*name)) {
2672 OK:
2673 			/* pathname or trailing symlink, done */
2674 			if (likely(!depth)) {
2675 				nd->dir_vfsuid = i_uid_into_vfsuid(idmap, nd->inode);
2676 				nd->dir_mode = nd->inode->i_mode;
2677 				nd->flags &= ~LOOKUP_PARENT;
2678 				return 0;
2679 			}
2680 			/* last component of nested symlink */
2681 			name = nd->stack[--depth].name;
2682 			link = walk_component(nd, 0);
2683 		} else {
2684 			/* not the last component */
2685 			link = walk_component(nd, WALK_MORE);
2686 		}
2687 		if (unlikely(link)) {
2688 			if (IS_ERR(link))
2689 				return PTR_ERR(link);
2690 			/* a symlink to follow */
2691 			nd->stack[depth++].name = name;
2692 			name = link;
2693 			continue;
2694 		}
2695 		if (unlikely(!d_can_lookup(nd->path.dentry))) {
2696 			if (nd->flags & LOOKUP_RCU) {
2697 				if (!try_to_unlazy(nd))
2698 					return -ECHILD;
2699 			}
2700 			return -ENOTDIR;
2701 		}
2702 	}
2703 }
2704 
2705 /* must be paired with terminate_walk() */
2706 static const char *path_init(struct nameidata *nd, unsigned flags)
2707 {
2708 	int error;
2709 	const char *s = nd->pathname;
2710 
2711 	/* LOOKUP_CACHED requires RCU, ask caller to retry */
2712 	if (unlikely((flags & (LOOKUP_RCU | LOOKUP_CACHED)) == LOOKUP_CACHED))
2713 		return ERR_PTR(-EAGAIN);
2714 
2715 	if (unlikely(!*s))
2716 		flags &= ~LOOKUP_RCU;
2717 	if (flags & LOOKUP_RCU)
2718 		rcu_read_lock();
2719 	else
2720 		nd->seq = nd->next_seq = 0;
2721 
2722 	nd->flags = flags;
2723 	nd->state |= ND_JUMPED;
2724 
2725 	nd->m_seq = __read_seqcount_begin(&mount_lock.seqcount);
2726 	nd->r_seq = __read_seqcount_begin(&rename_lock.seqcount);
2727 	smp_rmb();
2728 
2729 	if (unlikely(nd->state & ND_ROOT_PRESET)) {
2730 		struct dentry *root = nd->root.dentry;
2731 		struct inode *inode = root->d_inode;
2732 		if (*s && unlikely(!d_can_lookup(root)))
2733 			return ERR_PTR(-ENOTDIR);
2734 		nd->path = nd->root;
2735 		nd->inode = inode;
2736 		if (flags & LOOKUP_RCU) {
2737 			nd->seq = read_seqcount_begin(&nd->path.dentry->d_seq);
2738 			nd->root_seq = nd->seq;
2739 		} else {
2740 			path_get(&nd->path);
2741 		}
2742 		return s;
2743 	}
2744 
2745 	nd->root.mnt = NULL;
2746 
2747 	/* Absolute pathname -- fetch the root (LOOKUP_IN_ROOT uses nd->dfd). */
2748 	if (*s == '/' && likely(!(flags & LOOKUP_IN_ROOT))) {
2749 		error = nd_jump_root(nd);
2750 		if (unlikely(error))
2751 			return ERR_PTR(error);
2752 		return s;
2753 	}
2754 
2755 	/* Relative pathname -- get the starting-point it is relative to. */
2756 	if (nd->dfd == AT_FDCWD) {
2757 		if (flags & LOOKUP_RCU) {
2758 			struct fs_struct *fs = current->fs;
2759 			unsigned seq;
2760 
2761 			do {
2762 				seq = read_seqbegin(&fs->seq);
2763 				nd->path = fs->pwd;
2764 				nd->inode = nd->path.dentry->d_inode;
2765 				nd->seq = __read_seqcount_begin(&nd->path.dentry->d_seq);
2766 			} while (read_seqretry(&fs->seq, seq));
2767 		} else {
2768 			get_fs_pwd(current->fs, &nd->path);
2769 			nd->inode = nd->path.dentry->d_inode;
2770 		}
2771 	} else {
2772 		/* Caller must check execute permissions on the starting path component */
2773 		CLASS(fd_raw, f)(nd->dfd);
2774 		struct dentry *dentry;
2775 
2776 		if (fd_empty(f))
2777 			return ERR_PTR(-EBADF);
2778 
2779 		if (flags & LOOKUP_LINKAT_EMPTY) {
2780 			if (fd_file(f)->f_cred != current_cred() &&
2781 			    !ns_capable(fd_file(f)->f_cred->user_ns, CAP_DAC_READ_SEARCH))
2782 				return ERR_PTR(-ENOENT);
2783 		}
2784 
2785 		dentry = fd_file(f)->f_path.dentry;
2786 
2787 		if (*s && unlikely(!d_can_lookup(dentry)))
2788 			return ERR_PTR(-ENOTDIR);
2789 
2790 		nd->path = fd_file(f)->f_path;
2791 		if (flags & LOOKUP_RCU) {
2792 			nd->inode = nd->path.dentry->d_inode;
2793 			nd->seq = read_seqcount_begin(&nd->path.dentry->d_seq);
2794 		} else {
2795 			path_get(&nd->path);
2796 			nd->inode = nd->path.dentry->d_inode;
2797 		}
2798 	}
2799 
2800 	/* For scoped-lookups we need to set the root to the dirfd as well. */
2801 	if (unlikely(flags & LOOKUP_IS_SCOPED)) {
2802 		nd->root = nd->path;
2803 		if (flags & LOOKUP_RCU) {
2804 			nd->root_seq = nd->seq;
2805 		} else {
2806 			path_get(&nd->root);
2807 			nd->state |= ND_ROOT_GRABBED;
2808 		}
2809 	}
2810 	return s;
2811 }
2812 
2813 static inline const char *lookup_last(struct nameidata *nd)
2814 {
2815 	if (nd->last_type == LAST_NORM && nd->last.name[nd->last.len])
2816 		nd->flags |= LOOKUP_FOLLOW | LOOKUP_DIRECTORY;
2817 
2818 	return walk_component(nd, WALK_TRAILING);
2819 }
2820 
2821 static int handle_lookup_down(struct nameidata *nd)
2822 {
2823 	if (!(nd->flags & LOOKUP_RCU))
2824 		dget(nd->path.dentry);
2825 	nd->next_seq = nd->seq;
2826 	return PTR_ERR(step_into(nd, WALK_NOFOLLOW, nd->path.dentry));
2827 }
2828 
2829 /* Returns 0 and nd will be valid on success; Returns error, otherwise. */
2830 static int path_lookupat(struct nameidata *nd, unsigned flags, struct path *path)
2831 {
2832 	const char *s = path_init(nd, flags);
2833 	int err;
2834 
2835 	if (unlikely(flags & LOOKUP_DOWN) && !IS_ERR(s)) {
2836 		err = handle_lookup_down(nd);
2837 		if (unlikely(err < 0))
2838 			s = ERR_PTR(err);
2839 	}
2840 
2841 	while (!(err = link_path_walk(s, nd)) &&
2842 	       (s = lookup_last(nd)) != NULL)
2843 		;
2844 	if (!err && unlikely(nd->flags & LOOKUP_MOUNTPOINT)) {
2845 		err = handle_lookup_down(nd);
2846 		nd->state &= ~ND_JUMPED; // no d_weak_revalidate(), please...
2847 	}
2848 	if (!err)
2849 		err = complete_walk(nd);
2850 
2851 	if (!err && nd->flags & LOOKUP_DIRECTORY)
2852 		if (!d_can_lookup(nd->path.dentry))
2853 			err = -ENOTDIR;
2854 	if (!err) {
2855 		*path = nd->path;
2856 		nd->path.mnt = NULL;
2857 		nd->path.dentry = NULL;
2858 	}
2859 	terminate_walk(nd);
2860 	return err;
2861 }
2862 
2863 int filename_lookup(int dfd, struct filename *name, unsigned flags,
2864 		    struct path *path, const struct path *root)
2865 {
2866 	int retval;
2867 	struct nameidata nd;
2868 	if (IS_ERR(name))
2869 		return PTR_ERR(name);
2870 	set_nameidata(&nd, dfd, name, root);
2871 	retval = path_lookupat(&nd, flags | LOOKUP_RCU, path);
2872 	if (unlikely(retval == -ECHILD))
2873 		retval = path_lookupat(&nd, flags, path);
2874 	if (unlikely(retval == -ESTALE))
2875 		retval = path_lookupat(&nd, flags | LOOKUP_REVAL, path);
2876 
2877 	if (likely(!retval))
2878 		audit_inode(name, path->dentry,
2879 			    flags & LOOKUP_MOUNTPOINT ? AUDIT_INODE_NOEVAL : 0);
2880 	restore_nameidata();
2881 	return retval;
2882 }
2883 
2884 /* Returns 0 and nd will be valid on success; Returns error, otherwise. */
2885 static int path_parentat(struct nameidata *nd, unsigned flags,
2886 				struct path *parent)
2887 {
2888 	const char *s = path_init(nd, flags);
2889 	int err = link_path_walk(s, nd);
2890 	if (!err)
2891 		err = complete_walk(nd);
2892 	if (!err) {
2893 		*parent = nd->path;
2894 		nd->path.mnt = NULL;
2895 		nd->path.dentry = NULL;
2896 	}
2897 	terminate_walk(nd);
2898 	return err;
2899 }
2900 
2901 /* Note: this does not consume "name" */
2902 static int __filename_parentat(int dfd, struct filename *name,
2903 			       unsigned int flags, struct path *parent,
2904 			       struct qstr *last, enum last_type *type,
2905 			       const struct path *root)
2906 {
2907 	int retval;
2908 	struct nameidata nd;
2909 
2910 	if (IS_ERR(name))
2911 		return PTR_ERR(name);
2912 	set_nameidata(&nd, dfd, name, root);
2913 	retval = path_parentat(&nd, flags | LOOKUP_RCU, parent);
2914 	if (unlikely(retval == -ECHILD))
2915 		retval = path_parentat(&nd, flags, parent);
2916 	if (unlikely(retval == -ESTALE))
2917 		retval = path_parentat(&nd, flags | LOOKUP_REVAL, parent);
2918 	if (likely(!retval)) {
2919 		*last = nd.last;
2920 		*type = nd.last_type;
2921 		audit_inode(name, parent->dentry, AUDIT_INODE_PARENT);
2922 	}
2923 	restore_nameidata();
2924 	return retval;
2925 }
2926 
2927 static int filename_parentat(int dfd, struct filename *name,
2928 			     unsigned int flags, struct path *parent,
2929 			     struct qstr *last, enum last_type *type)
2930 {
2931 	return __filename_parentat(dfd, name, flags, parent, last, type, NULL);
2932 }
2933 
2934 static struct dentry *__start_dirop(struct dentry *parent, struct qstr *name,
2935 				    unsigned int lookup_flags,
2936 				    unsigned int state)
2937 {
2938 	struct dentry *dentry;
2939 	struct inode *dir = d_inode(parent);
2940 
2941 	if (state == TASK_KILLABLE) {
2942 		int ret = down_write_killable_nested(&dir->i_rwsem,
2943 						     I_MUTEX_PARENT);
2944 		if (ret)
2945 			return ERR_PTR(ret);
2946 	} else {
2947 		inode_lock_nested(dir, I_MUTEX_PARENT);
2948 	}
2949 	dentry = lookup_one_qstr_excl(name, parent, lookup_flags);
2950 	if (IS_ERR(dentry))
2951 		inode_unlock(dir);
2952 	return dentry;
2953 }
2954 
2955 /**
2956  * start_dirop - begin a create or remove dirop, performing locking and lookup
2957  * @parent:       the dentry of the parent in which the operation will occur
2958  * @name:         a qstr holding the name within that parent
2959  * @lookup_flags: intent and other lookup flags.
2960  *
2961  * The lookup is performed and necessary locks are taken so that, on success,
2962  * the returned dentry can be operated on safely.
2963  * The qstr must already have the hash value calculated.
2964  *
2965  * Returns: a locked dentry, or an error.
2966  *
2967  */
2968 struct dentry *start_dirop(struct dentry *parent, struct qstr *name,
2969 			   unsigned int lookup_flags)
2970 {
2971 	return __start_dirop(parent, name, lookup_flags, TASK_NORMAL);
2972 }
2973 
2974 /**
2975  * end_dirop - signal completion of a dirop
2976  * @de: the dentry which was returned by start_dirop or similar.
2977  *
2978  * If the de is an error, nothing happens. Otherwise any lock taken to
2979  * protect the dentry is dropped and the dentry itself is release (dput()).
2980  */
2981 void end_dirop(struct dentry *de)
2982 {
2983 	if (!IS_ERR(de)) {
2984 		inode_unlock(de->d_parent->d_inode);
2985 		dput(de);
2986 	}
2987 }
2988 EXPORT_SYMBOL(end_dirop);
2989 
2990 /* does lookup, returns the object with parent locked */
2991 struct dentry *start_removing_path(const char *name, struct path *path)
2992 {
2993 	CLASS(filename_kernel, filename)(name);
2994 	struct path parent_path __free(path_put) = {};
2995 	struct dentry *d;
2996 	struct qstr last;
2997 	enum last_type type;
2998 	int error;
2999 
3000 	error = filename_parentat(AT_FDCWD, filename, 0, &parent_path, &last,
3001 			&type);
3002 	if (error)
3003 		return ERR_PTR(error);
3004 	if (unlikely(type != LAST_NORM))
3005 		return ERR_PTR(-EINVAL);
3006 	/* don't fail immediately if it's r/o, at least try to report other errors */
3007 	error = mnt_want_write(parent_path.mnt);
3008 	d = start_dirop(parent_path.dentry, &last, 0);
3009 	if (IS_ERR(d))
3010 		goto drop;
3011 	if (error)
3012 		goto fail;
3013 	path->dentry = no_free_ptr(parent_path.dentry);
3014 	path->mnt = no_free_ptr(parent_path.mnt);
3015 	return d;
3016 
3017 fail:
3018 	end_dirop(d);
3019 	d = ERR_PTR(error);
3020 drop:
3021 	if (!error)
3022 		mnt_drop_write(parent_path.mnt);
3023 	return d;
3024 }
3025 
3026 /**
3027  * kern_path_parent: lookup path returning parent and target
3028  * @name: path name
3029  * @path: path to store parent in
3030  *
3031  * The path @name should end with a normal component, not "." or ".." or "/".
3032  * A lookup is performed and if successful the parent information
3033  * is store in @parent and the dentry is returned.
3034  *
3035  * The dentry maybe negative, the parent will be positive.
3036  *
3037  * Returns:  dentry or error.
3038  */
3039 struct dentry *kern_path_parent(const char *name, struct path *path)
3040 {
3041 	struct path parent_path __free(path_put) = {};
3042 	CLASS(filename_kernel, filename)(name);
3043 	struct dentry *d;
3044 	struct qstr last;
3045 	enum last_type type;
3046 	int error;
3047 
3048 	error = filename_parentat(AT_FDCWD, filename, 0, &parent_path, &last, &type);
3049 	if (error)
3050 		return ERR_PTR(error);
3051 	if (unlikely(type != LAST_NORM))
3052 		return ERR_PTR(-EINVAL);
3053 
3054 	d = lookup_noperm_unlocked(&last, parent_path.dentry);
3055 	if (IS_ERR(d))
3056 		return d;
3057 	path->dentry = no_free_ptr(parent_path.dentry);
3058 	path->mnt = no_free_ptr(parent_path.mnt);
3059 	return d;
3060 }
3061 
3062 int kern_path(const char *name, unsigned int flags, struct path *path)
3063 {
3064 	CLASS(filename_kernel, filename)(name);
3065 	return filename_lookup(AT_FDCWD, filename, flags, path, NULL);
3066 }
3067 EXPORT_SYMBOL(kern_path);
3068 
3069 /**
3070  * vfs_path_parent_lookup - lookup a parent path relative to a dentry-vfsmount pair
3071  * @filename: filename structure
3072  * @flags: lookup flags
3073  * @parent: pointer to struct path to fill
3074  * @last: last component
3075  * @root: pointer to struct path of the base directory
3076  */
3077 int vfs_path_parent_lookup(struct filename *filename, unsigned int flags,
3078 			   struct path *parent, struct qstr *last,
3079 			   const struct path *root)
3080 {
3081 	enum last_type type;
3082 	int err =  __filename_parentat(AT_FDCWD, filename, flags, parent, last,
3083 				       &type, root);
3084 	if (err)
3085 		return err;
3086 	if (unlikely(type != LAST_NORM)) {
3087 		path_put(parent);
3088 		return -EINVAL;
3089 	}
3090 	return 0;
3091 }
3092 EXPORT_SYMBOL(vfs_path_parent_lookup);
3093 
3094 /**
3095  * vfs_path_lookup - lookup a file path relative to a dentry-vfsmount pair
3096  * @dentry:  pointer to dentry of the base directory
3097  * @mnt: pointer to vfs mount of the base directory
3098  * @name: pointer to file name
3099  * @flags: lookup flags
3100  * @path: pointer to struct path to fill
3101  */
3102 int vfs_path_lookup(struct dentry *dentry, struct vfsmount *mnt,
3103 		    const char *name, unsigned int flags,
3104 		    struct path *path)
3105 {
3106 	CLASS(filename_kernel, filename)(name);
3107 	struct path root = {.mnt = mnt, .dentry = dentry};
3108 
3109 	/* the first argument of filename_lookup() is ignored with root */
3110 	return filename_lookup(AT_FDCWD, filename, flags, path, &root);
3111 }
3112 EXPORT_SYMBOL(vfs_path_lookup);
3113 
3114 int lookup_noperm_common(struct qstr *qname, struct dentry *base)
3115 {
3116 	const char *name = qname->name;
3117 	u32 len = qname->len;
3118 
3119 	qname->hash = full_name_hash(base, name, len);
3120 	if (!len)
3121 		return -EACCES;
3122 
3123 	if (name_is_dot_dotdot(name, len))
3124 		return -EACCES;
3125 
3126 	while (len--) {
3127 		unsigned int c = *(const unsigned char *)name++;
3128 		if (c == '/' || c == '\0')
3129 			return -EACCES;
3130 	}
3131 	/*
3132 	 * See if the low-level filesystem might want
3133 	 * to use its own hash..
3134 	 */
3135 	if (base->d_flags & DCACHE_OP_HASH) {
3136 		int err = base->d_op->d_hash(base, qname);
3137 		if (err < 0)
3138 			return err;
3139 	}
3140 	return 0;
3141 }
3142 
3143 static int lookup_one_common(struct mnt_idmap *idmap,
3144 			     struct qstr *qname, struct dentry *base)
3145 {
3146 	int err;
3147 	err = lookup_noperm_common(qname, base);
3148 	if (err < 0)
3149 		return err;
3150 	return inode_permission(idmap, base->d_inode, MAY_EXEC);
3151 }
3152 
3153 /**
3154  * try_lookup_noperm - filesystem helper to lookup single pathname component
3155  * @name:	qstr storing pathname component to lookup
3156  * @base:	base directory to lookup from
3157  *
3158  * Look up a dentry by name in the dcache, returning NULL if it does not
3159  * currently exist or an error if there is a problem with the name.
3160  * The function does not try to create a dentry and if one
3161  * is found it doesn't try to revalidate it.
3162  *
3163  * Note that this routine is purely a helper for filesystem usage and should
3164  * not be called by generic code.  It does no permission checking.
3165  *
3166  * No locks need be held - only a counted reference to @base is needed.
3167  *
3168  * Returns:
3169  *   - ref-counted dentry on success, or
3170  *   - %NULL if name could not be found, or
3171  *   - ERR_PTR(-EACCES) if name is dot or dotdot or contains a slash or nul, or
3172  *   - ERR_PTR() if fs provide ->d_hash, and this returned an error.
3173  */
3174 struct dentry *try_lookup_noperm(struct qstr *name, struct dentry *base)
3175 {
3176 	int err;
3177 
3178 	err = lookup_noperm_common(name, base);
3179 	if (err)
3180 		return ERR_PTR(err);
3181 
3182 	return d_lookup(base, name);
3183 }
3184 EXPORT_SYMBOL(try_lookup_noperm);
3185 
3186 /**
3187  * lookup_noperm - filesystem helper to lookup single pathname component
3188  * @name:	qstr storing pathname component to lookup
3189  * @base:	base directory to lookup from
3190  *
3191  * Note that this routine is purely a helper for filesystem usage and should
3192  * not be called by generic code.  It does no permission checking.
3193  *
3194  * The caller must hold base->i_rwsem.
3195  */
3196 struct dentry *lookup_noperm(struct qstr *name, struct dentry *base)
3197 {
3198 	struct dentry *dentry;
3199 	int err;
3200 
3201 	WARN_ON_ONCE(!inode_is_locked(base->d_inode));
3202 
3203 	err = lookup_noperm_common(name, base);
3204 	if (err)
3205 		return ERR_PTR(err);
3206 
3207 	dentry = lookup_dcache(name, base, 0);
3208 	return dentry ? dentry : __lookup_slow(name, base, 0);
3209 }
3210 EXPORT_SYMBOL(lookup_noperm);
3211 
3212 /**
3213  * lookup_one - lookup single pathname component
3214  * @idmap:	idmap of the mount the lookup is performed from
3215  * @name:	qstr holding pathname component to lookup
3216  * @base:	base directory to lookup from
3217  *
3218  * This can be used for in-kernel filesystem clients such as file servers.
3219  *
3220  * The caller must hold base->i_rwsem.
3221  */
3222 struct dentry *lookup_one(struct mnt_idmap *idmap, struct qstr *name,
3223 			  struct dentry *base)
3224 {
3225 	struct dentry *dentry;
3226 	int err;
3227 
3228 	WARN_ON_ONCE(!inode_is_locked(base->d_inode));
3229 
3230 	err = lookup_one_common(idmap, name, base);
3231 	if (err)
3232 		return ERR_PTR(err);
3233 
3234 	dentry = lookup_dcache(name, base, 0);
3235 	return dentry ? dentry : __lookup_slow(name, base, 0);
3236 }
3237 EXPORT_SYMBOL(lookup_one);
3238 
3239 /**
3240  * lookup_one_unlocked - lookup single pathname component
3241  * @idmap:	idmap of the mount the lookup is performed from
3242  * @name:	qstr olding pathname component to lookup
3243  * @base:	base directory to lookup from
3244  *
3245  * This can be used for in-kernel filesystem clients such as file servers.
3246  *
3247  * Unlike lookup_one, it should be called without the parent
3248  * i_rwsem held, and will take the i_rwsem itself if necessary.
3249  *
3250  * Returns: - A dentry, possibly negative, or
3251  *	    - same errors as try_lookup_noperm() or
3252  *	    - ERR_PTR(-ENOENT) if parent has been removed, or
3253  *	    - ERR_PTR(-EACCES) if parent directory is not searchable.
3254  */
3255 struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap, struct qstr *name,
3256 				   struct dentry *base)
3257 {
3258 	int err;
3259 	struct dentry *ret;
3260 
3261 	err = lookup_one_common(idmap, name, base);
3262 	if (err)
3263 		return ERR_PTR(err);
3264 
3265 	ret = lookup_dcache(name, base, 0);
3266 	if (!ret)
3267 		ret = lookup_slow(name, base, 0);
3268 	return ret;
3269 }
3270 EXPORT_SYMBOL(lookup_one_unlocked);
3271 
3272 /**
3273  * lookup_one_positive_killable - lookup single pathname component
3274  * @idmap:	idmap of the mount the lookup is performed from
3275  * @name:	qstr olding pathname component to lookup
3276  * @base:	base directory to lookup from
3277  *
3278  * This helper will yield ERR_PTR(-ENOENT) on negatives. The helper returns
3279  * known positive or ERR_PTR(). This is what most of the users want.
3280  *
3281  * Note that pinned negative with unlocked parent _can_ become positive at any
3282  * time, so callers of lookup_one_unlocked() need to be very careful; pinned
3283  * positives have >d_inode stable, so this one avoids such problems.
3284  *
3285  * This can be used for in-kernel filesystem clients such as file servers.
3286  *
3287  * It should be called without the parent i_rwsem held, and will take
3288  * the i_rwsem itself if necessary.  If a fatal signal is pending or
3289  * delivered, it will return %-EINTR if the lock is needed.
3290  *
3291  * Returns: A dentry, possibly negative, or
3292  *	   - same errors as lookup_one_unlocked() or
3293  *	   - ERR_PTR(-EINTR) if a fatal signal is pending.
3294  */
3295 struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap,
3296 					    struct qstr *name,
3297 					    struct dentry *base)
3298 {
3299 	int err;
3300 	struct dentry *ret;
3301 
3302 	err = lookup_one_common(idmap, name, base);
3303 	if (err)
3304 		return ERR_PTR(err);
3305 
3306 	ret = lookup_dcache(name, base, 0);
3307 	if (!ret)
3308 		ret = lookup_slow_killable(name, base, 0);
3309 	if (!IS_ERR(ret) && d_flags_negative(smp_load_acquire(&ret->d_flags))) {
3310 		dput(ret);
3311 		ret = ERR_PTR(-ENOENT);
3312 	}
3313 	return ret;
3314 }
3315 EXPORT_SYMBOL(lookup_one_positive_killable);
3316 
3317 /**
3318  * lookup_one_positive_unlocked - lookup single pathname component
3319  * @idmap:	idmap of the mount the lookup is performed from
3320  * @name:	qstr holding pathname component to lookup
3321  * @base:	base directory to lookup from
3322  *
3323  * This helper will yield ERR_PTR(-ENOENT) on negatives. The helper returns
3324  * known positive or ERR_PTR(). This is what most of the users want.
3325  *
3326  * Note that pinned negative with unlocked parent _can_ become positive at any
3327  * time, so callers of lookup_one_unlocked() need to be very careful; pinned
3328  * positives have >d_inode stable, so this one avoids such problems.
3329  *
3330  * This can be used for in-kernel filesystem clients such as file servers.
3331  *
3332  * The helper should be called without i_rwsem held.
3333  *
3334  * Returns: A positive dentry, or
3335  *	   - ERR_PTR(-ENOENT) if the name could not be found, or
3336  *	   - same errors as lookup_one_unlocked().
3337  */
3338 struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap,
3339 					    struct qstr *name,
3340 					    struct dentry *base)
3341 {
3342 	struct dentry *ret = lookup_one_unlocked(idmap, name, base);
3343 
3344 	if (!IS_ERR(ret) && d_flags_negative(smp_load_acquire(&ret->d_flags))) {
3345 		dput(ret);
3346 		ret = ERR_PTR(-ENOENT);
3347 	}
3348 	return ret;
3349 }
3350 EXPORT_SYMBOL(lookup_one_positive_unlocked);
3351 
3352 /**
3353  * lookup_noperm_unlocked - filesystem helper to lookup single pathname component
3354  * @name:	pathname component to lookup
3355  * @base:	base directory to lookup from
3356  *
3357  * Note that this routine is purely a helper for filesystem usage and should
3358  * not be called by generic code. It does no permission checking.
3359  *
3360  * Unlike lookup_noperm(), it should be called without the parent
3361  * i_rwsem held, and will take the i_rwsem itself if necessary.
3362  *
3363  * Unlike try_lookup_noperm() it *does* revalidate the dentry if it already
3364  * existed.
3365  *
3366  * Returns: A dentry, possibly negative, or
3367  *	   - ERR_PTR(-ENOENT) if parent has been removed, or
3368  *	   - same errors as try_lookup_noperm()
3369  */
3370 struct dentry *lookup_noperm_unlocked(struct qstr *name, struct dentry *base)
3371 {
3372 	struct dentry *ret;
3373 	int err;
3374 
3375 	err = lookup_noperm_common(name, base);
3376 	if (err)
3377 		return ERR_PTR(err);
3378 
3379 	ret = lookup_dcache(name, base, 0);
3380 	if (!ret)
3381 		ret = lookup_slow(name, base, 0);
3382 	return ret;
3383 }
3384 EXPORT_SYMBOL(lookup_noperm_unlocked);
3385 
3386 /*
3387  * Like lookup_noperm_unlocked(), except that it yields ERR_PTR(-ENOENT)
3388  * on negatives.  Returns known positive or ERR_PTR(); that's what
3389  * most of the users want.  Note that pinned negative with unlocked parent
3390  * _can_ become positive at any time, so callers of lookup_noperm_unlocked()
3391  * need to be very careful; pinned positives have ->d_inode stable, so
3392  * this one avoids such problems.
3393  *
3394  * Returns: A positive dentry, or
3395  *	   - ERR_PTR(-ENOENT) if name cannot be found or parent has been removed, or
3396  *	   - same errors as try_lookup_noperm()
3397  */
3398 struct dentry *lookup_noperm_positive_unlocked(struct qstr *name,
3399 					       struct dentry *base)
3400 {
3401 	struct dentry *ret;
3402 
3403 	ret = lookup_noperm_unlocked(name, base);
3404 	if (!IS_ERR(ret) && d_flags_negative(smp_load_acquire(&ret->d_flags))) {
3405 		dput(ret);
3406 		ret = ERR_PTR(-ENOENT);
3407 	}
3408 	return ret;
3409 }
3410 EXPORT_SYMBOL(lookup_noperm_positive_unlocked);
3411 
3412 /**
3413  * start_creating - prepare to create a given name with permission checking
3414  * @idmap:  idmap of the mount
3415  * @parent: directory in which to prepare to create the name
3416  * @name:   the name to be created
3417  *
3418  * Locks are taken and a lookup is performed prior to creating
3419  * an object in a directory.  Permission checking (MAY_EXEC) is performed
3420  * against @idmap.
3421  *
3422  * If the name already exists, a positive dentry is returned, so
3423  * behaviour is similar to O_CREAT without O_EXCL, which doesn't fail
3424  * with -EEXIST.
3425  *
3426  * Returns: a negative or positive dentry, or an error.
3427  */
3428 struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent,
3429 			      struct qstr *name)
3430 {
3431 	int err = lookup_one_common(idmap, name, parent);
3432 
3433 	if (err)
3434 		return ERR_PTR(err);
3435 	return start_dirop(parent, name, LOOKUP_CREATE);
3436 }
3437 EXPORT_SYMBOL(start_creating);
3438 
3439 /**
3440  * start_removing - prepare to remove a given name with permission checking
3441  * @idmap:  idmap of the mount
3442  * @parent: directory in which to find the name
3443  * @name:   the name to be removed
3444  *
3445  * Locks are taken and a lookup in performed prior to removing
3446  * an object from a directory.  Permission checking (MAY_EXEC) is performed
3447  * against @idmap.
3448  *
3449  * If the name doesn't exist, an error is returned.
3450  *
3451  * end_removing() should be called when removal is complete, or aborted.
3452  *
3453  * Returns: a positive dentry, or an error.
3454  */
3455 struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent,
3456 			      struct qstr *name)
3457 {
3458 	int err = lookup_one_common(idmap, name, parent);
3459 
3460 	if (err)
3461 		return ERR_PTR(err);
3462 	return start_dirop(parent, name, 0);
3463 }
3464 EXPORT_SYMBOL(start_removing);
3465 
3466 /**
3467  * start_creating_killable - prepare to create a given name with permission checking
3468  * @idmap:  idmap of the mount
3469  * @parent: directory in which to prepare to create the name
3470  * @name:   the name to be created
3471  *
3472  * Locks are taken and a lookup in performed prior to creating
3473  * an object in a directory.  Permission checking (MAY_EXEC) is performed
3474  * against @idmap.
3475  *
3476  * If the name already exists, a positive dentry is returned.
3477  *
3478  * If a signal is received or was already pending, the function aborts
3479  * with -EINTR;
3480  *
3481  * Returns: a negative or positive dentry, or an error.
3482  */
3483 struct dentry *start_creating_killable(struct mnt_idmap *idmap,
3484 				       struct dentry *parent,
3485 				       struct qstr *name)
3486 {
3487 	int err = lookup_one_common(idmap, name, parent);
3488 
3489 	if (err)
3490 		return ERR_PTR(err);
3491 	return __start_dirop(parent, name, LOOKUP_CREATE, TASK_KILLABLE);
3492 }
3493 EXPORT_SYMBOL(start_creating_killable);
3494 
3495 /**
3496  * start_removing_killable - prepare to remove a given name with permission checking
3497  * @idmap:  idmap of the mount
3498  * @parent: directory in which to find the name
3499  * @name:   the name to be removed
3500  *
3501  * Locks are taken and a lookup in performed prior to removing
3502  * an object from a directory.  Permission checking (MAY_EXEC) is performed
3503  * against @idmap.
3504  *
3505  * If the name doesn't exist, an error is returned.
3506  *
3507  * end_removing() should be called when removal is complete, or aborted.
3508  *
3509  * If a signal is received or was already pending, the function aborts
3510  * with -EINTR;
3511  *
3512  * Returns: a positive dentry, or an error.
3513  */
3514 struct dentry *start_removing_killable(struct mnt_idmap *idmap,
3515 				       struct dentry *parent,
3516 				       struct qstr *name)
3517 {
3518 	int err = lookup_one_common(idmap, name, parent);
3519 
3520 	if (err)
3521 		return ERR_PTR(err);
3522 	return __start_dirop(parent, name, 0, TASK_KILLABLE);
3523 }
3524 EXPORT_SYMBOL(start_removing_killable);
3525 
3526 /**
3527  * start_creating_noperm - prepare to create a given name without permission checking
3528  * @parent: directory in which to prepare to create the name
3529  * @name:   the name to be created
3530  *
3531  * Locks are taken and a lookup in performed prior to creating
3532  * an object in a directory.
3533  *
3534  * If the name already exists, a positive dentry is returned.
3535  *
3536  * Returns: a negative or positive dentry, or an error.
3537  */
3538 struct dentry *start_creating_noperm(struct dentry *parent,
3539 				     struct qstr *name)
3540 {
3541 	int err = lookup_noperm_common(name, parent);
3542 
3543 	if (err)
3544 		return ERR_PTR(err);
3545 	return start_dirop(parent, name, LOOKUP_CREATE);
3546 }
3547 EXPORT_SYMBOL(start_creating_noperm);
3548 
3549 /**
3550  * start_removing_noperm - prepare to remove a given name without permission checking
3551  * @parent: directory in which to find the name
3552  * @name:   the name to be removed
3553  *
3554  * Locks are taken and a lookup in performed prior to removing
3555  * an object from a directory.
3556  *
3557  * If the name doesn't exist, an error is returned.
3558  *
3559  * end_removing() should be called when removal is complete, or aborted.
3560  *
3561  * Returns: a positive dentry, or an error.
3562  */
3563 struct dentry *start_removing_noperm(struct dentry *parent,
3564 				     struct qstr *name)
3565 {
3566 	int err = lookup_noperm_common(name, parent);
3567 
3568 	if (err)
3569 		return ERR_PTR(err);
3570 	return start_dirop(parent, name, 0);
3571 }
3572 EXPORT_SYMBOL(start_removing_noperm);
3573 
3574 /**
3575  * start_creating_dentry - prepare to create a given dentry
3576  * @parent: directory from which dentry should be removed
3577  * @child:  the dentry to be removed
3578  *
3579  * A lock is taken to protect the dentry again other dirops and
3580  * the validity of the dentry is checked: correct parent and still hashed.
3581  *
3582  * If the dentry is valid and negative a reference is taken and
3583  * returned.  If not an error is returned.
3584  *
3585  * end_creating() should be called when creation is complete, or aborted.
3586  *
3587  * Returns: the valid dentry, or an error.
3588  */
3589 struct dentry *start_creating_dentry(struct dentry *parent,
3590 				     struct dentry *child)
3591 {
3592 	inode_lock_nested(parent->d_inode, I_MUTEX_PARENT);
3593 	if (unlikely(IS_DEADDIR(parent->d_inode) ||
3594 		     child->d_parent != parent ||
3595 		     d_unhashed(child))) {
3596 		inode_unlock(parent->d_inode);
3597 		return ERR_PTR(-EINVAL);
3598 	}
3599 	if (d_is_positive(child)) {
3600 		inode_unlock(parent->d_inode);
3601 		return ERR_PTR(-EEXIST);
3602 	}
3603 	return dget(child);
3604 }
3605 EXPORT_SYMBOL(start_creating_dentry);
3606 
3607 /**
3608  * start_removing_dentry - prepare to remove a given dentry
3609  * @parent: directory from which dentry should be removed
3610  * @child:  the dentry to be removed
3611  *
3612  * A lock is taken to protect the dentry again other dirops and
3613  * the validity of the dentry is checked: correct parent and still hashed.
3614  *
3615  * If the dentry is valid and positive, a reference is taken and
3616  * returned.  If not an error is returned.
3617  *
3618  * end_removing() should be called when removal is complete, or aborted.
3619  *
3620  * Returns: the valid dentry, or an error.
3621  */
3622 struct dentry *start_removing_dentry(struct dentry *parent,
3623 				     struct dentry *child)
3624 {
3625 	inode_lock_nested(parent->d_inode, I_MUTEX_PARENT);
3626 	if (unlikely(IS_DEADDIR(parent->d_inode) ||
3627 		     child->d_parent != parent ||
3628 		     d_unhashed(child))) {
3629 		inode_unlock(parent->d_inode);
3630 		return ERR_PTR(-EINVAL);
3631 	}
3632 	if (d_is_negative(child)) {
3633 		inode_unlock(parent->d_inode);
3634 		return ERR_PTR(-ENOENT);
3635 	}
3636 	return dget(child);
3637 }
3638 EXPORT_SYMBOL(start_removing_dentry);
3639 
3640 #ifdef CONFIG_UNIX98_PTYS
3641 int path_pts(struct path *path)
3642 {
3643 	/* Find something mounted on "pts" in the same directory as
3644 	 * the input path.
3645 	 */
3646 	struct dentry *parent = dget_parent(path->dentry);
3647 	struct dentry *child;
3648 
3649 	if (unlikely(!path_connected(path->mnt, parent))) {
3650 		dput(parent);
3651 		return -ENOENT;
3652 	}
3653 	dput(path->dentry);
3654 	path->dentry = parent;
3655 	child = d_hash_and_lookup(parent, &QSTR("pts"));
3656 	if (IS_ERR_OR_NULL(child))
3657 		return -ENOENT;
3658 
3659 	path->dentry = child;
3660 	dput(parent);
3661 	follow_down(path, 0);
3662 	return 0;
3663 }
3664 #endif
3665 
3666 int user_path_at(int dfd, const char __user *name, unsigned flags,
3667 		 struct path *path)
3668 {
3669 	CLASS(filename_flags, filename)(name, flags);
3670 	return filename_lookup(dfd, filename, flags, path, NULL);
3671 }
3672 EXPORT_SYMBOL(user_path_at);
3673 
3674 int __check_sticky(struct mnt_idmap *idmap, struct inode *dir,
3675 		   struct inode *inode)
3676 {
3677 	kuid_t fsuid = current_fsuid();
3678 
3679 	if (vfsuid_eq_kuid(i_uid_into_vfsuid(idmap, inode), fsuid))
3680 		return 0;
3681 	if (vfsuid_eq_kuid(i_uid_into_vfsuid(idmap, dir), fsuid))
3682 		return 0;
3683 	return !capable_wrt_inode_uidgid(idmap, inode, CAP_FOWNER);
3684 }
3685 EXPORT_SYMBOL(__check_sticky);
3686 
3687 /*
3688  *	Check whether we can remove a link victim from directory dir, check
3689  *  whether the type of victim is right.
3690  *  1. We can't do it if dir is read-only (done in permission())
3691  *  2. We should have write and exec permissions on dir
3692  *  3. We can't remove anything from append-only dir
3693  *  4. We can't do anything with immutable dir (done in permission())
3694  *  5. If the sticky bit on dir is set we should either
3695  *	a. be owner of dir, or
3696  *	b. be owner of victim, or
3697  *	c. have CAP_FOWNER capability
3698  *  6. If the victim is append-only or immutable we can't do antyhing with
3699  *     links pointing to it.
3700  *  7. If the victim has an unknown uid or gid we can't change the inode.
3701  *  8. If we were asked to remove a directory and victim isn't one - ENOTDIR.
3702  *  9. If we were asked to remove a non-directory and victim isn't one - EISDIR.
3703  * 10. We can't remove a root or mountpoint.
3704  * 11. We don't allow removal of NFS sillyrenamed files; it's handled by
3705  *     nfs_async_unlink().
3706  */
3707 int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir,
3708 		      struct dentry *victim, bool isdir)
3709 {
3710 	struct inode *inode = d_backing_inode(victim);
3711 	int error;
3712 
3713 	if (d_is_negative(victim))
3714 		return -ENOENT;
3715 	BUG_ON(!inode);
3716 
3717 	BUG_ON(victim->d_parent->d_inode != dir);
3718 
3719 	/* Inode writeback is not safe when the uid or gid are invalid. */
3720 	if (!vfsuid_valid(i_uid_into_vfsuid(idmap, inode)) ||
3721 	    !vfsgid_valid(i_gid_into_vfsgid(idmap, inode)))
3722 		return -EOVERFLOW;
3723 
3724 	audit_inode_child(dir, victim, AUDIT_TYPE_CHILD_DELETE);
3725 
3726 	error = inode_permission(idmap, dir, MAY_WRITE | MAY_EXEC);
3727 	if (error)
3728 		return error;
3729 	if (IS_APPEND(dir))
3730 		return -EPERM;
3731 
3732 	if (check_sticky(idmap, dir, inode) || IS_APPEND(inode) ||
3733 	    IS_IMMUTABLE(inode) || IS_SWAPFILE(inode) ||
3734 	    HAS_UNMAPPED_ID(idmap, inode))
3735 		return -EPERM;
3736 	if (isdir) {
3737 		if (!d_is_dir(victim))
3738 			return -ENOTDIR;
3739 		if (IS_ROOT(victim))
3740 			return -EBUSY;
3741 	} else if (d_is_dir(victim))
3742 		return -EISDIR;
3743 	if (IS_DEADDIR(dir))
3744 		return -ENOENT;
3745 	if (victim->d_flags & DCACHE_NFSFS_RENAMED)
3746 		return -EBUSY;
3747 	return 0;
3748 }
3749 EXPORT_SYMBOL(may_delete_dentry);
3750 
3751 /*	Check whether we can create an object with dentry child in directory
3752  *  dir.
3753  *  1. We can't do it if child already exists (open has special treatment for
3754  *     this case, but since we are inlined it's OK)
3755  *  2. We can't do it if dir is read-only (done in permission())
3756  *  3. We can't do it if the fs can't represent the fsuid or fsgid.
3757  *  4. We should have write and exec permissions on dir
3758  *  5. We can't do it if dir is immutable (done in permission())
3759  */
3760 int may_create_dentry(struct mnt_idmap *idmap,
3761 		      struct inode *dir, struct dentry *child)
3762 {
3763 	audit_inode_child(dir, child, AUDIT_TYPE_CHILD_CREATE);
3764 	if (child->d_inode)
3765 		return -EEXIST;
3766 	if (IS_DEADDIR(dir))
3767 		return -ENOENT;
3768 	if (!fsuidgid_has_mapping(dir->i_sb, idmap))
3769 		return -EOVERFLOW;
3770 
3771 	return inode_permission(idmap, dir, MAY_WRITE | MAY_EXEC);
3772 }
3773 EXPORT_SYMBOL(may_create_dentry);
3774 
3775 // p1 != p2, both are on the same filesystem, ->s_vfs_rename_mutex is held
3776 static struct dentry *lock_two_directories(struct dentry *p1, struct dentry *p2)
3777 {
3778 	struct dentry *p = p1, *q = p2, *r;
3779 
3780 	while ((r = p->d_parent) != p2 && r != p)
3781 		p = r;
3782 	if (r == p2) {
3783 		// p is a child of p2 and an ancestor of p1 or p1 itself
3784 		inode_lock_nested(p2->d_inode, I_MUTEX_PARENT);
3785 		inode_lock_nested(p1->d_inode, I_MUTEX_PARENT2);
3786 		return p;
3787 	}
3788 	// p is the root of connected component that contains p1
3789 	// p2 does not occur on the path from p to p1
3790 	while ((r = q->d_parent) != p1 && r != p && r != q)
3791 		q = r;
3792 	if (r == p1) {
3793 		// q is a child of p1 and an ancestor of p2 or p2 itself
3794 		inode_lock_nested(p1->d_inode, I_MUTEX_PARENT);
3795 		inode_lock_nested(p2->d_inode, I_MUTEX_PARENT2);
3796 		return q;
3797 	} else if (likely(r == p)) {
3798 		// both p2 and p1 are descendents of p
3799 		inode_lock_nested(p1->d_inode, I_MUTEX_PARENT);
3800 		inode_lock_nested(p2->d_inode, I_MUTEX_PARENT2);
3801 		return NULL;
3802 	} else { // no common ancestor at the time we'd been called
3803 		mutex_unlock(&p1->d_sb->s_vfs_rename_mutex);
3804 		return ERR_PTR(-EXDEV);
3805 	}
3806 }
3807 
3808 /*
3809  * p1 and p2 should be directories on the same fs.
3810  */
3811 static struct dentry *lock_rename(struct dentry *p1, struct dentry *p2)
3812 {
3813 	if (p1 == p2) {
3814 		inode_lock_nested(p1->d_inode, I_MUTEX_PARENT);
3815 		return NULL;
3816 	}
3817 
3818 	mutex_lock(&p1->d_sb->s_vfs_rename_mutex);
3819 	return lock_two_directories(p1, p2);
3820 }
3821 
3822 /*
3823  * c1 and p2 should be on the same fs.
3824  */
3825 static struct dentry *lock_rename_child(struct dentry *c1, struct dentry *p2)
3826 {
3827 	if (READ_ONCE(c1->d_parent) == p2) {
3828 		/*
3829 		 * hopefully won't need to touch ->s_vfs_rename_mutex at all.
3830 		 */
3831 		inode_lock_nested(p2->d_inode, I_MUTEX_PARENT);
3832 		/*
3833 		 * now that p2 is locked, nobody can move in or out of it,
3834 		 * so the test below is safe.
3835 		 */
3836 		if (likely(c1->d_parent == p2))
3837 			return NULL;
3838 
3839 		/*
3840 		 * c1 got moved out of p2 while we'd been taking locks;
3841 		 * unlock and fall back to slow case.
3842 		 */
3843 		inode_unlock(p2->d_inode);
3844 	}
3845 
3846 	mutex_lock(&c1->d_sb->s_vfs_rename_mutex);
3847 	/*
3848 	 * nobody can move out of any directories on this fs.
3849 	 */
3850 	if (likely(c1->d_parent != p2))
3851 		return lock_two_directories(c1->d_parent, p2);
3852 
3853 	/*
3854 	 * c1 got moved into p2 while we were taking locks;
3855 	 * we need p2 locked and ->s_vfs_rename_mutex unlocked,
3856 	 * for consistency with lock_rename().
3857 	 */
3858 	inode_lock_nested(p2->d_inode, I_MUTEX_PARENT);
3859 	mutex_unlock(&c1->d_sb->s_vfs_rename_mutex);
3860 	return NULL;
3861 }
3862 
3863 static void unlock_rename(struct dentry *p1, struct dentry *p2)
3864 {
3865 	inode_unlock(p1->d_inode);
3866 	if (p1 != p2) {
3867 		inode_unlock(p2->d_inode);
3868 		mutex_unlock(&p1->d_sb->s_vfs_rename_mutex);
3869 	}
3870 }
3871 
3872 /**
3873  * __start_renaming - lookup and lock names for rename
3874  * @rd:           rename data containing parents and flags, and
3875  *                for receiving found dentries
3876  * @lookup_flags: extra flags to pass to ->lookup (e.g. LOOKUP_REVAL,
3877  *                LOOKUP_NO_SYMLINKS etc).
3878  * @old_last:     name of object in @rd.old_parent
3879  * @new_last:     name of object in @rd.new_parent
3880  *
3881  * Look up two names and ensure locks are in place for
3882  * rename.
3883  *
3884  * On success the found dentries are stored in @rd.old_dentry,
3885  * @rd.new_dentry and an extra ref is taken on @rd.old_parent.
3886  * These references and the lock are dropped by end_renaming().
3887  *
3888  * The passed in qstrs must have the hash calculated, and no permission
3889  * checking is performed.
3890  *
3891  * Returns: zero or an error.
3892  */
3893 static int
3894 __start_renaming(struct renamedata *rd, int lookup_flags,
3895 		 struct qstr *old_last, struct qstr *new_last)
3896 {
3897 	struct dentry *trap;
3898 	struct dentry *d1, *d2;
3899 	int target_flags = LOOKUP_RENAME_TARGET | LOOKUP_CREATE;
3900 	int err;
3901 
3902 	if (rd->flags & RENAME_EXCHANGE)
3903 		target_flags = 0;
3904 	if (rd->flags & RENAME_NOREPLACE)
3905 		target_flags |= LOOKUP_EXCL;
3906 
3907 	trap = lock_rename(rd->old_parent, rd->new_parent);
3908 	if (IS_ERR(trap))
3909 		return PTR_ERR(trap);
3910 
3911 	d1 = lookup_one_qstr_excl(old_last, rd->old_parent,
3912 				  lookup_flags);
3913 	err = PTR_ERR(d1);
3914 	if (IS_ERR(d1))
3915 		goto out_unlock;
3916 
3917 	d2 = lookup_one_qstr_excl(new_last, rd->new_parent,
3918 				  lookup_flags | target_flags);
3919 	err = PTR_ERR(d2);
3920 	if (IS_ERR(d2))
3921 		goto out_dput_d1;
3922 
3923 	if (d1 == trap) {
3924 		/* source is an ancestor of target */
3925 		err = -EINVAL;
3926 		goto out_dput_d2;
3927 	}
3928 
3929 	if (d2 == trap) {
3930 		/* target is an ancestor of source */
3931 		if (rd->flags & RENAME_EXCHANGE)
3932 			err = -EINVAL;
3933 		else
3934 			err = -ENOTEMPTY;
3935 		goto out_dput_d2;
3936 	}
3937 
3938 	rd->old_dentry = d1;
3939 	rd->new_dentry = d2;
3940 	dget(rd->old_parent);
3941 	return 0;
3942 
3943 out_dput_d2:
3944 	dput(d2);
3945 out_dput_d1:
3946 	dput(d1);
3947 out_unlock:
3948 	unlock_rename(rd->old_parent, rd->new_parent);
3949 	return err;
3950 }
3951 
3952 /**
3953  * start_renaming - lookup and lock names for rename with permission checking
3954  * @rd:           rename data containing parents and flags, and
3955  *                for receiving found dentries
3956  * @lookup_flags: extra flags to pass to ->lookup (e.g. LOOKUP_REVAL,
3957  *                LOOKUP_NO_SYMLINKS etc).
3958  * @old_last:     name of object in @rd.old_parent
3959  * @new_last:     name of object in @rd.new_parent
3960  *
3961  * Look up two names and ensure locks are in place for
3962  * rename.
3963  *
3964  * On success the found dentries are stored in @rd.old_dentry,
3965  * @rd.new_dentry.  Also the refcount on @rd->old_parent is increased.
3966  * These references and the lock are dropped by end_renaming().
3967  *
3968  * The passed in qstrs need not have the hash calculated, and basic
3969  * eXecute permission checking is performed against @rd.mnt_idmap.
3970  *
3971  * Returns: zero or an error.
3972  */
3973 int start_renaming(struct renamedata *rd, int lookup_flags,
3974 		   struct qstr *old_last, struct qstr *new_last)
3975 {
3976 	int err;
3977 
3978 	err = lookup_one_common(rd->mnt_idmap, old_last, rd->old_parent);
3979 	if (err)
3980 		return err;
3981 	err = lookup_one_common(rd->mnt_idmap, new_last, rd->new_parent);
3982 	if (err)
3983 		return err;
3984 	return __start_renaming(rd, lookup_flags, old_last, new_last);
3985 }
3986 EXPORT_SYMBOL(start_renaming);
3987 
3988 static int
3989 __start_renaming_dentry(struct renamedata *rd, int lookup_flags,
3990 			struct dentry *old_dentry, struct qstr *new_last)
3991 {
3992 	struct dentry *trap;
3993 	struct dentry *d2;
3994 	int target_flags = LOOKUP_RENAME_TARGET | LOOKUP_CREATE;
3995 	int err;
3996 
3997 	if (rd->flags & RENAME_EXCHANGE)
3998 		target_flags = 0;
3999 	if (rd->flags & RENAME_NOREPLACE)
4000 		target_flags |= LOOKUP_EXCL;
4001 
4002 	/* Already have the dentry - need to be sure to lock the correct parent */
4003 	trap = lock_rename_child(old_dentry, rd->new_parent);
4004 	if (IS_ERR(trap))
4005 		return PTR_ERR(trap);
4006 	if (d_unhashed(old_dentry) ||
4007 	    (rd->old_parent && rd->old_parent != old_dentry->d_parent)) {
4008 		/* dentry was removed, or moved and explicit parent requested */
4009 		err = -EINVAL;
4010 		goto out_unlock;
4011 	}
4012 
4013 	d2 = lookup_one_qstr_excl(new_last, rd->new_parent,
4014 				  lookup_flags | target_flags);
4015 	err = PTR_ERR(d2);
4016 	if (IS_ERR(d2))
4017 		goto out_unlock;
4018 
4019 	if (old_dentry == trap) {
4020 		/* source is an ancestor of target */
4021 		err = -EINVAL;
4022 		goto out_dput_d2;
4023 	}
4024 
4025 	if (d2 == trap) {
4026 		/* target is an ancestor of source */
4027 		if (rd->flags & RENAME_EXCHANGE)
4028 			err = -EINVAL;
4029 		else
4030 			err = -ENOTEMPTY;
4031 		goto out_dput_d2;
4032 	}
4033 
4034 	rd->old_dentry = dget(old_dentry);
4035 	rd->new_dentry = d2;
4036 	rd->old_parent = dget(old_dentry->d_parent);
4037 	return 0;
4038 
4039 out_dput_d2:
4040 	dput(d2);
4041 out_unlock:
4042 	unlock_rename(old_dentry->d_parent, rd->new_parent);
4043 	return err;
4044 }
4045 
4046 /**
4047  * start_renaming_dentry - lookup and lock name for rename with permission checking
4048  * @rd:           rename data containing parents and flags, and
4049  *                for receiving found dentries
4050  * @lookup_flags: extra flags to pass to ->lookup (e.g. LOOKUP_REVAL,
4051  *                LOOKUP_NO_SYMLINKS etc).
4052  * @old_dentry:   dentry of name to move
4053  * @new_last:     name of target in @rd.new_parent
4054  *
4055  * Look up target name and ensure locks are in place for
4056  * rename.
4057  *
4058  * On success the found dentry is stored in @rd.new_dentry and
4059  * @rd.old_parent is confirmed to be the parent of @old_dentry.  If it
4060  * was originally %NULL, it is set.  In either case a reference is taken
4061  * so that end_renaming() can have a stable reference to unlock.
4062  *
4063  * References and the lock can be dropped with end_renaming()
4064  *
4065  * The passed in qstr need not have the hash calculated, and basic
4066  * eXecute permission checking is performed against @rd.mnt_idmap.
4067  *
4068  * Returns: zero or an error.
4069  */
4070 int start_renaming_dentry(struct renamedata *rd, int lookup_flags,
4071 			  struct dentry *old_dentry, struct qstr *new_last)
4072 {
4073 	int err;
4074 
4075 	err = lookup_one_common(rd->mnt_idmap, new_last, rd->new_parent);
4076 	if (err)
4077 		return err;
4078 	return __start_renaming_dentry(rd, lookup_flags, old_dentry, new_last);
4079 }
4080 EXPORT_SYMBOL(start_renaming_dentry);
4081 
4082 /**
4083  * start_renaming_two_dentries - Lock to dentries in given parents for rename
4084  * @rd:           rename data containing parent
4085  * @old_dentry:   dentry of name to move
4086  * @new_dentry:   dentry to move to
4087  *
4088  * Ensure locks are in place for rename and check parentage is still correct.
4089  *
4090  * On success the two dentries are stored in @rd.old_dentry and
4091  * @rd.new_dentry and @rd.old_parent and @rd.new_parent are confirmed to
4092  * be the parents of the dentries.
4093  *
4094  * References and the lock can be dropped with end_renaming()
4095  *
4096  * Returns: zero or an error.
4097  */
4098 int
4099 start_renaming_two_dentries(struct renamedata *rd,
4100 			    struct dentry *old_dentry, struct dentry *new_dentry)
4101 {
4102 	struct dentry *trap;
4103 	int err;
4104 
4105 	/* Already have the dentry - need to be sure to lock the correct parent */
4106 	trap = lock_rename_child(old_dentry, rd->new_parent);
4107 	if (IS_ERR(trap))
4108 		return PTR_ERR(trap);
4109 	err = -EINVAL;
4110 	if (d_unhashed(old_dentry) ||
4111 	    (rd->old_parent && rd->old_parent != old_dentry->d_parent))
4112 		/* old_dentry was removed, or moved and explicit parent requested */
4113 		goto out_unlock;
4114 	if (d_unhashed(new_dentry) ||
4115 	    rd->new_parent != new_dentry->d_parent)
4116 		/* new_dentry was removed or moved */
4117 		goto out_unlock;
4118 
4119 	if (old_dentry == trap)
4120 		/* source is an ancestor of target */
4121 		goto out_unlock;
4122 
4123 	if (new_dentry == trap) {
4124 		/* target is an ancestor of source */
4125 		if (rd->flags & RENAME_EXCHANGE)
4126 			err = -EINVAL;
4127 		else
4128 			err = -ENOTEMPTY;
4129 		goto out_unlock;
4130 	}
4131 
4132 	err = -EEXIST;
4133 	if (d_is_positive(new_dentry) && (rd->flags & RENAME_NOREPLACE))
4134 		goto out_unlock;
4135 
4136 	rd->old_dentry = dget(old_dentry);
4137 	rd->new_dentry = dget(new_dentry);
4138 	rd->old_parent = dget(old_dentry->d_parent);
4139 	return 0;
4140 
4141 out_unlock:
4142 	unlock_rename(old_dentry->d_parent, rd->new_parent);
4143 	return err;
4144 }
4145 EXPORT_SYMBOL(start_renaming_two_dentries);
4146 
4147 void end_renaming(struct renamedata *rd)
4148 {
4149 	unlock_rename(rd->old_parent, rd->new_parent);
4150 	dput(rd->old_dentry);
4151 	dput(rd->new_dentry);
4152 	dput(rd->old_parent);
4153 }
4154 EXPORT_SYMBOL(end_renaming);
4155 
4156 /**
4157  * vfs_prepare_mode - prepare the mode to be used for a new inode
4158  * @idmap:	idmap of the mount the inode was found from
4159  * @dir:	parent directory of the new inode
4160  * @mode:	mode of the new inode
4161  * @mask_perms:	allowed permission by the vfs
4162  * @type:	type of file to be created
4163  *
4164  * This helper consolidates and enforces vfs restrictions on the @mode of a new
4165  * object to be created.
4166  *
4167  * Umask stripping depends on whether the filesystem supports POSIX ACLs (see
4168  * the kernel documentation for mode_strip_umask()). Moving umask stripping
4169  * after setgid stripping allows the same ordering for both non-POSIX ACL and
4170  * POSIX ACL supporting filesystems.
4171  *
4172  * Returns: mode to be passed to the filesystem
4173  */
4174 static inline umode_t vfs_prepare_mode(struct mnt_idmap *idmap,
4175 				       const struct inode *dir, umode_t mode,
4176 				       umode_t mask_perms, umode_t type)
4177 {
4178 	mode = mode_strip_sgid(idmap, dir, mode);
4179 	mode = mode_strip_umask(dir, mode);
4180 
4181 	/*
4182 	 * Apply the vfs mandated allowed permission mask and set the type of
4183 	 * file to be created before we call into the filesystem.
4184 	 */
4185 	mode &= (mask_perms & ~S_IFMT);
4186 	mode |= (type & S_IFMT);
4187 
4188 	return mode;
4189 }
4190 
4191 /**
4192  * vfs_create - create new file
4193  * @idmap:	idmap of the mount the inode was found from
4194  * @dentry:	dentry of the child file
4195  * @mode:	mode of the child file
4196  * @di:		returns parent inode, if the inode is delegated.
4197  *
4198  * Create a new file.
4199  *
4200  * If the inode has been found through an idmapped mount the idmap of
4201  * the vfsmount must be passed through @idmap. This function will then take
4202  * care to map the inode according to @idmap before checking permissions.
4203  * On non-idmapped mounts or if permission checking is to be performed on the
4204  * raw inode simply pass @nop_mnt_idmap.
4205  */
4206 int vfs_create(struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode,
4207 	       struct delegated_inode *di)
4208 {
4209 	struct inode *dir = d_inode(dentry->d_parent);
4210 	int error;
4211 
4212 	error = may_create_dentry(idmap, dir, dentry);
4213 	if (error)
4214 		return error;
4215 
4216 	if (!dir->i_op->create)
4217 		return -EACCES;	/* shouldn't it be ENOSYS? */
4218 
4219 	mode = vfs_prepare_mode(idmap, dir, mode, S_IALLUGO, S_IFREG);
4220 	error = security_inode_create(dir, dentry, mode);
4221 	if (error)
4222 		return error;
4223 	error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, di);
4224 	if (error)
4225 		return error;
4226 	error = dir->i_op->create(idmap, dir, dentry, mode);
4227 	if (!error)
4228 		fsnotify_create(dir, dentry);
4229 	return error;
4230 }
4231 EXPORT_SYMBOL(vfs_create);
4232 
4233 int vfs_mkobj(struct dentry *dentry, umode_t mode,
4234 		int (*f)(struct dentry *, umode_t, void *),
4235 		void *arg)
4236 {
4237 	struct inode *dir = dentry->d_parent->d_inode;
4238 	int error = may_create_dentry(&nop_mnt_idmap, dir, dentry);
4239 	if (error)
4240 		return error;
4241 
4242 	mode &= S_IALLUGO;
4243 	mode |= S_IFREG;
4244 	error = security_inode_create(dir, dentry, mode);
4245 	if (error)
4246 		return error;
4247 	error = f(dentry, mode, arg);
4248 	if (!error)
4249 		fsnotify_create(dir, dentry);
4250 	return error;
4251 }
4252 EXPORT_SYMBOL(vfs_mkobj);
4253 
4254 bool may_open_dev(const struct path *path)
4255 {
4256 	return !(path->mnt->mnt_flags & MNT_NODEV) &&
4257 		!(path->mnt->mnt_sb->s_iflags & SB_I_NODEV);
4258 }
4259 
4260 static int may_open(struct mnt_idmap *idmap, const struct path *path,
4261 		    int acc_mode, int flag)
4262 {
4263 	struct dentry *dentry = path->dentry;
4264 	struct inode *inode = dentry->d_inode;
4265 	int error;
4266 
4267 	if (!inode)
4268 		return -ENOENT;
4269 
4270 	switch (inode->i_mode & S_IFMT) {
4271 	case S_IFLNK:
4272 		return -ELOOP;
4273 	case S_IFDIR:
4274 		if (acc_mode & MAY_WRITE)
4275 			return -EISDIR;
4276 		if (acc_mode & MAY_EXEC)
4277 			return -EACCES;
4278 		break;
4279 	case S_IFBLK:
4280 	case S_IFCHR:
4281 		if (!may_open_dev(path))
4282 			return -EACCES;
4283 		fallthrough;
4284 	case S_IFIFO:
4285 	case S_IFSOCK:
4286 		if (acc_mode & MAY_EXEC)
4287 			return -EACCES;
4288 		flag &= ~O_TRUNC;
4289 		break;
4290 	case S_IFREG:
4291 		if ((acc_mode & MAY_EXEC) && path_noexec(path))
4292 			return -EACCES;
4293 		break;
4294 	default:
4295 		VFS_BUG_ON_INODE(!IS_ANON_FILE(inode), inode);
4296 	}
4297 
4298 	error = inode_permission(idmap, inode, MAY_OPEN | acc_mode);
4299 	if (error)
4300 		return error;
4301 
4302 	/*
4303 	 * An append-only file must be opened in append mode for writing.
4304 	 */
4305 	if (IS_APPEND(inode)) {
4306 		if  ((flag & O_ACCMODE) != O_RDONLY && !(flag & O_APPEND))
4307 			return -EPERM;
4308 		if (flag & O_TRUNC)
4309 			return -EPERM;
4310 	}
4311 
4312 	/* O_NOATIME can only be set by the owner or superuser */
4313 	if (flag & O_NOATIME && !inode_owner_or_capable(idmap, inode))
4314 		return -EPERM;
4315 
4316 	return 0;
4317 }
4318 
4319 static int handle_truncate(struct mnt_idmap *idmap, struct file *filp)
4320 {
4321 	const struct path *path = &filp->f_path;
4322 	struct inode *inode = path->dentry->d_inode;
4323 	int error = get_write_access(inode);
4324 	if (error)
4325 		return error;
4326 
4327 	error = security_file_truncate(filp);
4328 	if (!error) {
4329 		error = do_truncate(idmap, path->dentry, 0,
4330 				    ATTR_MTIME|ATTR_CTIME|ATTR_OPEN,
4331 				    filp);
4332 	}
4333 	put_write_access(inode);
4334 	return error;
4335 }
4336 
4337 static inline int open_to_namei_flags(int flag)
4338 {
4339 	if ((flag & O_ACCMODE) == 3)
4340 		flag--;
4341 	return flag;
4342 }
4343 
4344 static int may_o_create(struct mnt_idmap *idmap,
4345 			const struct path *dir, struct dentry *dentry,
4346 			umode_t mode)
4347 {
4348 	int error = security_path_mknod(dir, dentry, mode, 0);
4349 	if (error)
4350 		return error;
4351 
4352 	if (!fsuidgid_has_mapping(dir->dentry->d_sb, idmap))
4353 		return -EOVERFLOW;
4354 
4355 	error = inode_permission(idmap, dir->dentry->d_inode,
4356 				 MAY_WRITE | MAY_EXEC);
4357 	if (error)
4358 		return error;
4359 
4360 	return security_inode_create(dir->dentry->d_inode, dentry, mode);
4361 }
4362 
4363 /**
4364  * atomic_open() - atomically look up, create and open a file
4365  * @path:          parent directory path
4366  * @dentry:        child to ->atomic_open()
4367  * @file:          file to attach child to
4368  * @open_flag:     open flags
4369  * @mode:          create mode
4370  * @create_error:  return value from may_o_create()
4371  *
4372  * Attempt to look up, create and open @dentry, which must be negative, in a
4373  * single call into the filesystem.
4374  *
4375  * If a non-error dentry is returned then: when FMODE_OPENED is set,
4376  * the file will have been attached to @file by the filesystem calling
4377  * finish_open(). If FMODE_OPENED isn't set, the filesystem instead called
4378  * finish_no_open() and the caller will need to perform the open themselves.
4379  *
4380  * FMODE_CREATED is set when the call to ->atomic_open() actually created
4381  * the file.
4382  *
4383  * Returns: the opened or looked-up dentry, or ERR_PTR() on failure.  The
4384  * reference to @dentry is consumed in either case.
4385  */
4386 static struct dentry *atomic_open(const struct path *path, struct dentry *dentry,
4387 				  struct file *file,
4388 				  int open_flag, umode_t mode, int create_error)
4389 {
4390 	struct dentry *const DENTRY_NOT_SET = (void *) -1UL;
4391 	struct inode *dir_inode = path->dentry->d_inode;
4392 	int error;
4393 
4394 	file->__f_path.dentry = DENTRY_NOT_SET;
4395 	file->__f_path.mnt = path->mnt;
4396 	error = dir_inode->i_op->atomic_open(dir_inode, dentry, file,
4397 				       open_to_namei_flags(open_flag), mode);
4398 	d_lookup_done(dentry);
4399 
4400 	if (!error) {
4401 		if (file->f_mode & FMODE_OPENED) {
4402 			/* finish_open() called */
4403 			struct dentry *opened = file->f_path.dentry;
4404 
4405 			if (unlikely(opened != dentry)) {
4406 				dput(dentry);
4407 				dentry = dget(opened);
4408 			}
4409 		} else if (likely(file->f_path.dentry != DENTRY_NOT_SET)) {
4410 			/* finish_no_open() called */
4411 			struct dentry *replaced = file->f_path.dentry;
4412 
4413 			if (replaced) {
4414 				dput(dentry);
4415 				dentry = replaced;
4416 			}
4417 			if (unlikely(d_is_negative(dentry)))
4418 				error = -ENOENT;
4419 		} else {
4420 			const char *fsname = dentry->d_sb->s_type->name;
4421 
4422 			WARN(1, "%s: ->atomic_open() left file->f_path.dentry unset!\n",
4423 			     fsname);
4424 			error = -EIO;
4425 		}
4426 	}
4427 
4428 	if (error) {
4429 		if (unlikely(create_error) && error == -ENOENT) {
4430 			/*
4431 			 * Should have done a create, but errored before.
4432 			 * Some filesystems return -ENOENT directly instead of
4433 			 * calling finish_no_open() with a negative dentry;
4434 			 * either way it should only mean the child doesn't exist,
4435 			 * so a refused create is safe to record here.
4436 			 */
4437 			audit_inode_child(dir_inode, dentry, AUDIT_TYPE_CHILD_CREATE);
4438 			error = create_error;
4439 		}
4440 		dput(dentry);
4441 		dentry = ERR_PTR(error);
4442 	}
4443 	return dentry;
4444 }
4445 
4446 /*
4447  * Look up and maybe create and open the last component.
4448  *
4449  * Takes the parent inode lock itself, exclusive if O_CREAT was requested and
4450  * shared otherwise, and drops it again before returning.  The caller must not
4451  * hold it.
4452  *
4453  * On success returns the dentry of the last component.  If FMODE_OPENED is set
4454  * on file->f_mode the file was also opened and attached to @file; otherwise
4455  * only lookup and creation were performed and the caller has to open it.  In
4456  * the latter case the dentry may be negative if O_CREAT hadn't been specified.
4457  *
4458  * Returns ERR_PTR() on failure.
4459  */
4460 static struct dentry *lookup_open(struct nameidata *nd, struct file *file,
4461 				  const struct open_flags *op)
4462 {
4463 	struct delegated_inode delegated_inode = { };
4464 	struct mnt_idmap *idmap;
4465 	struct dentry *dir = nd->path.dentry;
4466 	struct inode *dir_inode = dir->d_inode;
4467 	int open_flag;
4468 	struct dentry *dentry;
4469 	int error, create_error;
4470 	umode_t mode;
4471 	bool got_write;
4472 
4473 retry:
4474 	open_flag = op->open_flag;
4475 	got_write = false;
4476 	mode = op->mode;
4477 	create_error = 0;
4478 
4479 	if (open_flag & (O_CREAT | O_TRUNC | O_WRONLY | O_RDWR)) {
4480 		got_write = !mnt_want_write(nd->path.mnt);
4481 		/*
4482 		 * do _not_ fail yet - we might not need that or fail with
4483 		 * a different error; we'll be dropping this one anyway.
4484 		 */
4485 	}
4486 	if (open_flag & O_CREAT)
4487 		inode_lock(dir_inode);
4488 	else
4489 		inode_lock_shared(dir_inode);
4490 
4491 	if (unlikely(IS_DEADDIR(dir_inode))) {
4492 		dentry = ERR_PTR(-ENOENT);
4493 		goto out;
4494 	}
4495 
4496 	file->f_mode &= ~FMODE_CREATED;
4497 	dentry = d_lookup(dir, &nd->last);
4498 	for (;;) {
4499 		if (!dentry) {
4500 			dentry = d_alloc_parallel(dir, &nd->last);
4501 			if (IS_ERR(dentry))
4502 				goto out;
4503 		}
4504 		if (d_in_lookup(dentry))
4505 			break;
4506 
4507 		error = d_revalidate(dir_inode, &nd->last, dentry, nd->flags);
4508 		if (likely(error > 0))
4509 			break;
4510 		if (error)
4511 			goto out_dput;
4512 		d_invalidate(dentry);
4513 		dput(dentry);
4514 		dentry = NULL;
4515 	}
4516 	if (dentry->d_inode) {
4517 		/* Cached positive dentry: will open in do_open(). */
4518 		goto out;
4519 	}
4520 
4521 	if (open_flag & O_CREAT)
4522 		audit_inode(nd->name, dir, AUDIT_INODE_PARENT);
4523 
4524 	/*
4525 	 * Checking write permission is tricky, bacuse we don't know if we are
4526 	 * going to actually need it: O_CREAT opens should work as long as the
4527 	 * file exists.  But checking existence breaks atomicity.  The trick is
4528 	 * to check access and if not granted clear O_CREAT from the flags.
4529 	 *
4530 	 * Another problem is returing the "right" error value (e.g. for an
4531 	 * O_EXCL open we want to return EEXIST not EROFS).
4532 	 */
4533 	if (unlikely(!got_write))
4534 		open_flag &= ~O_TRUNC;
4535 	idmap = mnt_idmap(nd->path.mnt);
4536 	if (open_flag & O_CREAT) {
4537 		if (open_flag & O_EXCL)
4538 			open_flag &= ~O_TRUNC;
4539 		mode = vfs_prepare_mode(idmap, dir_inode, mode, mode, mode);
4540 		if (likely(got_write))
4541 			create_error = may_o_create(idmap, &nd->path,
4542 						    dentry, mode);
4543 		else
4544 			create_error = -EROFS;
4545 	}
4546 	if (create_error)
4547 		open_flag &= ~O_CREAT;
4548 	if (dir_inode->i_op->atomic_open) {
4549 		if (nd->flags & LOOKUP_DIRECTORY)
4550 			open_flag |= O_DIRECTORY;
4551 		dentry = atomic_open(&nd->path, dentry, file, open_flag, mode,
4552 				     create_error);
4553 		goto out;
4554 	}
4555 
4556 	if (d_in_lookup(dentry)) {
4557 		struct dentry *res = dir_inode->i_op->lookup(dir_inode, dentry,
4558 							     nd->flags);
4559 		d_lookup_done(dentry);
4560 		if (unlikely(res)) {
4561 			if (IS_ERR(res)) {
4562 				error = PTR_ERR(res);
4563 				goto out_dput;
4564 			}
4565 			dput(dentry);
4566 			dentry = res;
4567 		}
4568 	}
4569 	if (dentry->d_inode || !(op->open_flag & O_CREAT)) {
4570 		/*
4571 		 * No need to create a file.  If lookup returned a positive
4572 		 * dentry, the file will be opened in do_open().
4573 		 */
4574 		goto out;
4575 	}
4576 
4577 	/* Negative dentry with O_CREAT flag set */
4578 	audit_inode_child(dir_inode, dentry, AUDIT_TYPE_CHILD_CREATE);
4579 
4580 	if (unlikely(create_error)) {
4581 		/* should have done a create, but we already errored */
4582 		error = create_error;
4583 		goto out_dput;
4584 	}
4585 
4586 	error = try_break_deleg(dir_inode, LEASE_BREAK_DIR_CREATE, &delegated_inode);
4587 	if (error)
4588 		goto out_dput;
4589 
4590 	file->f_mode |= FMODE_CREATED;
4591 	if (!dir_inode->i_op->create) {
4592 		error = -EACCES;
4593 		goto out_dput;
4594 	}
4595 
4596 	error = dir_inode->i_op->create(idmap, dir_inode, dentry, mode);
4597 	if (error)
4598 		goto out_dput;
4599 out:
4600 	if (!IS_ERR(dentry)) {
4601 		if (file->f_mode & FMODE_CREATED)
4602 			fsnotify_create(dir_inode, dentry);
4603 		if (file->f_mode & FMODE_OPENED)
4604 			fsnotify_open(file);
4605 	}
4606 	if ((open_flag & O_CREAT) || create_error)
4607 		inode_unlock(dir_inode);
4608 	else
4609 		inode_unlock_shared(dir_inode);
4610 
4611 	if (got_write)
4612 		mnt_drop_write(nd->path.mnt);
4613 
4614 	if (is_delegated(&delegated_inode)) {
4615 		/* Must have come through out_dput: dentry is an ERR_PTR() */
4616 		error = break_deleg_wait(&delegated_inode);
4617 
4618 		if (!error)
4619 			goto retry;
4620 		dentry = ERR_PTR(error);
4621 	}
4622 
4623 	return dentry;
4624 
4625 out_dput:
4626 	dput(dentry);
4627 	dentry = ERR_PTR(error);
4628 	goto out;
4629 }
4630 
4631 /**
4632  * vfs_lookup_open - open and possibly create a regular file
4633  * @parent: directory to contain file
4634  * @last: final component of file name
4635  * @open_flag: O_flags
4636  * @mode: initial permissions for file
4637  *
4638  * Open a file after lookup and/or create.  This provides similar
4639  * functionality to open_last_lookups() for non-VFS users, particularly
4640  * nfsd.
4641  * It uses ->atomic_open or ->lookup / ->create / ->open as appropriate.
4642  *
4643  * If the fs object found is not a regular file then an error is returned.
4644  * In some cases, related errors are repurposed so that the caller can
4645  * determine the type of file found from the error.
4646  * -EISDIR : a directory was found
4647  * -ELOOP  : a symlink was found
4648  * -ENODEV : a block or character device special file was found
4649  * -EFTYPE : any other non-regular file was found, such as FIFO or SOCK.
4650  *           or ->atomic_open responded to __O_REGULAR.
4651  *
4652  * Returns: the opened struct file, or an error.
4653  */
4654 struct file *vfs_lookup_open(struct path *parent, struct qstr *last,
4655 			     int open_flag, umode_t mode)
4656 {
4657 	struct file *file __free(fput) = NULL;
4658 	struct nameidata nd = {};
4659 	struct open_flags op = {};
4660 	struct dentry *dentry;
4661 	int error = 0;
4662 
4663 	WARN_ONCE(mode & ~S_IALLUGO, "mode must only have permission bits");
4664 	WARN_ONCE(open_flag & ~(O_ACCMODE|O_CREAT|O_EXCL|O_TRUNC|__O_REGULAR),
4665 		  "open_flag has unsupported flags");
4666 
4667 	mode |= S_IFREG;
4668 	open_flag |= __O_REGULAR;
4669 
4670 	error = lookup_noperm_common(last, parent->dentry);
4671 	if (error)
4672 		return ERR_PTR(error);
4673 
4674 	file = alloc_empty_file(open_flag, current_cred());
4675 	if (IS_ERR(file))
4676 		return file;
4677 
4678 	nd.path = *parent;
4679 	nd.last = *last;
4680 	nd.flags = LOOKUP_OPEN;
4681 	if (open_flag & O_CREAT) {
4682 		nd.flags |= LOOKUP_CREATE;
4683 		if (open_flag & O_EXCL)
4684 			nd.flags |= LOOKUP_EXCL;
4685 	}
4686 	op.open_flag = open_flag;
4687 	op.mode = mode;
4688 	dentry = lookup_open(&nd, file, &op);
4689 
4690 	if (IS_ERR(dentry))
4691 		return ERR_CAST(dentry);
4692 
4693 	if (d_really_is_negative(dentry)) {
4694 		error = -ENOENT;
4695 	} else if (!(file->f_mode & FMODE_CREATED) && (open_flag & O_EXCL)) {
4696 		error = -EEXIST;
4697 	} else if ((dentry->d_inode->i_mode & S_IFMT) != S_IFREG) {
4698 		switch (dentry->d_inode->i_mode & S_IFMT) {
4699 		case S_IFDIR:
4700 			error = -EISDIR;
4701 			break;
4702 		case S_IFLNK:
4703 			error = -ELOOP;
4704 			break;
4705 		case S_IFBLK:
4706 		case S_IFCHR:
4707 			error = -ENODEV;
4708 			break;
4709 		case S_IFIFO:
4710 		case S_IFSOCK:
4711 		default:
4712 			error = -EFTYPE;
4713 			break;
4714 		}
4715 	} else if (!(file->f_mode & FMODE_OPENED)) {
4716 		nd.path.dentry = dentry;
4717 		error = vfs_open(&nd.path, file);
4718 	}
4719 	dput(dentry);
4720 
4721 	if (error)
4722 		return ERR_PTR(error);
4723 	return no_free_ptr(file);
4724 }
4725 EXPORT_SYMBOL_FOR_MODULES(vfs_lookup_open, "nfsd");
4726 
4727 static inline bool trailing_slashes(struct nameidata *nd)
4728 {
4729 	return (bool)nd->last.name[nd->last.len];
4730 }
4731 
4732 static struct dentry *lookup_fast_for_open(struct nameidata *nd, int open_flag)
4733 {
4734 	struct dentry *dentry;
4735 
4736 	if (open_flag & O_CREAT) {
4737 		if (trailing_slashes(nd))
4738 			return ERR_PTR(-EISDIR);
4739 
4740 		/* Don't bother on an O_EXCL create */
4741 		if (open_flag & O_EXCL)
4742 			return NULL;
4743 	}
4744 
4745 	if (trailing_slashes(nd))
4746 		nd->flags |= LOOKUP_FOLLOW | LOOKUP_DIRECTORY;
4747 
4748 	dentry = lookup_fast(nd);
4749 	if (IS_ERR_OR_NULL(dentry))
4750 		return dentry;
4751 
4752 	if (open_flag & O_CREAT) {
4753 		/* Discard negative dentries. Need inode_lock to do the create */
4754 		if (!dentry->d_inode) {
4755 			if (!(nd->flags & LOOKUP_RCU))
4756 				dput(dentry);
4757 			dentry = NULL;
4758 		}
4759 	}
4760 	return dentry;
4761 }
4762 
4763 static const char *open_last_lookups(struct nameidata *nd,
4764 		   struct file *file, const struct open_flags *op)
4765 {
4766 	int open_flag = op->open_flag;
4767 	struct dentry *dentry;
4768 	const char *res;
4769 
4770 	nd->flags |= op->intent;
4771 
4772 	if (nd->last_type != LAST_NORM) {
4773 		if (nd->depth)
4774 			put_link(nd);
4775 		return handle_dots(nd, nd->last_type);
4776 	}
4777 
4778 	/* We _can_ be in RCU mode here */
4779 	dentry = lookup_fast_for_open(nd, open_flag);
4780 	if (IS_ERR(dentry))
4781 		return ERR_CAST(dentry);
4782 
4783 	if (likely(dentry))
4784 		goto finish_lookup;
4785 
4786 	if (!(open_flag & O_CREAT)) {
4787 		if (WARN_ON_ONCE(nd->flags & LOOKUP_RCU))
4788 			return ERR_PTR(-ECHILD);
4789 	} else {
4790 		if (nd->flags & LOOKUP_RCU) {
4791 			if (!try_to_unlazy(nd))
4792 				return ERR_PTR(-ECHILD);
4793 		}
4794 	}
4795 
4796 	dentry = lookup_open(nd, file, op);
4797 	if (IS_ERR(dentry))
4798 		return ERR_CAST(dentry);
4799 
4800 	if (file->f_mode & (FMODE_OPENED | FMODE_CREATED)) {
4801 		dput(nd->path.dentry);
4802 		nd->path.dentry = dentry;
4803 		return NULL;
4804 	}
4805 
4806 finish_lookup:
4807 	if (nd->depth)
4808 		put_link(nd);
4809 	res = step_into(nd, WALK_TRAILING, dentry);
4810 	if (unlikely(res))
4811 		nd->flags &= ~(LOOKUP_OPEN|LOOKUP_CREATE|LOOKUP_EXCL);
4812 	return res;
4813 }
4814 
4815 /*
4816  * Handle the last step of open()
4817  */
4818 static int do_open(struct nameidata *nd,
4819 		   struct file *file, const struct open_flags *op)
4820 {
4821 	struct mnt_idmap *idmap;
4822 	int open_flag = op->open_flag;
4823 	bool do_truncate;
4824 	int acc_mode;
4825 	int error;
4826 
4827 	if (!(file->f_mode & (FMODE_OPENED | FMODE_CREATED))) {
4828 		error = complete_walk(nd);
4829 		if (error)
4830 			return error;
4831 	}
4832 	if (!(file->f_mode & FMODE_CREATED))
4833 		audit_inode(nd->name, nd->path.dentry, 0);
4834 	idmap = mnt_idmap(nd->path.mnt);
4835 	if (open_flag & O_CREAT) {
4836 		if ((open_flag & O_EXCL) && !(file->f_mode & FMODE_CREATED))
4837 			return -EEXIST;
4838 		if (d_is_dir(nd->path.dentry))
4839 			return -EISDIR;
4840 		error = may_create_in_sticky(idmap, nd,
4841 					     d_backing_inode(nd->path.dentry));
4842 		if (unlikely(error))
4843 			return error;
4844 	}
4845 
4846 	if ((open_flag & __O_REGULAR) && !d_is_reg(nd->path.dentry))
4847 		return -EFTYPE;
4848 
4849 	if ((nd->flags & LOOKUP_DIRECTORY) && !d_can_lookup(nd->path.dentry))
4850 		return -ENOTDIR;
4851 
4852 	do_truncate = false;
4853 	acc_mode = op->acc_mode;
4854 	if (file->f_mode & FMODE_CREATED) {
4855 		/* Don't check for write permission, don't truncate */
4856 		open_flag &= ~O_TRUNC;
4857 		acc_mode = 0;
4858 	} else if (d_is_reg(nd->path.dentry) && open_flag & O_TRUNC) {
4859 		error = mnt_want_write(nd->path.mnt);
4860 		if (error)
4861 			return error;
4862 		do_truncate = true;
4863 	}
4864 	error = may_open(idmap, &nd->path, acc_mode, open_flag);
4865 	if (!error && !(file->f_mode & FMODE_OPENED))
4866 		error = vfs_open(&nd->path, file);
4867 	if (!error)
4868 		error = security_file_post_open(file, op->acc_mode);
4869 	if (!error && do_truncate)
4870 		error = handle_truncate(idmap, file);
4871 	if (unlikely(error > 0)) {
4872 		WARN_ON(1);
4873 		error = -EINVAL;
4874 	}
4875 	if (do_truncate)
4876 		mnt_drop_write(nd->path.mnt);
4877 	return error;
4878 }
4879 
4880 /**
4881  * vfs_tmpfile - create tmpfile
4882  * @idmap:	idmap of the mount the inode was found from
4883  * @parentpath:	pointer to the path of the base directory
4884  * @file:	file descriptor of the new tmpfile
4885  * @mode:	mode of the new tmpfile
4886  *
4887  * Create a temporary file.
4888  *
4889  * If the inode has been found through an idmapped mount the idmap of
4890  * the vfsmount must be passed through @idmap. This function will then take
4891  * care to map the inode according to @idmap before checking permissions.
4892  * On non-idmapped mounts or if permission checking is to be performed on the
4893  * raw inode simply pass @nop_mnt_idmap.
4894  */
4895 int vfs_tmpfile(struct mnt_idmap *idmap,
4896 		const struct path *parentpath,
4897 		struct file *file, umode_t mode)
4898 {
4899 	struct dentry *child;
4900 	struct inode *dir = d_inode(parentpath->dentry);
4901 	struct inode *inode;
4902 	int error;
4903 	int open_flag = file->f_flags;
4904 
4905 	/* A tmpfile is I_LINKABLE, so guard its owner like may_o_create(). */
4906 	if (!fsuidgid_has_mapping(dir->i_sb, idmap))
4907 		return -EOVERFLOW;
4908 
4909 	/* we want directory to be writable */
4910 	error = inode_permission(idmap, dir, MAY_WRITE | MAY_EXEC);
4911 	if (error)
4912 		return error;
4913 	if (!dir->i_op->tmpfile)
4914 		return -EOPNOTSUPP;
4915 	child = d_alloc(parentpath->dentry, &slash_name);
4916 	if (unlikely(!child))
4917 		return -ENOMEM;
4918 	file->__f_path.mnt = parentpath->mnt;
4919 	file->__f_path.dentry = child;
4920 	mode = vfs_prepare_mode(idmap, dir, mode, mode, mode);
4921 	error = dir->i_op->tmpfile(idmap, dir, file, mode);
4922 	dput(child);
4923 	if (file->f_mode & FMODE_OPENED)
4924 		fsnotify_open(file);
4925 	if (error)
4926 		return error;
4927 	/* Don't check for other permissions, the inode was just created */
4928 	error = may_open(idmap, &file->f_path, 0, file->f_flags);
4929 	if (error)
4930 		return error;
4931 	inode = file_inode(file);
4932 	if (!(open_flag & O_EXCL)) {
4933 		spin_lock(&inode->i_lock);
4934 		inode_state_set(inode, I_LINKABLE);
4935 		spin_unlock(&inode->i_lock);
4936 	}
4937 	security_inode_post_create_tmpfile(idmap, inode);
4938 	return 0;
4939 }
4940 
4941 /**
4942  * kernel_tmpfile_open - open a tmpfile for kernel internal use
4943  * @idmap:	idmap of the mount the inode was found from
4944  * @parentpath:	path of the base directory
4945  * @mode:	mode of the new tmpfile
4946  * @open_flag:	flags
4947  * @cred:	credentials for open
4948  *
4949  * Create and open a temporary file.  The file is not accounted in nr_files,
4950  * hence this is only for kernel internal use, and must not be installed into
4951  * file tables or such.
4952  */
4953 struct file *kernel_tmpfile_open(struct mnt_idmap *idmap,
4954 				 const struct path *parentpath,
4955 				 umode_t mode, int open_flag,
4956 				 const struct cred *cred)
4957 {
4958 	struct file *file;
4959 	int error;
4960 
4961 	file = alloc_empty_file_noaccount(open_flag, cred);
4962 	if (IS_ERR(file))
4963 		return file;
4964 
4965 	error = vfs_tmpfile(idmap, parentpath, file, mode);
4966 	if (error) {
4967 		fput(file);
4968 		file = ERR_PTR(error);
4969 	}
4970 	return file;
4971 }
4972 EXPORT_SYMBOL(kernel_tmpfile_open);
4973 
4974 static int do_tmpfile(struct nameidata *nd, unsigned flags,
4975 		const struct open_flags *op,
4976 		struct file *file)
4977 {
4978 	struct path path;
4979 	int error = path_lookupat(nd, flags | LOOKUP_DIRECTORY, &path);
4980 
4981 	if (unlikely(error))
4982 		return error;
4983 	error = mnt_want_write(path.mnt);
4984 	if (unlikely(error))
4985 		goto out;
4986 	error = vfs_tmpfile(mnt_idmap(path.mnt), &path, file, op->mode);
4987 	if (error)
4988 		goto out2;
4989 	audit_inode(nd->name, file->f_path.dentry, 0);
4990 out2:
4991 	mnt_drop_write(path.mnt);
4992 out:
4993 	path_put(&path);
4994 	return error;
4995 }
4996 
4997 static int do_o_path(struct nameidata *nd, unsigned flags, struct file *file)
4998 {
4999 	struct path path;
5000 	int error = path_lookupat(nd, flags, &path);
5001 	if (!error) {
5002 		audit_inode(nd->name, path.dentry, 0);
5003 		error = vfs_open(&path, file);
5004 		path_put(&path);
5005 	}
5006 	return error;
5007 }
5008 
5009 static struct file *path_openat(struct nameidata *nd,
5010 			const struct open_flags *op, unsigned flags)
5011 {
5012 	struct file *file;
5013 	int error;
5014 
5015 	file = alloc_empty_file(op->open_flag, current_cred());
5016 	if (IS_ERR(file))
5017 		return file;
5018 
5019 	if (unlikely(file->f_flags & __O_TMPFILE)) {
5020 		error = do_tmpfile(nd, flags, op, file);
5021 	} else if (unlikely(file->f_flags & O_PATH)) {
5022 		error = do_o_path(nd, flags, file);
5023 	} else {
5024 		const char *s = path_init(nd, flags);
5025 		while (!(error = link_path_walk(s, nd)) &&
5026 		       (s = open_last_lookups(nd, file, op)) != NULL)
5027 			;
5028 		if (!error)
5029 			error = do_open(nd, file, op);
5030 		terminate_walk(nd);
5031 	}
5032 	if (likely(!error)) {
5033 		if (likely(file->f_mode & FMODE_OPENED))
5034 			return file;
5035 		WARN_ON(1);
5036 		error = -EINVAL;
5037 	}
5038 	fput_close(file);
5039 	if (error == -EOPENSTALE) {
5040 		if (flags & LOOKUP_RCU)
5041 			error = -ECHILD;
5042 		else
5043 			error = -ESTALE;
5044 	}
5045 	return ERR_PTR(error);
5046 }
5047 
5048 struct file *do_file_open(int dfd, struct filename *pathname,
5049 		const struct open_flags *op)
5050 {
5051 	struct nameidata nd;
5052 	int flags = op->lookup_flags;
5053 	struct file *filp;
5054 
5055 	if (IS_ERR(pathname))
5056 		return ERR_CAST(pathname);
5057 	set_nameidata(&nd, dfd, pathname, NULL);
5058 	filp = path_openat(&nd, op, flags | LOOKUP_RCU);
5059 	if (unlikely(filp == ERR_PTR(-ECHILD)))
5060 		filp = path_openat(&nd, op, flags);
5061 	if (unlikely(filp == ERR_PTR(-ESTALE)))
5062 		filp = path_openat(&nd, op, flags | LOOKUP_REVAL);
5063 	restore_nameidata();
5064 	return filp;
5065 }
5066 
5067 struct file *do_file_open_root(const struct path *root,
5068 		const char *name, const struct open_flags *op)
5069 {
5070 	struct nameidata nd;
5071 	struct file *file;
5072 	int flags = op->lookup_flags;
5073 
5074 	if (d_is_symlink(root->dentry) && op->intent & LOOKUP_OPEN)
5075 		return ERR_PTR(-ELOOP);
5076 
5077 	CLASS(filename_kernel, filename)(name);
5078 	if (IS_ERR(filename))
5079 		return ERR_CAST(filename);
5080 
5081 	set_nameidata(&nd, -1, filename, root);
5082 	file = path_openat(&nd, op, flags | LOOKUP_RCU);
5083 	if (unlikely(file == ERR_PTR(-ECHILD)))
5084 		file = path_openat(&nd, op, flags);
5085 	if (unlikely(file == ERR_PTR(-ESTALE)))
5086 		file = path_openat(&nd, op, flags | LOOKUP_REVAL);
5087 	restore_nameidata();
5088 	return file;
5089 }
5090 
5091 static struct dentry *filename_create(int dfd, struct filename *name,
5092 				      struct path *path, unsigned int lookup_flags)
5093 {
5094 	struct dentry *dentry = ERR_PTR(-EEXIST);
5095 	struct qstr last;
5096 	bool want_dir = lookup_flags & LOOKUP_DIRECTORY;
5097 	unsigned int reval_flag = lookup_flags & LOOKUP_REVAL;
5098 	unsigned int create_flags = LOOKUP_CREATE | LOOKUP_EXCL;
5099 	enum last_type type;
5100 	int error;
5101 
5102 	error = filename_parentat(dfd, name, reval_flag, path, &last, &type);
5103 	if (error)
5104 		return ERR_PTR(error);
5105 
5106 	/*
5107 	 * Yucky last component or no last component at all?
5108 	 * (foo/., foo/.., /////)
5109 	 */
5110 	if (unlikely(type != LAST_NORM))
5111 		goto out;
5112 
5113 	/* don't fail immediately if it's r/o, at least try to report other errors */
5114 	error = mnt_want_write(path->mnt);
5115 	/*
5116 	 * Do the final lookup.  Suppress 'create' if there is a trailing
5117 	 * '/', and a directory wasn't requested.
5118 	 */
5119 	if (last.name[last.len] && !want_dir)
5120 		create_flags &= ~LOOKUP_CREATE;
5121 	dentry = start_dirop(path->dentry, &last, reval_flag | create_flags);
5122 	if (IS_ERR(dentry))
5123 		goto out_drop_write;
5124 
5125 	if (unlikely(error))
5126 		goto fail;
5127 
5128 	return dentry;
5129 fail:
5130 	end_dirop(dentry);
5131 	dentry = ERR_PTR(error);
5132 out_drop_write:
5133 	if (!error)
5134 		mnt_drop_write(path->mnt);
5135 out:
5136 	path_put(path);
5137 	return dentry;
5138 }
5139 
5140 struct dentry *start_creating_path(int dfd, const char *pathname,
5141 				   struct path *path, unsigned int lookup_flags)
5142 {
5143 	CLASS(filename_kernel, filename)(pathname);
5144 	return filename_create(dfd, filename, path, lookup_flags);
5145 }
5146 EXPORT_SYMBOL(start_creating_path);
5147 
5148 /**
5149  * end_creating_path - finish a code section started by start_creating_path()
5150  * @path: the path instantiated by start_creating_path()
5151  * @dentry: the dentry returned by start_creating_path()
5152  *
5153  * end_creating_path() will unlock and locks taken by start_creating_path()
5154  * and drop an references that were taken.  It should only be called
5155  * if start_creating_path() returned a non-error.
5156  * If vfs_mkdir() was called and it returned an error, that error *should*
5157  * be passed to end_creating_path() together with the path.
5158  */
5159 void end_creating_path(const struct path *path, struct dentry *dentry)
5160 {
5161 	end_creating(dentry);
5162 	mnt_drop_write(path->mnt);
5163 	path_put(path);
5164 }
5165 EXPORT_SYMBOL(end_creating_path);
5166 
5167 inline struct dentry *start_creating_user_path(
5168 	int dfd, const char __user *pathname,
5169 	struct path *path, unsigned int lookup_flags)
5170 {
5171 	CLASS(filename, filename)(pathname);
5172 	return filename_create(dfd, filename, path, lookup_flags);
5173 }
5174 EXPORT_SYMBOL(start_creating_user_path);
5175 
5176 /**
5177  * dentry_create - Create and open a file
5178  * @path: path to create
5179  * @flags: O\_ flags
5180  * @mode: mode bits for new file
5181  * @cred: credentials to use
5182  *
5183  * Caller must hold the parent directory's lock, and have prepared
5184  * a negative dentry, placed in @path->dentry, for the new file.
5185  *
5186  * Caller sets @path->mnt to the vfsmount of the filesystem where
5187  * the new file is to be created. The parent directory and the
5188  * negative dentry must reside on the same filesystem instance.
5189  *
5190  * On success, returns a ``struct file *``. Otherwise an ERR_PTR
5191  * is returned.
5192  */
5193 struct file *dentry_create(struct path *path, int flags, umode_t mode,
5194 			   const struct cred *cred)
5195 {
5196 	struct file *file __free(fput) = NULL;
5197 	struct dentry *dentry = path->dentry;
5198 	struct dentry *orig_dentry = dentry;
5199 	struct dentry *dir = dentry->d_parent;
5200 	struct inode *dir_inode = d_inode(dir);
5201 	struct mnt_idmap *idmap;
5202 	int error, create_error;
5203 
5204 	file = alloc_empty_file(flags, cred);
5205 	if (IS_ERR(file))
5206 		return file;
5207 
5208 	idmap = mnt_idmap(path->mnt);
5209 
5210 	if (dir_inode->i_op->atomic_open) {
5211 		path->dentry = dir;
5212 		mode = vfs_prepare_mode(idmap, dir_inode, mode, S_IALLUGO, S_IFREG);
5213 
5214 		create_error = may_o_create(idmap, path, dentry, mode);
5215 		if (create_error)
5216 			flags &= ~O_CREAT;
5217 
5218 		/* atomic_open will dput(dentry) on error */
5219 		dget(orig_dentry);
5220 		dentry = atomic_open(path, dentry, file, flags, mode, create_error);
5221 		error = PTR_ERR_OR_ZERO(dentry);
5222 
5223 		if (IS_ERR(dentry))
5224 			/* keep the original */
5225 			dentry = orig_dentry;
5226 		else
5227 			/* Drop the extra reference */
5228 			dput(orig_dentry);
5229 
5230 		if (!error) {
5231 			if (file->f_mode & FMODE_CREATED)
5232 				fsnotify_create(dir->d_inode, dentry);
5233 			if (file->f_mode & FMODE_OPENED)
5234 				fsnotify_open(file);
5235 		}
5236 
5237 		path->dentry = dentry;
5238 
5239 	} else {
5240 		error = vfs_create(mnt_idmap(path->mnt), path->dentry, mode, NULL);
5241 		if (!error)
5242 			error = vfs_open(path, file);
5243 	}
5244 	if (unlikely(error))
5245 		return ERR_PTR(error);
5246 
5247 	return no_free_ptr(file);
5248 }
5249 EXPORT_SYMBOL(dentry_create);
5250 
5251 /**
5252  * vfs_mknod - create device node or file
5253  * @idmap:		idmap of the mount the inode was found from
5254  * @dir:		inode of the parent directory
5255  * @dentry:		dentry of the child device node
5256  * @mode:		mode of the child device node
5257  * @dev:		device number of device to create
5258  * @delegated_inode:	returns parent inode, if the inode is delegated.
5259  *
5260  * Create a device node or file.
5261  *
5262  * If the inode has been found through an idmapped mount the idmap of
5263  * the vfsmount must be passed through @idmap. This function will then take
5264  * care to map the inode according to @idmap before checking permissions.
5265  * On non-idmapped mounts or if permission checking is to be performed on the
5266  * raw inode simply pass @nop_mnt_idmap.
5267  */
5268 int vfs_mknod(struct mnt_idmap *idmap, struct inode *dir,
5269 	      struct dentry *dentry, umode_t mode, dev_t dev,
5270 	      struct delegated_inode *delegated_inode)
5271 {
5272 	bool is_whiteout = S_ISCHR(mode) && dev == WHITEOUT_DEV;
5273 	int error = may_create_dentry(idmap, dir, dentry);
5274 
5275 	if (error)
5276 		return error;
5277 
5278 	if ((S_ISCHR(mode) || S_ISBLK(mode)) && !is_whiteout &&
5279 	    !capable(CAP_MKNOD))
5280 		return -EPERM;
5281 
5282 	if (!dir->i_op->mknod)
5283 		return -EPERM;
5284 
5285 	mode = vfs_prepare_mode(idmap, dir, mode, mode, mode);
5286 	error = devcgroup_inode_mknod(mode, dev);
5287 	if (error)
5288 		return error;
5289 
5290 	error = security_inode_mknod(dir, dentry, mode, dev);
5291 	if (error)
5292 		return error;
5293 
5294 	error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, delegated_inode);
5295 	if (error)
5296 		return error;
5297 
5298 	error = dir->i_op->mknod(idmap, dir, dentry, mode, dev);
5299 	if (!error)
5300 		fsnotify_create(dir, dentry);
5301 	return error;
5302 }
5303 EXPORT_SYMBOL(vfs_mknod);
5304 
5305 static int may_mknod(umode_t mode)
5306 {
5307 	switch (mode & S_IFMT) {
5308 	case S_IFREG:
5309 	case S_IFCHR:
5310 	case S_IFBLK:
5311 	case S_IFIFO:
5312 	case S_IFSOCK:
5313 	case 0: /* zero mode translates to S_IFREG */
5314 		return 0;
5315 	case S_IFDIR:
5316 		return -EPERM;
5317 	default:
5318 		return -EINVAL;
5319 	}
5320 }
5321 
5322 int filename_mknodat(int dfd, struct filename *name, umode_t mode,
5323 		     unsigned int dev)
5324 {
5325 	struct delegated_inode di = { };
5326 	struct mnt_idmap *idmap;
5327 	struct dentry *dentry;
5328 	struct path path;
5329 	int error;
5330 	unsigned int lookup_flags = 0;
5331 
5332 	error = may_mknod(mode);
5333 	if (error)
5334 		return error;
5335 retry:
5336 	dentry = filename_create(dfd, name, &path, lookup_flags);
5337 	if (IS_ERR(dentry))
5338 		return PTR_ERR(dentry);
5339 
5340 	error = security_path_mknod(&path, dentry,
5341 			mode_strip_umask(path.dentry->d_inode, mode), dev);
5342 	if (error)
5343 		goto out2;
5344 
5345 	idmap = mnt_idmap(path.mnt);
5346 	switch (mode & S_IFMT) {
5347 		case 0: case S_IFREG:
5348 			error = vfs_create(idmap, dentry, mode, &di);
5349 			if (!error)
5350 				security_path_post_mknod(idmap, dentry);
5351 			break;
5352 		case S_IFCHR: case S_IFBLK:
5353 			error = vfs_mknod(idmap, path.dentry->d_inode,
5354 					  dentry, mode, new_decode_dev(dev), &di);
5355 			break;
5356 		case S_IFIFO: case S_IFSOCK:
5357 			error = vfs_mknod(idmap, path.dentry->d_inode,
5358 					  dentry, mode, 0, &di);
5359 			break;
5360 	}
5361 out2:
5362 	end_creating_path(&path, dentry);
5363 	if (is_delegated(&di)) {
5364 		error = break_deleg_wait(&di);
5365 		if (!error)
5366 			goto retry;
5367 	}
5368 	if (retry_estale(error, lookup_flags)) {
5369 		lookup_flags |= LOOKUP_REVAL;
5370 		goto retry;
5371 	}
5372 	return error;
5373 }
5374 
5375 SYSCALL_DEFINE4(mknodat, int, dfd, const char __user *, filename, umode_t, mode,
5376 		unsigned int, dev)
5377 {
5378 	CLASS(filename, name)(filename);
5379 	return filename_mknodat(dfd, name, mode, dev);
5380 }
5381 
5382 SYSCALL_DEFINE3(mknod, const char __user *, filename, umode_t, mode, unsigned, dev)
5383 {
5384 	CLASS(filename, name)(filename);
5385 	return filename_mknodat(AT_FDCWD, name, mode, dev);
5386 }
5387 
5388 /**
5389  * vfs_mkdir - create directory returning correct dentry if possible
5390  * @idmap:		idmap of the mount the inode was found from
5391  * @dir:		inode of the parent directory
5392  * @dentry:		dentry of the child directory
5393  * @mode:		mode of the child directory
5394  * @delegated_inode:	returns parent inode, if the inode is delegated.
5395  *
5396  * Create a directory.
5397  *
5398  * If the inode has been found through an idmapped mount the idmap of
5399  * the vfsmount must be passed through @idmap. This function will then take
5400  * care to map the inode according to @idmap before checking permissions.
5401  * On non-idmapped mounts or if permission checking is to be performed on the
5402  * raw inode simply pass @nop_mnt_idmap.
5403  *
5404  * In the event that the filesystem does not use the *@dentry but leaves it
5405  * negative or unhashes it and possibly splices a different one returning it,
5406  * the original dentry is dput() and the alternate is returned.
5407  *
5408  * In case of an error the dentry is dput() and an ERR_PTR() is returned.
5409  */
5410 struct dentry *vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
5411 			 struct dentry *dentry, umode_t mode,
5412 			 struct delegated_inode *delegated_inode)
5413 {
5414 	int error;
5415 	unsigned max_links = dir->i_sb->s_max_links;
5416 	struct dentry *de;
5417 
5418 	error = may_create_dentry(idmap, dir, dentry);
5419 	if (error)
5420 		goto err;
5421 
5422 	error = -EPERM;
5423 	if (!dir->i_op->mkdir)
5424 		goto err;
5425 
5426 	mode = vfs_prepare_mode(idmap, dir, mode, S_IRWXUGO | S_ISVTX, S_IFDIR);
5427 	error = security_inode_mkdir(dir, dentry, mode);
5428 	if (error)
5429 		goto err;
5430 
5431 	error = -EMLINK;
5432 	if (max_links && dir->i_nlink >= max_links)
5433 		goto err;
5434 
5435 	error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, delegated_inode);
5436 	if (error)
5437 		goto err;
5438 
5439 	de = dir->i_op->mkdir(idmap, dir, dentry, mode);
5440 	error = PTR_ERR(de);
5441 	if (IS_ERR(de))
5442 		goto err;
5443 	if (de) {
5444 		dput(dentry);
5445 		dentry = de;
5446 	}
5447 	fsnotify_mkdir(dir, dentry);
5448 	return dentry;
5449 
5450 err:
5451 	end_creating(dentry);
5452 	return ERR_PTR(error);
5453 }
5454 EXPORT_SYMBOL(vfs_mkdir);
5455 
5456 int filename_mkdirat(int dfd, struct filename *name, umode_t mode)
5457 {
5458 	struct dentry *dentry;
5459 	struct path path;
5460 	int error;
5461 	unsigned int lookup_flags = LOOKUP_DIRECTORY;
5462 	struct delegated_inode delegated_inode = { };
5463 
5464 retry:
5465 	dentry = filename_create(dfd, name, &path, lookup_flags);
5466 	if (IS_ERR(dentry))
5467 		return PTR_ERR(dentry);
5468 
5469 	error = security_path_mkdir(&path, dentry,
5470 			mode_strip_umask(path.dentry->d_inode, mode));
5471 	if (!error) {
5472 		dentry = vfs_mkdir(mnt_idmap(path.mnt), path.dentry->d_inode,
5473 				   dentry, mode, &delegated_inode);
5474 		if (IS_ERR(dentry))
5475 			error = PTR_ERR(dentry);
5476 	}
5477 	end_creating_path(&path, dentry);
5478 	if (is_delegated(&delegated_inode)) {
5479 		error = break_deleg_wait(&delegated_inode);
5480 		if (!error)
5481 			goto retry;
5482 	}
5483 	if (retry_estale(error, lookup_flags)) {
5484 		lookup_flags |= LOOKUP_REVAL;
5485 		goto retry;
5486 	}
5487 	return error;
5488 }
5489 
5490 SYSCALL_DEFINE3(mkdirat, int, dfd, const char __user *, pathname, umode_t, mode)
5491 {
5492 	CLASS(filename, name)(pathname);
5493 	return filename_mkdirat(dfd, name, mode);
5494 }
5495 
5496 SYSCALL_DEFINE2(mkdir, const char __user *, pathname, umode_t, mode)
5497 {
5498 	CLASS(filename, name)(pathname);
5499 	return filename_mkdirat(AT_FDCWD, name, mode);
5500 }
5501 
5502 /**
5503  * vfs_rmdir - remove directory
5504  * @idmap:		idmap of the mount the inode was found from
5505  * @dir:		inode of the parent directory
5506  * @dentry:		dentry of the child directory
5507  * @delegated_inode:	returns parent inode, if it's delegated.
5508  *
5509  * Remove a directory.
5510  *
5511  * If the inode has been found through an idmapped mount the idmap of
5512  * the vfsmount must be passed through @idmap. This function will then take
5513  * care to map the inode according to @idmap before checking permissions.
5514  * On non-idmapped mounts or if permission checking is to be performed on the
5515  * raw inode simply pass @nop_mnt_idmap.
5516  */
5517 int vfs_rmdir(struct mnt_idmap *idmap, struct inode *dir,
5518 	      struct dentry *dentry, struct delegated_inode *delegated_inode)
5519 {
5520 	int error = may_delete_dentry(idmap, dir, dentry, true);
5521 
5522 	if (error)
5523 		return error;
5524 
5525 	if (!dir->i_op->rmdir)
5526 		return -EPERM;
5527 
5528 	dget(dentry);
5529 	inode_lock(dentry->d_inode);
5530 
5531 	error = -EBUSY;
5532 	if (is_local_mountpoint(dentry) ||
5533 	    (dentry->d_inode->i_flags & S_KERNEL_FILE))
5534 		goto out;
5535 
5536 	error = security_inode_rmdir(dir, dentry);
5537 	if (error)
5538 		goto out;
5539 
5540 	error = try_break_deleg(dir, LEASE_BREAK_DIR_DELETE, delegated_inode);
5541 	if (error)
5542 		goto out;
5543 
5544 	error = dir->i_op->rmdir(dir, dentry);
5545 	if (error)
5546 		goto out;
5547 
5548 	shrink_dcache_parent(dentry);
5549 	dentry->d_inode->i_flags |= S_DEAD;
5550 	dont_mount(dentry);
5551 	detach_mounts(dentry);
5552 
5553 out:
5554 	inode_unlock(dentry->d_inode);
5555 	dput(dentry);
5556 	if (!error)
5557 		d_delete_notify(dir, dentry);
5558 	return error;
5559 }
5560 EXPORT_SYMBOL(vfs_rmdir);
5561 
5562 int filename_rmdir(int dfd, struct filename *name)
5563 {
5564 	int error;
5565 	struct dentry *dentry;
5566 	struct path path;
5567 	struct qstr last;
5568 	enum last_type type;
5569 	unsigned int lookup_flags = 0;
5570 	struct delegated_inode delegated_inode = { };
5571 retry:
5572 	error = filename_parentat(dfd, name, lookup_flags, &path, &last, &type);
5573 	if (error)
5574 		return error;
5575 
5576 	switch (type) {
5577 	case LAST_NORM:
5578 		break;
5579 	case LAST_DOTDOT:
5580 		error = -ENOTEMPTY;
5581 		goto exit2;
5582 	case LAST_DOT:
5583 		error = -EINVAL;
5584 		goto exit2;
5585 	case LAST_ROOT:
5586 		error = -EBUSY;
5587 		goto exit2;
5588 	}
5589 
5590 	error = mnt_want_write(path.mnt);
5591 	if (error)
5592 		goto exit2;
5593 
5594 	dentry = start_dirop(path.dentry, &last, lookup_flags);
5595 	error = PTR_ERR(dentry);
5596 	if (IS_ERR(dentry))
5597 		goto exit3;
5598 	error = security_path_rmdir(&path, dentry);
5599 	if (error)
5600 		goto exit4;
5601 	error = vfs_rmdir(mnt_idmap(path.mnt), path.dentry->d_inode,
5602 			  dentry, &delegated_inode);
5603 exit4:
5604 	end_dirop(dentry);
5605 exit3:
5606 	mnt_drop_write(path.mnt);
5607 exit2:
5608 	path_put(&path);
5609 	if (is_delegated(&delegated_inode)) {
5610 		error = break_deleg_wait(&delegated_inode);
5611 		if (!error)
5612 			goto retry;
5613 	}
5614 	if (retry_estale(error, lookup_flags)) {
5615 		lookup_flags |= LOOKUP_REVAL;
5616 		goto retry;
5617 	}
5618 	return error;
5619 }
5620 
5621 SYSCALL_DEFINE1(rmdir, const char __user *, pathname)
5622 {
5623 	CLASS(filename, name)(pathname);
5624 	return filename_rmdir(AT_FDCWD, name);
5625 }
5626 
5627 /**
5628  * vfs_unlink - unlink a filesystem object
5629  * @idmap:	idmap of the mount the inode was found from
5630  * @dir:	parent directory
5631  * @dentry:	victim
5632  * @delegated_inode: returns victim inode, if the inode is delegated.
5633  *
5634  * The caller must hold dir->i_rwsem exclusively.
5635  *
5636  * If vfs_unlink discovers a delegation, it will return -EWOULDBLOCK and
5637  * return a reference to the inode in delegated_inode.  The caller
5638  * should then break the delegation on that inode and retry.  Because
5639  * breaking a delegation may take a long time, the caller should drop
5640  * dir->i_rwsem before doing so.
5641  *
5642  * Alternatively, a caller may pass NULL for delegated_inode.  This may
5643  * be appropriate for callers that expect the underlying filesystem not
5644  * to be NFS exported.
5645  *
5646  * If the inode has been found through an idmapped mount the idmap of
5647  * the vfsmount must be passed through @idmap. This function will then take
5648  * care to map the inode according to @idmap before checking permissions.
5649  * On non-idmapped mounts or if permission checking is to be performed on the
5650  * raw inode simply pass @nop_mnt_idmap.
5651  */
5652 int vfs_unlink(struct mnt_idmap *idmap, struct inode *dir,
5653 	       struct dentry *dentry, struct delegated_inode *delegated_inode)
5654 {
5655 	struct inode *target = dentry->d_inode;
5656 	int error = may_delete_dentry(idmap, dir, dentry, false);
5657 
5658 	if (error)
5659 		return error;
5660 
5661 	if (!dir->i_op->unlink)
5662 		return -EPERM;
5663 
5664 	inode_lock(target);
5665 	if (IS_SWAPFILE(target))
5666 		error = -EPERM;
5667 	else if (is_local_mountpoint(dentry))
5668 		error = -EBUSY;
5669 	else {
5670 		error = security_inode_unlink(dir, dentry);
5671 		if (!error) {
5672 			error = try_break_deleg(dir, LEASE_BREAK_DIR_DELETE, delegated_inode);
5673 			if (error)
5674 				goto out;
5675 			error = try_break_deleg(target, 0, delegated_inode);
5676 			if (error)
5677 				goto out;
5678 			error = dir->i_op->unlink(dir, dentry);
5679 			if (!error) {
5680 				dont_mount(dentry);
5681 				detach_mounts(dentry);
5682 			}
5683 		}
5684 	}
5685 out:
5686 	inode_unlock(target);
5687 
5688 	/* We don't d_delete() NFS sillyrenamed files--they still exist. */
5689 	if (!error && dentry->d_flags & DCACHE_NFSFS_RENAMED) {
5690 		fsnotify_unlink(dir, dentry);
5691 	} else if (!error) {
5692 		fsnotify_link_count(target);
5693 		d_delete_notify(dir, dentry);
5694 	}
5695 
5696 	return error;
5697 }
5698 EXPORT_SYMBOL(vfs_unlink);
5699 
5700 /*
5701  * Make sure that the actual truncation of the file will occur outside its
5702  * directory's i_rwsem.  Truncate can take a long time if there is a lot of
5703  * writeout happening, and we don't want to prevent access to the directory
5704  * while waiting on the I/O.
5705  */
5706 int filename_unlinkat(int dfd, struct filename *name)
5707 {
5708 	int error;
5709 	struct dentry *dentry;
5710 	struct path path;
5711 	struct qstr last;
5712 	enum last_type type;
5713 	struct inode *inode;
5714 	struct delegated_inode delegated_inode = { };
5715 	unsigned int lookup_flags = 0;
5716 retry:
5717 	error = filename_parentat(dfd, name, lookup_flags, &path, &last, &type);
5718 	if (error)
5719 		return error;
5720 
5721 	error = -EISDIR;
5722 	if (type != LAST_NORM)
5723 		goto exit_path_put;
5724 
5725 	error = mnt_want_write(path.mnt);
5726 	if (error)
5727 		goto exit_path_put;
5728 retry_deleg:
5729 	dentry = start_dirop(path.dentry, &last, lookup_flags);
5730 	error = PTR_ERR(dentry);
5731 	if (IS_ERR(dentry))
5732 		goto exit_drop_write;
5733 
5734 	/* Why not before? Because we want correct error value */
5735 	if (unlikely(last.name[last.len])) {
5736 		if (d_is_dir(dentry))
5737 			error = -EISDIR;
5738 		else
5739 			error = -ENOTDIR;
5740 		end_dirop(dentry);
5741 		goto exit_drop_write;
5742 	}
5743 	inode = dentry->d_inode;
5744 	ihold(inode);
5745 	error = security_path_unlink(&path, dentry);
5746 	if (error)
5747 		goto exit_end_dirop;
5748 	error = vfs_unlink(mnt_idmap(path.mnt), path.dentry->d_inode,
5749 			   dentry, &delegated_inode);
5750 exit_end_dirop:
5751 	end_dirop(dentry);
5752 	iput(inode);	/* truncate the inode here */
5753 	if (is_delegated(&delegated_inode)) {
5754 		error = break_deleg_wait(&delegated_inode);
5755 		if (!error)
5756 			goto retry_deleg;
5757 	}
5758 exit_drop_write:
5759 	mnt_drop_write(path.mnt);
5760 exit_path_put:
5761 	path_put(&path);
5762 	if (retry_estale(error, lookup_flags)) {
5763 		lookup_flags |= LOOKUP_REVAL;
5764 		goto retry;
5765 	}
5766 	return error;
5767 }
5768 
5769 SYSCALL_DEFINE3(unlinkat, int, dfd, const char __user *, pathname, int, flag)
5770 {
5771 	if ((flag & ~AT_REMOVEDIR) != 0)
5772 		return -EINVAL;
5773 
5774 	CLASS(filename, name)(pathname);
5775 	if (flag & AT_REMOVEDIR)
5776 		return filename_rmdir(dfd, name);
5777 	return filename_unlinkat(dfd, name);
5778 }
5779 
5780 SYSCALL_DEFINE1(unlink, const char __user *, pathname)
5781 {
5782 	CLASS(filename, name)(pathname);
5783 	return filename_unlinkat(AT_FDCWD, name);
5784 }
5785 
5786 /**
5787  * vfs_symlink - create symlink
5788  * @idmap:	idmap of the mount the inode was found from
5789  * @dir:	inode of the parent directory
5790  * @dentry:	dentry of the child symlink file
5791  * @oldname:	name of the file to link to
5792  * @delegated_inode: returns victim inode, if the inode is delegated.
5793  *
5794  * Create a symlink.
5795  *
5796  * If the inode has been found through an idmapped mount the idmap of
5797  * the vfsmount must be passed through @idmap. This function will then take
5798  * care to map the inode according to @idmap before checking permissions.
5799  * On non-idmapped mounts or if permission checking is to be performed on the
5800  * raw inode simply pass @nop_mnt_idmap.
5801  */
5802 int vfs_symlink(struct mnt_idmap *idmap, struct inode *dir,
5803 		struct dentry *dentry, const char *oldname,
5804 		struct delegated_inode *delegated_inode)
5805 {
5806 	int error;
5807 
5808 	error = may_create_dentry(idmap, dir, dentry);
5809 	if (error)
5810 		return error;
5811 
5812 	if (!dir->i_op->symlink)
5813 		return -EPERM;
5814 
5815 	error = security_inode_symlink(dir, dentry, oldname);
5816 	if (error)
5817 		return error;
5818 
5819 	error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, delegated_inode);
5820 	if (error)
5821 		return error;
5822 
5823 	error = dir->i_op->symlink(idmap, dir, dentry, oldname);
5824 	if (!error)
5825 		fsnotify_create(dir, dentry);
5826 	return error;
5827 }
5828 EXPORT_SYMBOL(vfs_symlink);
5829 
5830 int filename_symlinkat(struct filename *from, int newdfd, struct filename *to)
5831 {
5832 	int error;
5833 	struct dentry *dentry;
5834 	struct path path;
5835 	unsigned int lookup_flags = 0;
5836 	struct delegated_inode delegated_inode = { };
5837 
5838 	if (IS_ERR(from))
5839 		return PTR_ERR(from);
5840 
5841 retry:
5842 	dentry = filename_create(newdfd, to, &path, lookup_flags);
5843 	if (IS_ERR(dentry))
5844 		return PTR_ERR(dentry);
5845 
5846 	error = security_path_symlink(&path, dentry, from->name);
5847 	if (!error)
5848 		error = vfs_symlink(mnt_idmap(path.mnt), path.dentry->d_inode,
5849 				    dentry, from->name, &delegated_inode);
5850 	end_creating_path(&path, dentry);
5851 	if (is_delegated(&delegated_inode)) {
5852 		error = break_deleg_wait(&delegated_inode);
5853 		if (!error)
5854 			goto retry;
5855 	}
5856 	if (retry_estale(error, lookup_flags)) {
5857 		lookup_flags |= LOOKUP_REVAL;
5858 		goto retry;
5859 	}
5860 	return error;
5861 }
5862 
5863 SYSCALL_DEFINE3(symlinkat, const char __user *, oldname,
5864 		int, newdfd, const char __user *, newname)
5865 {
5866 	CLASS(filename, old)(oldname);
5867 	CLASS(filename, new)(newname);
5868 	return filename_symlinkat(old, newdfd, new);
5869 }
5870 
5871 SYSCALL_DEFINE2(symlink, const char __user *, oldname, const char __user *, newname)
5872 {
5873 	CLASS(filename, old)(oldname);
5874 	CLASS(filename, new)(newname);
5875 	return filename_symlinkat(old, AT_FDCWD, new);
5876 }
5877 
5878 /**
5879  * vfs_link - create a new link
5880  * @old_dentry:	object to be linked
5881  * @idmap:	idmap of the mount
5882  * @dir:	new parent
5883  * @new_dentry:	where to create the new link
5884  * @delegated_inode: returns inode needing a delegation break
5885  *
5886  * The caller must hold dir->i_rwsem exclusively.
5887  *
5888  * If vfs_link discovers a delegation on the to-be-linked file in need
5889  * of breaking, it will return -EWOULDBLOCK and return a reference to the
5890  * inode in delegated_inode.  The caller should then break the delegation
5891  * and retry.  Because breaking a delegation may take a long time, the
5892  * caller should drop the i_rwsem before doing so.
5893  *
5894  * Alternatively, a caller may pass NULL for delegated_inode.  This may
5895  * be appropriate for callers that expect the underlying filesystem not
5896  * to be NFS exported.
5897  *
5898  * If the inode has been found through an idmapped mount the idmap of
5899  * the vfsmount must be passed through @idmap. This function will then take
5900  * care to map the inode according to @idmap before checking permissions.
5901  * On non-idmapped mounts or if permission checking is to be performed on the
5902  * raw inode simply pass @nop_mnt_idmap.
5903  */
5904 int vfs_link(struct dentry *old_dentry, struct mnt_idmap *idmap,
5905 	     struct inode *dir, struct dentry *new_dentry,
5906 	     struct delegated_inode *delegated_inode)
5907 {
5908 	struct inode *inode = old_dentry->d_inode;
5909 	unsigned max_links = dir->i_sb->s_max_links;
5910 	int error;
5911 
5912 	if (!inode)
5913 		return -ENOENT;
5914 
5915 	error = may_create_dentry(idmap, dir, new_dentry);
5916 	if (error)
5917 		return error;
5918 
5919 	if (dir->i_sb != inode->i_sb)
5920 		return -EXDEV;
5921 
5922 	/*
5923 	 * A link to an append-only or immutable file cannot be created.
5924 	 */
5925 	if (IS_APPEND(inode) || IS_IMMUTABLE(inode))
5926 		return -EPERM;
5927 	/*
5928 	 * Updating the link count will likely cause i_uid and i_gid to
5929 	 * be written back improperly if their true value is unknown to
5930 	 * the vfs.
5931 	 */
5932 	if (HAS_UNMAPPED_ID(idmap, inode))
5933 		return -EPERM;
5934 	if (!dir->i_op->link)
5935 		return -EPERM;
5936 	if (S_ISDIR(inode->i_mode))
5937 		return -EPERM;
5938 
5939 	error = security_inode_link(old_dentry, dir, new_dentry);
5940 	if (error)
5941 		return error;
5942 
5943 	inode_lock(inode);
5944 	/* Make sure we don't allow creating hardlink to an unlinked file */
5945 	if (inode->i_nlink == 0 && !(inode_state_read_once(inode) & I_LINKABLE))
5946 		error =  -ENOENT;
5947 	else if (max_links && inode->i_nlink >= max_links)
5948 		error = -EMLINK;
5949 	else {
5950 		error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, delegated_inode);
5951 		if (!error)
5952 			error = try_break_deleg(inode, 0, delegated_inode);
5953 		if (!error)
5954 			error = dir->i_op->link(old_dentry, dir, new_dentry);
5955 	}
5956 
5957 	if (!error && (inode_state_read_once(inode) & I_LINKABLE)) {
5958 		spin_lock(&inode->i_lock);
5959 		inode_state_clear(inode, I_LINKABLE);
5960 		spin_unlock(&inode->i_lock);
5961 	}
5962 	inode_unlock(inode);
5963 	if (!error)
5964 		fsnotify_link(dir, inode, new_dentry);
5965 	return error;
5966 }
5967 EXPORT_SYMBOL(vfs_link);
5968 
5969 /*
5970  * Hardlinks are often used in delicate situations.  We avoid
5971  * security-related surprises by not following symlinks on the
5972  * newname.  --KAB
5973  *
5974  * We don't follow them on the oldname either to be compatible
5975  * with linux 2.0, and to avoid hard-linking to directories
5976  * and other special files.  --ADM
5977 */
5978 int filename_linkat(int olddfd, struct filename *old,
5979 		    int newdfd, struct filename *new, int flags)
5980 {
5981 	struct mnt_idmap *idmap;
5982 	struct dentry *new_dentry;
5983 	struct path old_path, new_path;
5984 	struct delegated_inode delegated_inode = { };
5985 	int how = 0;
5986 	int error;
5987 
5988 	if ((flags & ~(AT_SYMLINK_FOLLOW | AT_EMPTY_PATH)) != 0)
5989 		return -EINVAL;
5990 	/*
5991 	 * To use null names we require CAP_DAC_READ_SEARCH or
5992 	 * that the open-time creds of the dfd matches current.
5993 	 * This ensures that not everyone will be able to create
5994 	 * a hardlink using the passed file descriptor.
5995 	 */
5996 	if (flags & AT_EMPTY_PATH)
5997 		how |= LOOKUP_LINKAT_EMPTY;
5998 
5999 	if (flags & AT_SYMLINK_FOLLOW)
6000 		how |= LOOKUP_FOLLOW;
6001 retry:
6002 	error = filename_lookup(olddfd, old, how, &old_path, NULL);
6003 	if (error)
6004 		return error;
6005 
6006 	new_dentry = filename_create(newdfd, new, &new_path,
6007 					(how & LOOKUP_REVAL));
6008 	error = PTR_ERR(new_dentry);
6009 	if (IS_ERR(new_dentry))
6010 		goto out_putpath;
6011 
6012 	error = -EXDEV;
6013 	if (old_path.mnt != new_path.mnt)
6014 		goto out_dput;
6015 	idmap = mnt_idmap(new_path.mnt);
6016 	error = may_linkat(idmap, &old_path);
6017 	if (unlikely(error))
6018 		goto out_dput;
6019 	error = security_path_link(old_path.dentry, &new_path, new_dentry);
6020 	if (error)
6021 		goto out_dput;
6022 	error = vfs_link(old_path.dentry, idmap, new_path.dentry->d_inode,
6023 			 new_dentry, &delegated_inode);
6024 out_dput:
6025 	end_creating_path(&new_path, new_dentry);
6026 	if (is_delegated(&delegated_inode)) {
6027 		error = break_deleg_wait(&delegated_inode);
6028 		if (!error) {
6029 			path_put(&old_path);
6030 			goto retry;
6031 		}
6032 	}
6033 	if (retry_estale(error, how)) {
6034 		path_put(&old_path);
6035 		how |= LOOKUP_REVAL;
6036 		goto retry;
6037 	}
6038 out_putpath:
6039 	path_put(&old_path);
6040 	return error;
6041 }
6042 
6043 SYSCALL_DEFINE5(linkat, int, olddfd, const char __user *, oldname,
6044 		int, newdfd, const char __user *, newname, int, flags)
6045 {
6046 	CLASS(filename_uflags, old)(oldname, flags);
6047 	CLASS(filename, new)(newname);
6048 	return filename_linkat(olddfd, old, newdfd, new, flags);
6049 }
6050 
6051 SYSCALL_DEFINE2(link, const char __user *, oldname, const char __user *, newname)
6052 {
6053 	CLASS(filename, old)(oldname);
6054 	CLASS(filename, new)(newname);
6055 	return filename_linkat(AT_FDCWD, old, AT_FDCWD, new, 0);
6056 }
6057 
6058 /**
6059  * vfs_rename - rename a filesystem object
6060  * @rd:		pointer to &struct renamedata info
6061  *
6062  * The caller must hold multiple mutexes--see lock_rename()).
6063  *
6064  * If vfs_rename discovers a delegation in need of breaking at either
6065  * the source or destination, it will return -EWOULDBLOCK and return a
6066  * reference to the inode in delegated_inode.  The caller should then
6067  * break the delegation and retry.  Because breaking a delegation may
6068  * take a long time, the caller should drop all locks before doing
6069  * so.
6070  *
6071  * Alternatively, a caller may pass NULL for delegated_inode.  This may
6072  * be appropriate for callers that expect the underlying filesystem not
6073  * to be NFS exported.
6074  *
6075  * The worst of all namespace operations - renaming directory. "Perverted"
6076  * doesn't even start to describe it. Somebody in UCB had a heck of a trip...
6077  * Problems:
6078  *
6079  *	a) we can get into loop creation.
6080  *	b) race potential - two innocent renames can create a loop together.
6081  *	   That's where 4.4BSD screws up. Current fix: serialization on
6082  *	   sb->s_vfs_rename_mutex. We might be more accurate, but that's another
6083  *	   story.
6084  *	c) we may have to lock up to _four_ objects - parents and victim (if it exists),
6085  *	   and source (if it's a non-directory or a subdirectory that moves to
6086  *	   different parent).
6087  *	   And that - after we got ->i_rwsem on parents (until then we don't know
6088  *	   whether the target exists).  Solution: try to be smart with locking
6089  *	   order for inodes.  We rely on the fact that tree topology may change
6090  *	   only under ->s_vfs_rename_mutex _and_ that parent of the object we
6091  *	   move will be locked.  Thus we can rank directories by the tree
6092  *	   (ancestors first) and rank all non-directories after them.
6093  *	   That works since everybody except rename does "lock parent, lookup,
6094  *	   lock child" and rename is under ->s_vfs_rename_mutex.
6095  *	   HOWEVER, it relies on the assumption that any object with ->lookup()
6096  *	   has no more than 1 dentry.  If "hybrid" objects will ever appear,
6097  *	   we'd better make sure that there's no link(2) for them.
6098  *	d) conversion from fhandle to dentry may come in the wrong moment - when
6099  *	   we are removing the target. Solution: we will have to grab ->i_rwsem
6100  *	   in the fhandle_to_dentry code. [FIXME - current nfsfh.c relies on
6101  *	   ->i_rwsem on parents, which works but leads to some truly excessive
6102  *	   locking].
6103  */
6104 int vfs_rename(struct renamedata *rd)
6105 {
6106 	int error;
6107 	struct inode *old_dir = d_inode(rd->old_parent);
6108 	struct inode *new_dir = d_inode(rd->new_parent);
6109 	struct dentry *old_dentry = rd->old_dentry;
6110 	struct dentry *new_dentry = rd->new_dentry;
6111 	struct delegated_inode *delegated_inode = rd->delegated_inode;
6112 	unsigned int flags = rd->flags;
6113 	bool is_dir = d_is_dir(old_dentry);
6114 	struct inode *source = old_dentry->d_inode;
6115 	struct inode *target = new_dentry->d_inode;
6116 	bool new_is_dir = false;
6117 	unsigned max_links = new_dir->i_sb->s_max_links;
6118 	struct name_snapshot old_name;
6119 	bool lock_old_subdir, lock_new_subdir;
6120 
6121 	if (source == target)
6122 		return 0;
6123 
6124 	error = may_delete_dentry(rd->mnt_idmap, old_dir, old_dentry, is_dir);
6125 	if (error)
6126 		return error;
6127 
6128 	if (!target) {
6129 		error = may_create_dentry(rd->mnt_idmap, new_dir, new_dentry);
6130 	} else {
6131 		new_is_dir = d_is_dir(new_dentry);
6132 
6133 		if (!(flags & RENAME_EXCHANGE))
6134 			error = may_delete_dentry(rd->mnt_idmap, new_dir,
6135 						  new_dentry, is_dir);
6136 		else
6137 			error = may_delete_dentry(rd->mnt_idmap, new_dir,
6138 						  new_dentry, new_is_dir);
6139 	}
6140 	if (error)
6141 		return error;
6142 
6143 	if (!old_dir->i_op->rename)
6144 		return -EPERM;
6145 
6146 	/*
6147 	 * If we are going to change the parent - check write permissions,
6148 	 * we'll need to flip '..'.
6149 	 */
6150 	if (new_dir != old_dir) {
6151 		if (is_dir) {
6152 			error = inode_permission(rd->mnt_idmap, source,
6153 						 MAY_WRITE);
6154 			if (error)
6155 				return error;
6156 		}
6157 		if ((flags & RENAME_EXCHANGE) && new_is_dir) {
6158 			error = inode_permission(rd->mnt_idmap, target,
6159 						 MAY_WRITE);
6160 			if (error)
6161 				return error;
6162 		}
6163 	}
6164 
6165 	error = security_inode_rename(old_dir, old_dentry, new_dir, new_dentry,
6166 				      flags);
6167 	if (error)
6168 		return error;
6169 
6170 	take_dentry_name_snapshot(&old_name, old_dentry);
6171 	dget(new_dentry);
6172 	/*
6173 	 * Lock children.
6174 	 * The source subdirectory needs to be locked on cross-directory
6175 	 * rename or cross-directory exchange since its parent changes.
6176 	 * The target subdirectory needs to be locked on cross-directory
6177 	 * exchange due to parent change and on any rename due to becoming
6178 	 * a victim.
6179 	 * Non-directories need locking in all cases (for NFS reasons);
6180 	 * they get locked after any subdirectories (in inode address order).
6181 	 *
6182 	 * NOTE: WE ONLY LOCK UNRELATED DIRECTORIES IN CROSS-DIRECTORY CASE.
6183 	 * NEVER, EVER DO THAT WITHOUT ->s_vfs_rename_mutex.
6184 	 */
6185 	lock_old_subdir = new_dir != old_dir;
6186 	lock_new_subdir = new_dir != old_dir || !(flags & RENAME_EXCHANGE);
6187 	if (is_dir) {
6188 		if (lock_old_subdir)
6189 			inode_lock_nested(source, I_MUTEX_CHILD);
6190 		if (target && (!new_is_dir || lock_new_subdir))
6191 			inode_lock(target);
6192 	} else if (new_is_dir) {
6193 		if (lock_new_subdir)
6194 			inode_lock_nested(target, I_MUTEX_CHILD);
6195 		inode_lock(source);
6196 	} else {
6197 		lock_two_nondirectories(source, target);
6198 	}
6199 
6200 	error = -EPERM;
6201 	if (IS_SWAPFILE(source) || (target && IS_SWAPFILE(target)))
6202 		goto out;
6203 
6204 	error = -EBUSY;
6205 	if (is_local_mountpoint(old_dentry) || is_local_mountpoint(new_dentry))
6206 		goto out;
6207 
6208 	if (max_links && new_dir != old_dir) {
6209 		error = -EMLINK;
6210 		if (is_dir && !new_is_dir && new_dir->i_nlink >= max_links)
6211 			goto out;
6212 		if ((flags & RENAME_EXCHANGE) && !is_dir && new_is_dir &&
6213 		    old_dir->i_nlink >= max_links)
6214 			goto out;
6215 	}
6216 	error = try_break_deleg(old_dir,
6217 				old_dir == new_dir ? LEASE_BREAK_DIR_RENAME :
6218 						     LEASE_BREAK_DIR_DELETE,
6219 				delegated_inode);
6220 	if (error)
6221 		goto out;
6222 	if (new_dir != old_dir) {
6223 		error = try_break_deleg(new_dir, LEASE_BREAK_DIR_CREATE, delegated_inode);
6224 		if (error)
6225 			goto out;
6226 	}
6227 	if (!is_dir) {
6228 		error = try_break_deleg(source, 0, delegated_inode);
6229 		if (error)
6230 			goto out;
6231 	}
6232 	if (target && !new_is_dir) {
6233 		error = try_break_deleg(target, 0, delegated_inode);
6234 		if (error)
6235 			goto out;
6236 	}
6237 	error = old_dir->i_op->rename(rd->mnt_idmap, old_dir, old_dentry,
6238 				      new_dir, new_dentry, flags);
6239 	if (error)
6240 		goto out;
6241 
6242 	if (!(flags & RENAME_EXCHANGE) && target) {
6243 		if (is_dir) {
6244 			shrink_dcache_parent(new_dentry);
6245 			target->i_flags |= S_DEAD;
6246 		}
6247 		dont_mount(new_dentry);
6248 		detach_mounts(new_dentry);
6249 	}
6250 	if (!(old_dir->i_sb->s_type->fs_flags & FS_RENAME_DOES_D_MOVE)) {
6251 		if (!(flags & RENAME_EXCHANGE))
6252 			d_move(old_dentry, new_dentry);
6253 		else
6254 			d_exchange(old_dentry, new_dentry);
6255 	}
6256 out:
6257 	if (!is_dir || lock_old_subdir)
6258 		inode_unlock(source);
6259 	if (target && (!new_is_dir || lock_new_subdir))
6260 		inode_unlock(target);
6261 	dput(new_dentry);
6262 	if (!error) {
6263 		fsnotify_move(old_dir, new_dir, &old_name.name, is_dir,
6264 			      !(flags & RENAME_EXCHANGE) ? target : NULL, old_dentry);
6265 		if (flags & RENAME_EXCHANGE) {
6266 			fsnotify_move(new_dir, old_dir, &old_dentry->d_name,
6267 				      new_is_dir, NULL, new_dentry);
6268 		}
6269 	}
6270 	release_dentry_name_snapshot(&old_name);
6271 
6272 	return error;
6273 }
6274 EXPORT_SYMBOL(vfs_rename);
6275 
6276 int filename_renameat2(int olddfd, struct filename *from,
6277 		       int newdfd, struct filename *to, unsigned int flags)
6278 {
6279 	struct renamedata rd;
6280 	struct path old_path, new_path;
6281 	struct qstr old_last, new_last;
6282 	enum last_type old_type, new_type;
6283 	struct delegated_inode delegated_inode = { };
6284 	unsigned int lookup_flags = 0;
6285 	bool should_retry = false;
6286 	int error;
6287 
6288 	if (flags & ~(RENAME_NOREPLACE | RENAME_EXCHANGE | RENAME_WHITEOUT))
6289 		return -EINVAL;
6290 
6291 	if ((flags & (RENAME_NOREPLACE | RENAME_WHITEOUT)) &&
6292 	    (flags & RENAME_EXCHANGE))
6293 		return -EINVAL;
6294 
6295 retry:
6296 	error = filename_parentat(olddfd, from, lookup_flags, &old_path,
6297 				  &old_last, &old_type);
6298 	if (error)
6299 		return error;
6300 
6301 	error = filename_parentat(newdfd, to, lookup_flags, &new_path, &new_last,
6302 				  &new_type);
6303 	if (error)
6304 		goto exit1;
6305 
6306 	error = -EXDEV;
6307 	if (old_path.mnt != new_path.mnt)
6308 		goto exit2;
6309 
6310 	error = -EBUSY;
6311 	if (old_type != LAST_NORM)
6312 		goto exit2;
6313 
6314 	if (flags & RENAME_NOREPLACE)
6315 		error = -EEXIST;
6316 	if (new_type != LAST_NORM)
6317 		goto exit2;
6318 
6319 	error = mnt_want_write(old_path.mnt);
6320 	if (error)
6321 		goto exit2;
6322 
6323 retry_deleg:
6324 	rd.old_parent	   = old_path.dentry;
6325 	rd.mnt_idmap	   = mnt_idmap(old_path.mnt);
6326 	rd.new_parent	   = new_path.dentry;
6327 	rd.delegated_inode = &delegated_inode;
6328 	rd.flags	   = flags;
6329 
6330 	error = __start_renaming(&rd, lookup_flags, &old_last, &new_last);
6331 	if (error)
6332 		goto exit_lock_rename;
6333 
6334 	if (flags & RENAME_EXCHANGE) {
6335 		if (!d_is_dir(rd.new_dentry)) {
6336 			error = -ENOTDIR;
6337 			if (new_last.name[new_last.len])
6338 				goto exit_unlock;
6339 		}
6340 	}
6341 	/* unless the source is a directory trailing slashes give -ENOTDIR */
6342 	if (!d_is_dir(rd.old_dentry)) {
6343 		error = -ENOTDIR;
6344 		if (old_last.name[old_last.len])
6345 			goto exit_unlock;
6346 		if (!(flags & RENAME_EXCHANGE) && new_last.name[new_last.len])
6347 			goto exit_unlock;
6348 	}
6349 
6350 	error = security_path_rename(&old_path, rd.old_dentry,
6351 				     &new_path, rd.new_dentry, flags);
6352 	if (error)
6353 		goto exit_unlock;
6354 
6355 	error = vfs_rename(&rd);
6356 exit_unlock:
6357 	end_renaming(&rd);
6358 exit_lock_rename:
6359 	if (is_delegated(&delegated_inode)) {
6360 		error = break_deleg_wait(&delegated_inode);
6361 		if (!error)
6362 			goto retry_deleg;
6363 	}
6364 	mnt_drop_write(old_path.mnt);
6365 exit2:
6366 	if (retry_estale(error, lookup_flags))
6367 		should_retry = true;
6368 	path_put(&new_path);
6369 exit1:
6370 	path_put(&old_path);
6371 	if (should_retry) {
6372 		should_retry = false;
6373 		lookup_flags |= LOOKUP_REVAL;
6374 		goto retry;
6375 	}
6376 	return error;
6377 }
6378 
6379 SYSCALL_DEFINE5(renameat2, int, olddfd, const char __user *, oldname,
6380 		int, newdfd, const char __user *, newname, unsigned int, flags)
6381 {
6382 	CLASS(filename, old)(oldname);
6383 	CLASS(filename, new)(newname);
6384 	return filename_renameat2(olddfd, old, newdfd, new, flags);
6385 }
6386 
6387 SYSCALL_DEFINE4(renameat, int, olddfd, const char __user *, oldname,
6388 		int, newdfd, const char __user *, newname)
6389 {
6390 	CLASS(filename, old)(oldname);
6391 	CLASS(filename, new)(newname);
6392 	return filename_renameat2(olddfd, old, newdfd, new, 0);
6393 }
6394 
6395 SYSCALL_DEFINE2(rename, const char __user *, oldname, const char __user *, newname)
6396 {
6397 	CLASS(filename, old)(oldname);
6398 	CLASS(filename, new)(newname);
6399 	return filename_renameat2(AT_FDCWD, old, AT_FDCWD, new, 0);
6400 }
6401 
6402 int readlink_copy(char __user *buffer, int buflen, const char *link, int linklen)
6403 {
6404 	int copylen;
6405 
6406 	copylen = linklen;
6407 	if (unlikely(copylen > (unsigned) buflen))
6408 		copylen = buflen;
6409 	if (copy_to_user(buffer, link, copylen))
6410 		copylen = -EFAULT;
6411 	return copylen;
6412 }
6413 
6414 /**
6415  * vfs_readlink - copy symlink body into userspace buffer
6416  * @dentry: dentry on which to get symbolic link
6417  * @buffer: user memory pointer
6418  * @buflen: size of buffer
6419  *
6420  * Does not touch atime.  That's up to the caller if necessary
6421  *
6422  * Does not call security hook.
6423  */
6424 int vfs_readlink(struct dentry *dentry, char __user *buffer, int buflen)
6425 {
6426 	struct inode *inode = d_inode(dentry);
6427 	DEFINE_DELAYED_CALL(done);
6428 	const char *link;
6429 	int res;
6430 
6431 	if (inode->i_opflags & IOP_CACHED_LINK)
6432 		return readlink_copy(buffer, buflen, inode->i_link, inode->i_linklen);
6433 
6434 	if (unlikely(!(inode->i_opflags & IOP_DEFAULT_READLINK))) {
6435 		if (unlikely(inode->i_op->readlink))
6436 			return inode->i_op->readlink(dentry, buffer, buflen);
6437 
6438 		if (!d_is_symlink(dentry))
6439 			return -EINVAL;
6440 
6441 		spin_lock(&inode->i_lock);
6442 		inode->i_opflags |= IOP_DEFAULT_READLINK;
6443 		spin_unlock(&inode->i_lock);
6444 	}
6445 
6446 	link = READ_ONCE(inode->i_link);
6447 	if (!link) {
6448 		link = inode->i_op->get_link(dentry, inode, &done);
6449 		if (IS_ERR(link))
6450 			return PTR_ERR(link);
6451 	}
6452 	res = readlink_copy(buffer, buflen, link, strlen(link));
6453 	do_delayed_call(&done);
6454 	return res;
6455 }
6456 EXPORT_SYMBOL(vfs_readlink);
6457 
6458 /**
6459  * vfs_get_link - get symlink body
6460  * @dentry: dentry on which to get symbolic link
6461  * @done: caller needs to free returned data with this
6462  *
6463  * Calls security hook and i_op->get_link() on the supplied inode.
6464  *
6465  * It does not touch atime.  That's up to the caller if necessary.
6466  *
6467  * Does not work on "special" symlinks like /proc/$$/fd/N
6468  */
6469 const char *vfs_get_link(struct dentry *dentry, struct delayed_call *done)
6470 {
6471 	const char *res = ERR_PTR(-EINVAL);
6472 	struct inode *inode = d_inode(dentry);
6473 
6474 	if (d_is_symlink(dentry)) {
6475 		res = ERR_PTR(security_inode_readlink(dentry));
6476 		if (!res)
6477 			res = inode->i_op->get_link(dentry, inode, done);
6478 	}
6479 	return res;
6480 }
6481 EXPORT_SYMBOL(vfs_get_link);
6482 
6483 /* get the link contents into pagecache */
6484 static char *__page_get_link(struct dentry *dentry, struct inode *inode,
6485 			     struct delayed_call *callback)
6486 {
6487 	struct folio *folio;
6488 	struct address_space *mapping = inode->i_mapping;
6489 
6490 	if (!dentry) {
6491 		folio = filemap_get_folio(mapping, 0);
6492 		if (IS_ERR(folio))
6493 			return ERR_PTR(-ECHILD);
6494 		if (!folio_test_uptodate(folio)) {
6495 			folio_put(folio);
6496 			return ERR_PTR(-ECHILD);
6497 		}
6498 	} else {
6499 		folio = read_mapping_folio(mapping, 0, NULL);
6500 		if (IS_ERR(folio))
6501 			return ERR_CAST(folio);
6502 	}
6503 	set_delayed_call(callback, page_put_link, folio);
6504 	BUG_ON(mapping_gfp_mask(mapping) & __GFP_HIGHMEM);
6505 	return folio_address(folio);
6506 }
6507 
6508 const char *page_get_link_raw(struct dentry *dentry, struct inode *inode,
6509 			      struct delayed_call *callback)
6510 {
6511 	return __page_get_link(dentry, inode, callback);
6512 }
6513 EXPORT_SYMBOL_GPL(page_get_link_raw);
6514 
6515 /**
6516  * page_get_link() - An implementation of the get_link inode_operation.
6517  * @dentry: The directory entry which is the symlink.
6518  * @inode: The inode for the symlink.
6519  * @callback: Used to drop the reference to the symlink.
6520  *
6521  * Filesystems which store their symlinks in the page cache should use
6522  * this to implement the get_link() member of their inode_operations.
6523  *
6524  * Return: A pointer to the NUL-terminated symlink.
6525  */
6526 const char *page_get_link(struct dentry *dentry, struct inode *inode,
6527 					struct delayed_call *callback)
6528 {
6529 	char *kaddr = __page_get_link(dentry, inode, callback);
6530 
6531 	if (!IS_ERR(kaddr))
6532 		nd_terminate_link(kaddr, inode->i_size, PAGE_SIZE - 1);
6533 	return kaddr;
6534 }
6535 EXPORT_SYMBOL(page_get_link);
6536 
6537 /**
6538  * page_put_link() - Drop the reference to the symlink.
6539  * @arg: The folio which contains the symlink.
6540  *
6541  * This is used internally by page_get_link().  It is exported for use
6542  * by filesystems which need to implement a variant of page_get_link()
6543  * themselves.  Despite the apparent symmetry, filesystems which use
6544  * page_get_link() do not need to call page_put_link().
6545  *
6546  * The argument, while it has a void pointer type, must be a pointer to
6547  * the folio which was retrieved from the page cache.  The delayed_call
6548  * infrastructure is used to drop the reference count once the caller
6549  * is done with the symlink.
6550  */
6551 void page_put_link(void *arg)
6552 {
6553 	folio_put(arg);
6554 }
6555 EXPORT_SYMBOL(page_put_link);
6556 
6557 int page_readlink(struct dentry *dentry, char __user *buffer, int buflen)
6558 {
6559 	const char *link;
6560 	int res;
6561 
6562 	DEFINE_DELAYED_CALL(done);
6563 	link = page_get_link(dentry, d_inode(dentry), &done);
6564 	res = PTR_ERR(link);
6565 	if (!IS_ERR(link))
6566 		res = readlink_copy(buffer, buflen, link, strlen(link));
6567 	do_delayed_call(&done);
6568 	return res;
6569 }
6570 EXPORT_SYMBOL(page_readlink);
6571 
6572 int page_symlink(struct inode *inode, const char *symname, int len)
6573 {
6574 	struct address_space *mapping = inode->i_mapping;
6575 	const struct address_space_operations *aops = mapping->a_ops;
6576 	bool nofs = !mapping_gfp_constraint(mapping, __GFP_FS);
6577 	struct folio *folio;
6578 	void *fsdata = NULL;
6579 	int err;
6580 	unsigned int flags;
6581 
6582 retry:
6583 	if (nofs)
6584 		flags = memalloc_nofs_save();
6585 	err = aops->write_begin(NULL, mapping, 0, len-1, &folio, &fsdata);
6586 	if (nofs)
6587 		memalloc_nofs_restore(flags);
6588 	if (err)
6589 		goto fail;
6590 
6591 	memcpy(folio_address(folio), symname, len - 1);
6592 
6593 	err = aops->write_end(NULL, mapping, 0, len - 1, len - 1,
6594 						folio, fsdata);
6595 	if (err < 0)
6596 		goto fail;
6597 	if (err < len-1)
6598 		goto retry;
6599 
6600 	mark_inode_dirty(inode);
6601 	return 0;
6602 fail:
6603 	return err;
6604 }
6605 EXPORT_SYMBOL(page_symlink);
6606 
6607 const struct inode_operations page_symlink_inode_operations = {
6608 	.get_link	= page_get_link,
6609 };
6610 EXPORT_SYMBOL(page_symlink_inode_operations);
6611