1 // SPDX-License-Identifier: GPL-2.0-only 2 /* 3 * linux/fs/namespace.c 4 * 5 * (C) Copyright Al Viro 2000, 2001 6 * 7 * Based on code from fs/super.c, copyright Linus Torvalds and others. 8 * Heavily rewritten. 9 */ 10 11 #include <linux/syscalls.h> 12 #include <linux/export.h> 13 #include <linux/capability.h> 14 #include <linux/mnt_namespace.h> 15 #include <linux/user_namespace.h> 16 #include <linux/namei.h> 17 #include <linux/security.h> 18 #include <linux/cred.h> 19 #include <linux/idr.h> 20 #include <linux/init.h> /* init_rootfs */ 21 #include <linux/fs_struct.h> /* get_fs_root et.al. */ 22 #include <linux/fsnotify.h> /* fsnotify_vfsmount_delete */ 23 #include <linux/file.h> 24 #include <linux/uaccess.h> 25 #include <linux/proc_ns.h> 26 #include <linux/magic.h> 27 #include <linux/memblock.h> 28 #include <linux/proc_fs.h> 29 #include <linux/task_work.h> 30 #include <linux/sched/task.h> 31 #include <uapi/linux/mount.h> 32 #include <linux/fs_context.h> 33 #include <linux/shmem_fs.h> 34 #include <linux/mnt_idmapping.h> 35 #include <linux/pidfs.h> 36 #include <linux/nstree.h> 37 38 #include "pnode.h" 39 #include "internal.h" 40 41 /* Maximum number of mounts in a mount namespace */ 42 static unsigned int sysctl_mount_max __read_mostly = 100000; 43 44 static unsigned int m_hash_mask __ro_after_init; 45 static unsigned int m_hash_shift __ro_after_init; 46 static unsigned int mp_hash_mask __ro_after_init; 47 static unsigned int mp_hash_shift __ro_after_init; 48 49 static __initdata unsigned long mhash_entries; 50 static int __init set_mhash_entries(char *str) 51 { 52 return kstrtoul(str, 0, &mhash_entries) == 0; 53 } 54 __setup("mhash_entries=", set_mhash_entries); 55 56 static __initdata unsigned long mphash_entries; 57 static int __init set_mphash_entries(char *str) 58 { 59 return kstrtoul(str, 0, &mphash_entries) == 0; 60 } 61 __setup("mphash_entries=", set_mphash_entries); 62 63 static char * __initdata initramfs_options; 64 static int __init initramfs_options_setup(char *str) 65 { 66 initramfs_options = str; 67 return 1; 68 } 69 70 __setup("initramfs_options=", initramfs_options_setup); 71 72 static u64 event; 73 static DEFINE_XARRAY_FLAGS(mnt_id_xa, XA_FLAGS_ALLOC); 74 static DEFINE_IDA(mnt_group_ida); 75 76 /* Don't allow confusion with old 32bit mount ID */ 77 #define MNT_UNIQUE_ID_OFFSET (1ULL << 31) 78 static u64 mnt_id_ctr = MNT_UNIQUE_ID_OFFSET; 79 80 static struct hlist_head *mount_hashtable __ro_after_init; 81 static struct hlist_head *mountpoint_hashtable __ro_after_init; 82 static struct kmem_cache *mnt_cache __ro_after_init; 83 struct vfsmount *knullfs __ro_after_init; /* private nullfs instance */ 84 static struct vfsmount *knullfs_file __ro_after_init; /* its regular file */ 85 static DECLARE_RWSEM(namespace_sem); 86 static HLIST_HEAD(unmounted); /* protected by namespace_sem */ 87 static LIST_HEAD(ex_mountpoints); /* protected by namespace_sem */ 88 static struct mnt_namespace *emptied_ns; /* protected by namespace_sem */ 89 90 static inline void namespace_lock(void); 91 static void namespace_unlock(void); 92 DEFINE_LOCK_GUARD_0(namespace_excl, namespace_lock(), namespace_unlock()) 93 DEFINE_LOCK_GUARD_0(namespace_shared, down_read(&namespace_sem), 94 up_read(&namespace_sem)) 95 96 DEFINE_FREE(mntput, struct vfsmount *, if (!IS_ERR(_T)) mntput(_T)) 97 98 #ifdef CONFIG_FSNOTIFY 99 LIST_HEAD(notify_list); /* protected by namespace_sem */ 100 #endif 101 102 enum mount_kattr_flags_t { 103 MOUNT_KATTR_RECURSE = (1 << 0), 104 MOUNT_KATTR_IDMAP_REPLACE = (1 << 1), 105 }; 106 107 struct mount_kattr { 108 unsigned int attr_set; 109 unsigned int attr_clr; 110 unsigned int propagation; 111 unsigned int lookup_flags; 112 enum mount_kattr_flags_t kflags; 113 struct user_namespace *mnt_userns; 114 struct mnt_idmap *mnt_idmap; 115 }; 116 117 /* /sys/fs */ 118 struct kobject *fs_kobj __ro_after_init; 119 EXPORT_SYMBOL_GPL(fs_kobj); 120 121 /* 122 * vfsmount lock may be taken for read to prevent changes to the 123 * vfsmount hash, ie. during mountpoint lookups or walking back 124 * up the tree. 125 * 126 * It should be taken for write in all cases where the vfsmount 127 * tree or hash is modified or when a vfsmount structure is modified. 128 */ 129 __cacheline_aligned_in_smp DEFINE_SEQLOCK(mount_lock); 130 131 static void mnt_ns_release(struct mnt_namespace *ns) 132 { 133 /* keep alive for {list,stat}mount() */ 134 if (ns && refcount_dec_and_test(&ns->passive)) { 135 put_user_ns(ns->user_ns); 136 kfree(ns); 137 } 138 } 139 DEFINE_FREE(mnt_ns_release, struct mnt_namespace *, 140 if (!IS_ERR(_T)) mnt_ns_release(_T)) 141 142 static void mnt_ns_release_rcu(struct rcu_head *rcu) 143 { 144 mnt_ns_release(container_of(rcu, struct mnt_namespace, ns.ns_rcu)); 145 } 146 147 static void mnt_ns_tree_remove(struct mnt_namespace *ns) 148 { 149 /* remove from global mount namespace list */ 150 if (ns_tree_active(ns)) 151 ns_tree_remove(ns); 152 153 call_rcu(&ns->ns.ns_rcu, mnt_ns_release_rcu); 154 } 155 156 /* 157 * Lookup a mount namespace by id and take a passive reference count. Taking a 158 * passive reference means the mount namespace can be emptied if e.g., the last 159 * task holding an active reference exits. To access the mounts of the 160 * namespace the @namespace_sem must first be acquired. If the namespace has 161 * already shut down before acquiring @namespace_sem, {list,stat}mount() will 162 * see that the mount rbtree of the namespace is empty. 163 * 164 * Note the lookup is lockless protected by a sequence counter. We only 165 * need to guard against false negatives as false positives aren't 166 * possible. So if we didn't find a mount namespace and the sequence 167 * counter has changed we need to retry. If the sequence counter is 168 * still the same we know the search actually failed. 169 */ 170 static struct mnt_namespace *lookup_mnt_ns(u64 mnt_ns_id) 171 { 172 struct mnt_namespace *mnt_ns; 173 struct ns_common *ns; 174 175 guard(rcu)(); 176 ns = ns_tree_lookup_rcu(mnt_ns_id, CLONE_NEWNS); 177 if (!ns) 178 return NULL; 179 180 /* 181 * The last reference count is put with RCU delay so we can 182 * unconditonally acquire a reference here. 183 */ 184 mnt_ns = container_of(ns, struct mnt_namespace, ns); 185 refcount_inc(&mnt_ns->passive); 186 return mnt_ns; 187 } 188 189 static inline void lock_mount_hash(void) 190 { 191 write_seqlock(&mount_lock); 192 } 193 194 static inline void unlock_mount_hash(void) 195 { 196 write_sequnlock(&mount_lock); 197 } 198 199 static inline struct hlist_head *m_hash(struct vfsmount *mnt, struct dentry *dentry) 200 { 201 unsigned long tmp = ((unsigned long)mnt / L1_CACHE_BYTES); 202 tmp += ((unsigned long)dentry / L1_CACHE_BYTES); 203 tmp = tmp + (tmp >> m_hash_shift); 204 return &mount_hashtable[tmp & m_hash_mask]; 205 } 206 207 static inline struct hlist_head *mp_hash(struct dentry *dentry) 208 { 209 unsigned long tmp = ((unsigned long)dentry / L1_CACHE_BYTES); 210 tmp = tmp + (tmp >> mp_hash_shift); 211 return &mountpoint_hashtable[tmp & mp_hash_mask]; 212 } 213 214 /* 215 * What an unmounted mount leaves behind at its unmounted parent instead of 216 * staying attached to it. A lookup on the parent at the mountpoint finds 217 * a stand-in for as long as the cover is there. The parent frees it. 218 */ 219 struct mnt_cover { 220 struct hlist_node node; /* parent->mnt_covers, RCU */ 221 struct hlist_node pin; /* mp->m_covers, keeps the mountpoint */ 222 struct dentry *dentry; 223 struct mountpoint *mp; 224 struct rcu_head rcu; 225 }; 226 227 static int mnt_alloc_id(struct mount *mnt) 228 { 229 int res; 230 231 xa_lock(&mnt_id_xa); 232 res = __xa_alloc(&mnt_id_xa, &mnt->mnt_id, mnt, xa_limit_31b, GFP_KERNEL); 233 if (!res) 234 mnt->mnt_id_unique = ++mnt_id_ctr; 235 xa_unlock(&mnt_id_xa); 236 return res; 237 } 238 239 static void mnt_free_id(struct mount *mnt) 240 { 241 xa_erase(&mnt_id_xa, mnt->mnt_id); 242 } 243 244 /* 245 * Allocate a new peer group ID 246 */ 247 static int mnt_alloc_group_id(struct mount *mnt) 248 { 249 int res = ida_alloc_min(&mnt_group_ida, 1, GFP_KERNEL); 250 251 if (res < 0) 252 return res; 253 mnt->mnt_group_id = res; 254 return 0; 255 } 256 257 /* 258 * Release a peer group ID 259 */ 260 void mnt_release_group_id(struct mount *mnt) 261 { 262 ida_free(&mnt_group_ida, mnt->mnt_group_id); 263 mnt->mnt_group_id = 0; 264 } 265 266 static inline void mnt_inc_count(struct mount *mnt) 267 { 268 #ifdef CONFIG_SMP 269 this_cpu_inc(mnt->mnt_pcp->mnt_gets); 270 #else 271 preempt_disable(); 272 mnt->mnt_count++; 273 preempt_enable(); 274 #endif 275 } 276 277 static inline void mnt_dec_count(struct mount *mnt) 278 { 279 #ifdef CONFIG_SMP 280 this_cpu_inc(mnt->mnt_pcp->mnt_puts); 281 #else 282 preempt_disable(); 283 mnt->mnt_count--; 284 preempt_enable(); 285 #endif 286 } 287 288 /* 289 * vfsmount lock must be held for write 290 */ 291 int mnt_get_count(struct mount *mnt) 292 { 293 #ifdef CONFIG_SMP 294 unsigned int gets = 0, puts = 0; 295 int cpu; 296 297 /* puts first, so a put counted here has its get counted below */ 298 for_each_possible_cpu(cpu) 299 puts += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_puts; 300 smp_mb(); /* pairs with the smp_wmb() in mntput_no_expire() */ 301 for_each_possible_cpu(cpu) 302 gets += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_gets; 303 304 return gets - puts; 305 #else 306 return mnt->mnt_count; 307 #endif 308 } 309 310 static struct mount *alloc_vfsmnt(const char *name) 311 { 312 struct mount *mnt = kmem_cache_zalloc(mnt_cache, GFP_KERNEL); 313 if (mnt) { 314 int err; 315 316 mnt->mnt_cover = kzalloc_obj(struct mnt_cover, GFP_KERNEL_ACCOUNT); 317 if (!mnt->mnt_cover) 318 goto out_free_cache; 319 320 err = mnt_alloc_id(mnt); 321 if (err) 322 goto out_free_cache; 323 324 if (name) 325 mnt->mnt_devname = kstrdup_const(name, 326 GFP_KERNEL_ACCOUNT); 327 else 328 mnt->mnt_devname = "none"; 329 if (!mnt->mnt_devname) 330 goto out_free_id; 331 332 #ifdef CONFIG_SMP 333 mnt->mnt_pcp = alloc_percpu(struct mnt_pcp); 334 if (!mnt->mnt_pcp) 335 goto out_free_devname; 336 337 this_cpu_inc(mnt->mnt_pcp->mnt_gets); 338 #else 339 mnt->mnt_count = 1; 340 mnt->mnt_writers = 0; 341 #endif 342 343 INIT_HLIST_NODE(&mnt->mnt_hash); 344 INIT_LIST_HEAD(&mnt->mnt_child); 345 INIT_LIST_HEAD(&mnt->mnt_mounts); 346 INIT_LIST_HEAD(&mnt->mnt_list); 347 INIT_LIST_HEAD(&mnt->mnt_expire); 348 INIT_LIST_HEAD(&mnt->mnt_share); 349 INIT_HLIST_HEAD(&mnt->mnt_slave_list); 350 INIT_HLIST_NODE(&mnt->mnt_slave); 351 INIT_HLIST_NODE(&mnt->mnt_mp_list); 352 INIT_HLIST_HEAD(&mnt->mnt_covers); 353 INIT_HLIST_NODE(&mnt->mnt_ns_visible); 354 #ifdef CONFIG_FSNOTIFY 355 INIT_LIST_HEAD(&mnt->to_notify); 356 #endif 357 RB_CLEAR_NODE(&mnt->mnt_node); 358 mnt->mnt.mnt_idmap = &nop_mnt_idmap; 359 } 360 return mnt; 361 362 #ifdef CONFIG_SMP 363 out_free_devname: 364 kfree_const(mnt->mnt_devname); 365 #endif 366 out_free_id: 367 mnt_free_id(mnt); 368 out_free_cache: 369 kfree(mnt->mnt_cover); 370 kmem_cache_free(mnt_cache, mnt); 371 return NULL; 372 } 373 374 /* 375 * Most r/o checks on a fs are for operations that take 376 * discrete amounts of time, like a write() or unlink(). 377 * We must keep track of when those operations start 378 * (for permission checks) and when they end, so that 379 * we can determine when writes are able to occur to 380 * a filesystem. 381 */ 382 /* 383 * __mnt_is_readonly: check whether a mount is read-only 384 * @mnt: the mount to check for its write status 385 * 386 * This shouldn't be used directly ouside of the VFS. 387 * It does not guarantee that the filesystem will stay 388 * r/w, just that it is right *now*. This can not and 389 * should not be used in place of IS_RDONLY(inode). 390 * mnt_want/drop_write() will _keep_ the filesystem 391 * r/w. 392 */ 393 bool __mnt_is_readonly(const struct vfsmount *mnt) 394 { 395 return (mnt->mnt_flags & MNT_READONLY) || sb_rdonly(mnt->mnt_sb); 396 } 397 EXPORT_SYMBOL_GPL(__mnt_is_readonly); 398 399 static inline void mnt_inc_writers(struct mount *mnt) 400 { 401 #ifdef CONFIG_SMP 402 this_cpu_inc(mnt->mnt_pcp->mnt_writers); 403 #else 404 mnt->mnt_writers++; 405 #endif 406 } 407 408 static inline void mnt_dec_writers(struct mount *mnt) 409 { 410 #ifdef CONFIG_SMP 411 this_cpu_dec(mnt->mnt_pcp->mnt_writers); 412 #else 413 mnt->mnt_writers--; 414 #endif 415 } 416 417 static unsigned int mnt_get_writers(struct mount *mnt) 418 { 419 #ifdef CONFIG_SMP 420 unsigned int count = 0; 421 int cpu; 422 423 for_each_possible_cpu(cpu) { 424 count += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_writers; 425 } 426 427 return count; 428 #else 429 return mnt->mnt_writers; 430 #endif 431 } 432 433 static int mnt_is_readonly(const struct vfsmount *mnt) 434 { 435 if (READ_ONCE(mnt->mnt_sb->s_readonly_remount)) 436 return 1; 437 /* 438 * The barrier pairs with the barrier in sb_start_ro_state_change() 439 * making sure if we don't see s_readonly_remount set yet, we also will 440 * not see any superblock / mount flag changes done by remount. 441 * It also pairs with the barrier in sb_end_ro_state_change() 442 * assuring that if we see s_readonly_remount already cleared, we will 443 * see the values of superblock / mount flags updated by remount. 444 */ 445 smp_rmb(); 446 return __mnt_is_readonly(mnt); 447 } 448 449 /* 450 * Most r/o & frozen checks on a fs are for operations that take discrete 451 * amounts of time, like a write() or unlink(). We must keep track of when 452 * those operations start (for permission checks) and when they end, so that we 453 * can determine when writes are able to occur to a filesystem. 454 */ 455 /** 456 * mnt_get_write_access - get write access to a mount without freeze protection 457 * @m: the mount on which to take a write 458 * 459 * This tells the low-level filesystem that a write is about to be performed to 460 * it, and makes sure that writes are allowed (mnt it read-write) before 461 * returning success. This operation does not protect against filesystem being 462 * frozen. When the write operation is finished, mnt_put_write_access() must be 463 * called. This is effectively a refcount. 464 */ 465 int mnt_get_write_access(struct vfsmount *m) 466 { 467 struct mount *mnt = real_mount(m); 468 int ret = 0; 469 470 preempt_disable(); 471 mnt_inc_writers(mnt); 472 /* 473 * The store to mnt_inc_writers must be visible before we pass 474 * WRITE_HOLD loop below, so that the slowpath can see our 475 * incremented count after it has set WRITE_HOLD. 476 */ 477 smp_mb(); 478 might_lock(&mount_lock.lock); 479 while (__test_write_hold(READ_ONCE(mnt->mnt_pprev_for_sb))) { 480 if (!IS_ENABLED(CONFIG_PREEMPT_RT)) { 481 cpu_relax(); 482 } else { 483 /* 484 * This prevents priority inversion, if the task 485 * setting WRITE_HOLD got preempted on a remote 486 * CPU, and it prevents life lock if the task setting 487 * WRITE_HOLD has a lower priority and is bound to 488 * the same CPU as the task that is spinning here. 489 */ 490 preempt_enable(); 491 read_seqlock_excl(&mount_lock); 492 read_sequnlock_excl(&mount_lock); 493 preempt_disable(); 494 } 495 } 496 /* 497 * The barrier pairs with the barrier sb_start_ro_state_change() making 498 * sure that if we see WRITE_HOLD cleared, we will also see 499 * s_readonly_remount set (or even SB_RDONLY / MNT_READONLY flags) in 500 * mnt_is_readonly() and bail in case we are racing with remount 501 * read-only. 502 */ 503 smp_rmb(); 504 if (mnt_is_readonly(m)) { 505 mnt_dec_writers(mnt); 506 ret = -EROFS; 507 } 508 preempt_enable(); 509 510 return ret; 511 } 512 EXPORT_SYMBOL_GPL(mnt_get_write_access); 513 514 /** 515 * mnt_want_write - get write access to a mount 516 * @m: the mount on which to take a write 517 * 518 * This tells the low-level filesystem that a write is about to be performed to 519 * it, and makes sure that writes are allowed (mount is read-write, filesystem 520 * is not frozen) before returning success. When the write operation is 521 * finished, mnt_drop_write() must be called. This is effectively a refcount. 522 */ 523 int mnt_want_write(struct vfsmount *m) 524 { 525 int ret; 526 527 sb_start_write(m->mnt_sb); 528 ret = mnt_get_write_access(m); 529 if (ret) 530 sb_end_write(m->mnt_sb); 531 return ret; 532 } 533 EXPORT_SYMBOL_GPL(mnt_want_write); 534 535 /** 536 * mnt_get_write_access_file - get write access to a file's mount 537 * @file: the file who's mount on which to take a write 538 * 539 * This is like mnt_get_write_access, but if @file is already open for write it 540 * skips incrementing mnt_writers (since the open file already has a reference) 541 * and instead only does the check for emergency r/o remounts. This must be 542 * paired with mnt_put_write_access_file. 543 */ 544 int mnt_get_write_access_file(struct file *file) 545 { 546 if (file->f_mode & FMODE_WRITER) { 547 /* 548 * Superblock may have become readonly while there are still 549 * writable fd's, e.g. due to a fs error with errors=remount-ro 550 */ 551 if (__mnt_is_readonly(file->f_path.mnt)) 552 return -EROFS; 553 return 0; 554 } 555 return mnt_get_write_access(file->f_path.mnt); 556 } 557 558 /** 559 * mnt_want_write_file - get write access to a file's mount 560 * @file: the file who's mount on which to take a write 561 * 562 * This is like mnt_want_write, but if the file is already open for writing it 563 * skips incrementing mnt_writers (since the open file already has a reference) 564 * and instead only does the freeze protection and the check for emergency r/o 565 * remounts. This must be paired with mnt_drop_write_file. 566 */ 567 int mnt_want_write_file(struct file *file) 568 { 569 int ret; 570 571 sb_start_write(file_inode(file)->i_sb); 572 ret = mnt_get_write_access_file(file); 573 if (ret) 574 sb_end_write(file_inode(file)->i_sb); 575 return ret; 576 } 577 EXPORT_SYMBOL_GPL(mnt_want_write_file); 578 579 /** 580 * mnt_put_write_access - give up write access to a mount 581 * @mnt: the mount on which to give up write access 582 * 583 * Tells the low-level filesystem that we are done 584 * performing writes to it. Must be matched with 585 * mnt_get_write_access() call above. 586 */ 587 void mnt_put_write_access(struct vfsmount *mnt) 588 { 589 preempt_disable(); 590 mnt_dec_writers(real_mount(mnt)); 591 preempt_enable(); 592 } 593 EXPORT_SYMBOL_GPL(mnt_put_write_access); 594 595 /** 596 * mnt_drop_write - give up write access to a mount 597 * @mnt: the mount on which to give up write access 598 * 599 * Tells the low-level filesystem that we are done performing writes to it and 600 * also allows filesystem to be frozen again. Must be matched with 601 * mnt_want_write() call above. 602 */ 603 void mnt_drop_write(struct vfsmount *mnt) 604 { 605 mnt_put_write_access(mnt); 606 sb_end_write(mnt->mnt_sb); 607 } 608 EXPORT_SYMBOL_GPL(mnt_drop_write); 609 610 void mnt_put_write_access_file(struct file *file) 611 { 612 if (!(file->f_mode & FMODE_WRITER)) 613 mnt_put_write_access(file->f_path.mnt); 614 } 615 616 void mnt_drop_write_file(struct file *file) 617 { 618 mnt_put_write_access_file(file); 619 sb_end_write(file_inode(file)->i_sb); 620 } 621 EXPORT_SYMBOL(mnt_drop_write_file); 622 623 /** 624 * mnt_hold_writers - prevent write access to the given mount 625 * @mnt: mnt to prevent write access to 626 * 627 * Prevents write access to @mnt if there are no active writers for @mnt. 628 * This function needs to be called and return successfully before changing 629 * properties of @mnt that need to remain stable for callers with write access 630 * to @mnt. 631 * 632 * After this functions has been called successfully callers must pair it with 633 * a call to mnt_unhold_writers() in order to stop preventing write access to 634 * @mnt. 635 * 636 * Context: This function expects to be in mount_locked_reader scope serializing 637 * setting WRITE_HOLD. 638 * Return: On success 0 is returned. 639 * On error, -EBUSY is returned. 640 */ 641 static inline int mnt_hold_writers(struct mount *mnt) 642 { 643 set_write_hold(mnt); 644 /* 645 * After storing WRITE_HOLD, we'll read the counters. This store 646 * should be visible before we do. 647 */ 648 smp_mb(); 649 650 /* 651 * With writers on hold, if this value is zero, then there are 652 * definitely no active writers (although held writers may subsequently 653 * increment the count, they'll have to wait, and decrement it after 654 * seeing MNT_READONLY). 655 * 656 * It is OK to have counter incremented on one CPU and decremented on 657 * another: the sum will add up correctly. The danger would be when we 658 * sum up each counter, if we read a counter before it is incremented, 659 * but then read another CPU's count which it has been subsequently 660 * decremented from -- we would see more decrements than we should. 661 * WRITE_HOLD protects against this scenario, because 662 * mnt_want_write first increments count, then smp_mb, then spins on 663 * WRITE_HOLD, so it can't be decremented by another CPU while 664 * we're counting up here. 665 */ 666 if (mnt_get_writers(mnt) > 0) 667 return -EBUSY; 668 669 return 0; 670 } 671 672 /** 673 * mnt_unhold_writers - stop preventing write access to the given mount 674 * @mnt: mnt to stop preventing write access to 675 * 676 * Stop preventing write access to @mnt allowing callers to gain write access 677 * to @mnt again. 678 * 679 * This function can only be called after a call to mnt_hold_writers(). 680 * 681 * Context: This function expects to be in the same mount_locked_reader scope 682 * as the matching mnt_hold_writers(). 683 */ 684 static inline void mnt_unhold_writers(struct mount *mnt) 685 { 686 if (!test_write_hold(mnt)) 687 return; 688 /* 689 * MNT_READONLY must become visible before ~WRITE_HOLD, so writers 690 * that become unheld will see MNT_READONLY. 691 */ 692 smp_wmb(); 693 clear_write_hold(mnt); 694 } 695 696 static inline void mnt_del_instance(struct mount *m) 697 { 698 struct mount **p = m->mnt_pprev_for_sb; 699 struct mount *next = m->mnt_next_for_sb; 700 701 if (next) 702 next->mnt_pprev_for_sb = p; 703 *p = next; 704 } 705 706 static inline void mnt_add_instance(struct mount *m, struct super_block *s) 707 { 708 struct mount *first = s->s_mounts; 709 710 if (first) 711 first->mnt_pprev_for_sb = &m->mnt_next_for_sb; 712 m->mnt_next_for_sb = first; 713 m->mnt_pprev_for_sb = &s->s_mounts; 714 s->s_mounts = m; 715 } 716 717 static int mnt_make_readonly(struct mount *mnt) 718 { 719 int ret; 720 721 ret = mnt_hold_writers(mnt); 722 if (!ret) 723 mnt->mnt.mnt_flags |= MNT_READONLY; 724 mnt_unhold_writers(mnt); 725 return ret; 726 } 727 728 int sb_prepare_remount_readonly(struct super_block *sb) 729 { 730 int err = 0; 731 732 /* Racy optimization. Recheck the counter under WRITE_HOLD */ 733 if (atomic_long_read(&sb->s_remove_count)) 734 return -EBUSY; 735 736 guard(mount_locked_reader)(); 737 738 for (struct mount *m = sb->s_mounts; m; m = m->mnt_next_for_sb) { 739 if (!(m->mnt.mnt_flags & MNT_READONLY)) { 740 err = mnt_hold_writers(m); 741 if (err) 742 break; 743 } 744 } 745 if (!err && atomic_long_read(&sb->s_remove_count)) 746 err = -EBUSY; 747 748 if (!err) 749 sb_start_ro_state_change(sb); 750 for (struct mount *m = sb->s_mounts; m; m = m->mnt_next_for_sb) { 751 if (test_write_hold(m)) 752 clear_write_hold(m); 753 } 754 755 return err; 756 } 757 758 static void free_vfsmnt(struct mount *mnt) 759 { 760 mnt_idmap_put(mnt_idmap(&mnt->mnt)); 761 /* NULL if it left it behind */ 762 kfree(mnt->mnt_cover); 763 kfree_const(mnt->mnt_devname); 764 #ifdef CONFIG_SMP 765 free_percpu(mnt->mnt_pcp); 766 #endif 767 kmem_cache_free(mnt_cache, mnt); 768 } 769 770 static void delayed_free_vfsmnt(struct rcu_head *head) 771 { 772 free_vfsmnt(container_of(head, struct mount, mnt_rcu)); 773 } 774 775 /* call under rcu_read_lock */ 776 int __legitimize_mnt(struct vfsmount *bastard, unsigned seq) 777 { 778 struct mount *mnt; 779 if (read_seqretry(&mount_lock, seq)) 780 return 1; 781 if (bastard == NULL) 782 return 0; 783 mnt = real_mount(bastard); 784 mnt_inc_count(mnt); 785 smp_mb(); /* see mntput_no_expire_slowpath() and do_umount() */ 786 if (likely(!read_seqretry(&mount_lock, seq))) 787 return 0; 788 lock_mount_hash(); 789 if (unlikely(bastard->mnt_flags & (MNT_SYNC_UMOUNT | MNT_DOOMED))) { 790 mnt_dec_count(mnt); 791 unlock_mount_hash(); 792 return 1; 793 } 794 unlock_mount_hash(); 795 /* caller will mntput() */ 796 return -1; 797 } 798 799 /* call under rcu_read_lock */ 800 static bool legitimize_mnt(struct vfsmount *bastard, unsigned seq) 801 { 802 int res = __legitimize_mnt(bastard, seq); 803 if (likely(!res)) 804 return true; 805 if (unlikely(res < 0)) { 806 rcu_read_unlock(); 807 mntput(bastard); 808 rcu_read_lock(); 809 } 810 return false; 811 } 812 813 /** 814 * __lookup_mnt - mount hash lookup 815 * @mnt: parent mount 816 * @dentry: dentry of mountpoint 817 * 818 * If @mnt has a child mount @c mounted on @dentry find and return it. 819 * If @mnt is unmounted and a child that was unmounted with it left its 820 * cover behind at @dentry, return the stand-in for it instead: knullfs 821 * for a directory, its regular file for anything else. 822 * Caller must either hold the spinlock component of @mount_lock or 823 * hold rcu_read_lock(), sample the seqcount component before the call 824 * and recheck it afterwards. 825 * 826 * Return: The child of @mnt mounted on @dentry, a stand-in or %NULL. 827 */ 828 struct mount *__lookup_mnt(struct vfsmount *mnt, struct dentry *dentry) 829 { 830 struct hlist_head *head = m_hash(mnt, dentry); 831 struct mnt_cover *cover; 832 struct mount *p; 833 834 hlist_for_each_entry_rcu(p, head, mnt_hash) 835 if (&p->mnt_parent->mnt == mnt && p->mnt_mountpoint == dentry) 836 return p; 837 /* an unmounted mount keeps the covers its unmounted children left */ 838 /* a lockless caller rechecks mount_lock after a miss, a stale flag is harmless */ 839 if (unlikely(data_race(mnt->mnt_flags) & MNT_UMOUNT)) { 840 hlist_for_each_entry_rcu(cover, &real_mount(mnt)->mnt_covers, node) 841 if (cover->dentry == dentry) 842 return real_mount(d_is_dir(dentry) ? knullfs : knullfs_file); 843 } 844 return NULL; 845 } 846 847 /** 848 * lookup_mnt - Return the child mount mounted at given location 849 * @path: location in the namespace 850 * 851 * Acquires and returns a new reference to mount at given location 852 * or %NULL if nothing is mounted there. 853 */ 854 struct vfsmount *lookup_mnt(const struct path *path) 855 { 856 struct mount *child_mnt; 857 struct vfsmount *m; 858 unsigned seq; 859 860 rcu_read_lock(); 861 do { 862 seq = read_seqbegin(&mount_lock); 863 child_mnt = __lookup_mnt(path->mnt, path->dentry); 864 m = child_mnt ? &child_mnt->mnt : NULL; 865 } while (!legitimize_mnt(m, seq)); 866 rcu_read_unlock(); 867 return m; 868 } 869 870 /* 871 * __is_local_mountpoint - Test to see if dentry is a mountpoint in the 872 * current mount namespace. 873 * 874 * The common case is dentries are not mountpoints at all and that 875 * test is handled inline. For the slow case when we are actually 876 * dealing with a mountpoint of some kind, walk through all of the 877 * mounts in the current mount namespace and test to see if the dentry 878 * is a mountpoint. 879 * 880 * The mount_hashtable is not usable in the context because we 881 * need to identify all mounts that may be in the current mount 882 * namespace not just a mount that happens to have some specified 883 * parent mount. 884 */ 885 bool __is_local_mountpoint(const struct dentry *dentry) 886 { 887 struct mnt_namespace *ns = current->nsproxy->mnt_ns; 888 struct mount *mnt, *n; 889 890 guard(namespace_shared)(); 891 892 rbtree_postorder_for_each_entry_safe(mnt, n, &ns->mounts, mnt_node) 893 if (mnt->mnt_mountpoint == dentry) 894 return true; 895 896 return false; 897 } 898 899 struct pinned_mountpoint { 900 struct hlist_node node; 901 struct mountpoint *mp; 902 struct mount *parent; 903 }; 904 905 static bool lookup_mountpoint(struct dentry *dentry, struct pinned_mountpoint *m) 906 { 907 struct hlist_head *chain = mp_hash(dentry); 908 struct mountpoint *mp; 909 910 hlist_for_each_entry(mp, chain, m_hash) { 911 if (mp->m_dentry == dentry) { 912 hlist_add_head(&m->node, &mp->m_list); 913 m->mp = mp; 914 return true; 915 } 916 } 917 return false; 918 } 919 920 static int get_mountpoint(struct dentry *dentry, struct pinned_mountpoint *m) 921 { 922 struct mountpoint *mp __free(kfree) = NULL; 923 bool found; 924 int ret; 925 926 if (d_mountpoint(dentry)) { 927 /* might be worth a WARN_ON() */ 928 if (d_unlinked(dentry)) 929 return -ENOENT; 930 mountpoint: 931 read_seqlock_excl(&mount_lock); 932 found = lookup_mountpoint(dentry, m); 933 read_sequnlock_excl(&mount_lock); 934 if (found) 935 return 0; 936 } 937 938 if (!mp) 939 mp = kmalloc_obj(struct mountpoint); 940 if (!mp) 941 return -ENOMEM; 942 943 /* Exactly one processes may set d_mounted */ 944 ret = d_set_mounted(dentry); 945 946 /* Someone else set d_mounted? */ 947 if (ret == -EBUSY) 948 goto mountpoint; 949 950 /* The dentry is not available as a mountpoint? */ 951 if (ret) 952 return ret; 953 954 /* Add the new mountpoint to the hash table */ 955 read_seqlock_excl(&mount_lock); 956 mp->m_dentry = dget(dentry); 957 hlist_add_head(&mp->m_hash, mp_hash(dentry)); 958 INIT_HLIST_HEAD(&mp->m_list); 959 INIT_HLIST_HEAD(&mp->m_covers); 960 hlist_add_head(&m->node, &mp->m_list); 961 m->mp = no_free_ptr(mp); 962 read_sequnlock_excl(&mount_lock); 963 return 0; 964 } 965 966 /* 967 * vfsmount lock must be held. Additionally, the caller is responsible 968 * for serializing calls for given disposal list. 969 */ 970 static void maybe_free_mountpoint(struct mountpoint *mp, struct list_head *list) 971 { 972 if (hlist_empty(&mp->m_list) && hlist_empty(&mp->m_covers)) { 973 struct dentry *dentry = mp->m_dentry; 974 spin_lock(&dentry->d_lock); 975 dentry->d_flags &= ~DCACHE_MOUNTED; 976 spin_unlock(&dentry->d_lock); 977 dput_to_list(dentry, list); 978 hlist_del(&mp->m_hash); 979 kfree(mp); 980 } 981 } 982 983 /* 984 * locks: mount_lock [read_seqlock_excl], namespace_sem [excl] 985 */ 986 static void unpin_mountpoint(struct pinned_mountpoint *m) 987 { 988 if (m->mp) { 989 hlist_del(&m->node); 990 maybe_free_mountpoint(m->mp, &ex_mountpoints); 991 } 992 } 993 994 static inline int check_mnt(const struct mount *mnt) 995 { 996 return mnt->mnt_ns == current->nsproxy->mnt_ns; 997 } 998 999 static inline bool check_anonymous_mnt(struct mount *mnt) 1000 { 1001 u64 seq; 1002 1003 if (!is_anon_ns(mnt->mnt_ns)) 1004 return false; 1005 1006 seq = mnt->mnt_ns->seq_origin; 1007 return !seq || (seq == current->nsproxy->mnt_ns->ns.ns_id); 1008 } 1009 1010 /* 1011 * vfsmount lock must be held for write 1012 */ 1013 static void touch_mnt_namespace(struct mnt_namespace *ns) 1014 { 1015 if (ns) { 1016 ns->event = ++event; 1017 wake_up_interruptible(&ns->poll); 1018 } 1019 } 1020 1021 /* 1022 * vfsmount lock must be held for write 1023 */ 1024 static void __touch_mnt_namespace(struct mnt_namespace *ns) 1025 { 1026 if (ns && ns->event != event) { 1027 ns->event = event; 1028 wake_up_interruptible(&ns->poll); 1029 } 1030 } 1031 1032 /* 1033 * locks: mount_lock[write_seqlock] 1034 */ 1035 static void __umount_mnt(struct mount *mnt, struct list_head *shrink_list) 1036 { 1037 struct mountpoint *mp; 1038 struct mount *parent = mnt->mnt_parent; 1039 if (unlikely(parent->overmount == mnt)) 1040 parent->overmount = NULL; 1041 mnt->mnt_parent = mnt; 1042 mnt->mnt_mountpoint = mnt->mnt.mnt_root; 1043 list_del_init(&mnt->mnt_child); 1044 hlist_del_init_rcu(&mnt->mnt_hash); 1045 hlist_del_init(&mnt->mnt_mp_list); 1046 mp = mnt->mnt_mp; 1047 mnt->mnt_mp = NULL; 1048 maybe_free_mountpoint(mp, shrink_list); 1049 } 1050 1051 /* 1052 * locks: mount_lock[write_seqlock], namespace_sem[excl] (for ex_mountpoints) 1053 */ 1054 static void umount_mnt(struct mount *mnt) 1055 { 1056 __umount_mnt(mnt, &ex_mountpoints); 1057 } 1058 1059 /* 1060 * @mnt is unmounted together with its parent and would have stayed attached 1061 * to it. Leave its cover behind before it is detached so that a lookup on the 1062 * parent at the mountpoint keeps finding a mount instead of what @mnt covered. 1063 * 1064 * locks: mount_lock[write_seqlock] 1065 */ 1066 static void leave_cover(struct mount *mnt) 1067 { 1068 struct mnt_cover *cover = mnt->mnt_cover; 1069 1070 mnt->mnt_cover = NULL; 1071 cover->dentry = mnt->mnt_mountpoint; 1072 cover->mp = mnt->mnt_mp; 1073 /* keeps the mountpoint once @mnt has let go of it */ 1074 hlist_add_head(&cover->pin, &cover->mp->m_covers); 1075 hlist_add_head_rcu(&cover->node, &mnt->mnt_parent->mnt_covers); 1076 } 1077 1078 /* 1079 * locks: mount_lock[write_seqlock] 1080 */ 1081 static void drop_cover(struct mnt_cover *cover, struct list_head *shrink_list) 1082 { 1083 hlist_del_rcu(&cover->node); 1084 hlist_del(&cover->pin); 1085 maybe_free_mountpoint(cover->mp, shrink_list); 1086 kfree_rcu(cover, rcu); 1087 } 1088 1089 /* 1090 * vfsmount lock must be held for write 1091 */ 1092 void mnt_set_mountpoint(struct mount *mnt, 1093 struct mountpoint *mp, 1094 struct mount *child_mnt) 1095 { 1096 child_mnt->mnt_mountpoint = mp->m_dentry; 1097 child_mnt->mnt_parent = mnt; 1098 child_mnt->mnt_mp = mp; 1099 hlist_add_head(&child_mnt->mnt_mp_list, &mp->m_list); 1100 } 1101 1102 static void make_visible(struct mount *mnt) 1103 { 1104 struct mount *parent = mnt->mnt_parent; 1105 if (unlikely(mnt->mnt_mountpoint == parent->mnt.mnt_root)) 1106 parent->overmount = mnt; 1107 hlist_add_head_rcu(&mnt->mnt_hash, 1108 m_hash(&parent->mnt, mnt->mnt_mountpoint)); 1109 list_add_tail(&mnt->mnt_child, &parent->mnt_mounts); 1110 } 1111 1112 /** 1113 * attach_mnt - mount a mount, attach to @mount_hashtable and parent's 1114 * list of child mounts 1115 * @parent: the parent 1116 * @mnt: the new mount 1117 * @mp: the new mountpoint 1118 * 1119 * Mount @mnt at @mp on @parent. Then attach @mnt 1120 * to @parent's child mount list and to @mount_hashtable. 1121 * 1122 * Note, when make_visible() is called @mnt->mnt_parent already points 1123 * to the correct parent. 1124 * 1125 * Context: This function expects namespace_lock() and lock_mount_hash() 1126 * to have been acquired in that order. 1127 */ 1128 static void attach_mnt(struct mount *mnt, struct mount *parent, 1129 struct mountpoint *mp) 1130 { 1131 mnt_set_mountpoint(parent, mp, mnt); 1132 make_visible(mnt); 1133 } 1134 1135 void mnt_change_mountpoint(struct mount *parent, struct mountpoint *mp, struct mount *mnt) 1136 { 1137 struct mountpoint *old_mp = mnt->mnt_mp; 1138 struct mount *old_parent = mnt->mnt_parent; 1139 1140 if (old_parent->overmount == mnt) 1141 old_parent->overmount = NULL; 1142 list_del_init(&mnt->mnt_child); 1143 hlist_del_init(&mnt->mnt_mp_list); 1144 hlist_del_init_rcu(&mnt->mnt_hash); 1145 1146 attach_mnt(mnt, parent, mp); 1147 1148 maybe_free_mountpoint(old_mp, &ex_mountpoints); 1149 } 1150 1151 static inline struct mount *node_to_mount(struct rb_node *node) 1152 { 1153 return node ? rb_entry(node, struct mount, mnt_node) : NULL; 1154 } 1155 1156 static void mnt_add_to_ns(struct mnt_namespace *ns, struct mount *mnt) 1157 { 1158 struct rb_node **link = &ns->mounts.rb_node; 1159 struct rb_node *parent = NULL; 1160 bool mnt_first_node = true, mnt_last_node = true; 1161 1162 WARN_ON(mnt_ns_attached(mnt)); 1163 WRITE_ONCE(mnt->mnt_ns, ns); 1164 while (*link) { 1165 parent = *link; 1166 if (mnt->mnt_id_unique < node_to_mount(parent)->mnt_id_unique) { 1167 link = &parent->rb_left; 1168 mnt_last_node = false; 1169 } else { 1170 link = &parent->rb_right; 1171 mnt_first_node = false; 1172 } 1173 } 1174 1175 if (mnt_last_node) 1176 ns->mnt_last_node = &mnt->mnt_node; 1177 if (mnt_first_node) 1178 ns->mnt_first_node = &mnt->mnt_node; 1179 rb_link_node(&mnt->mnt_node, parent, link); 1180 rb_insert_color(&mnt->mnt_node, &ns->mounts); 1181 1182 if ((mnt->mnt.mnt_sb->s_type->fs_flags & FS_USERNS_MOUNT_RESTRICTED) && 1183 mnt->mnt.mnt_root == mnt->mnt.mnt_sb->s_root) 1184 hlist_add_head(&mnt->mnt_ns_visible, &ns->mnt_visible_mounts); 1185 1186 mnt_notify_add(mnt); 1187 } 1188 1189 static struct mount *next_mnt(struct mount *p, struct mount *root) 1190 { 1191 struct list_head *next = p->mnt_mounts.next; 1192 if (next == &p->mnt_mounts) { 1193 while (1) { 1194 if (p == root) 1195 return NULL; 1196 next = p->mnt_child.next; 1197 if (next != &p->mnt_parent->mnt_mounts) 1198 break; 1199 p = p->mnt_parent; 1200 } 1201 } 1202 return list_entry(next, struct mount, mnt_child); 1203 } 1204 1205 static struct mount *skip_mnt_tree(struct mount *p) 1206 { 1207 struct list_head *prev = p->mnt_mounts.prev; 1208 while (prev != &p->mnt_mounts) { 1209 p = list_entry(prev, struct mount, mnt_child); 1210 prev = p->mnt_mounts.prev; 1211 } 1212 return p; 1213 } 1214 1215 /* 1216 * vfsmount lock must be held for write 1217 */ 1218 static void commit_tree(struct mount *mnt) 1219 { 1220 struct mnt_namespace *n = mnt->mnt_parent->mnt_ns; 1221 1222 if (!mnt_ns_attached(mnt)) { 1223 for (struct mount *m = mnt; m; m = next_mnt(m, mnt)) 1224 mnt_add_to_ns(n, m); 1225 n->nr_mounts += n->pending_mounts; 1226 n->pending_mounts = 0; 1227 } 1228 1229 make_visible(mnt); 1230 touch_mnt_namespace(n); 1231 } 1232 1233 static void setup_mnt(struct mount *m, struct dentry *root) 1234 { 1235 struct super_block *s = root->d_sb; 1236 1237 atomic_inc(&s->s_active); 1238 m->mnt.mnt_sb = s; 1239 m->mnt.mnt_root = dget(root); 1240 m->mnt_mountpoint = m->mnt.mnt_root; 1241 m->mnt_parent = m; 1242 1243 guard(mount_locked_reader)(); 1244 mnt_add_instance(m, s); 1245 } 1246 1247 /** 1248 * vfs_create_mount - Create a mount for a configured superblock 1249 * @fc: The configuration context with the superblock attached 1250 * 1251 * Create a mount to an already configured superblock. If necessary, the 1252 * caller should invoke vfs_get_tree() before calling this. 1253 * 1254 * Note that this does not attach the mount to anything. 1255 */ 1256 struct vfsmount *vfs_create_mount(struct fs_context *fc) 1257 { 1258 struct mount *mnt; 1259 1260 if (!fc->root) 1261 return ERR_PTR(-EINVAL); 1262 1263 mnt = alloc_vfsmnt(fc->source); 1264 if (!mnt) 1265 return ERR_PTR(-ENOMEM); 1266 1267 if (fc->sb_flags & SB_KERNMOUNT) 1268 mnt->mnt.mnt_flags = MNT_INTERNAL; 1269 1270 setup_mnt(mnt, fc->root); 1271 1272 return &mnt->mnt; 1273 } 1274 EXPORT_SYMBOL(vfs_create_mount); 1275 1276 struct vfsmount *fc_mount(struct fs_context *fc) 1277 { 1278 int err = vfs_get_tree(fc); 1279 if (!err) { 1280 up_write(&fc->root->d_sb->s_umount); 1281 return vfs_create_mount(fc); 1282 } 1283 return ERR_PTR(err); 1284 } 1285 EXPORT_SYMBOL(fc_mount); 1286 1287 struct vfsmount *fc_mount_longterm(struct fs_context *fc) 1288 { 1289 struct vfsmount *mnt = fc_mount(fc); 1290 if (!IS_ERR(mnt)) 1291 real_mount(mnt)->mnt_ns = MNT_NS_INTERNAL; 1292 return mnt; 1293 } 1294 EXPORT_SYMBOL(fc_mount_longterm); 1295 1296 struct vfsmount *vfs_kern_mount(struct file_system_type *type, 1297 int flags, const char *name, 1298 void *data) 1299 { 1300 struct fs_context *fc; 1301 struct vfsmount *mnt; 1302 int ret = 0; 1303 1304 if (!type) 1305 return ERR_PTR(-EINVAL); 1306 1307 fc = fs_context_for_mount(type, flags); 1308 if (IS_ERR(fc)) 1309 return ERR_CAST(fc); 1310 1311 if (name) 1312 ret = vfs_parse_fs_string(fc, "source", name); 1313 if (!ret) 1314 ret = parse_monolithic_mount_data(fc, data); 1315 if (!ret) 1316 mnt = fc_mount(fc); 1317 else 1318 mnt = ERR_PTR(ret); 1319 1320 put_fs_context(fc); 1321 return mnt; 1322 } 1323 EXPORT_SYMBOL_GPL(vfs_kern_mount); 1324 1325 static struct mount *clone_mnt(struct mount *old, struct dentry *root, 1326 int flag) 1327 { 1328 struct mount *mnt; 1329 int err; 1330 1331 mnt = alloc_vfsmnt(old->mnt_devname); 1332 if (!mnt) 1333 return ERR_PTR(-ENOMEM); 1334 1335 mnt->mnt.mnt_flags = READ_ONCE(old->mnt.mnt_flags) & 1336 ~MNT_INTERNAL_FLAGS; 1337 mnt->mnt_t_flags = old->mnt_t_flags & T_UNBINDABLE; 1338 1339 if (flag & (CL_SLAVE | CL_PRIVATE)) 1340 mnt->mnt_group_id = 0; /* not a peer of original */ 1341 else 1342 mnt->mnt_group_id = old->mnt_group_id; 1343 1344 if ((flag & CL_MAKE_SHARED) && !mnt->mnt_group_id) { 1345 err = mnt_alloc_group_id(mnt); 1346 if (err) 1347 goto out_free; 1348 } 1349 1350 if (mnt->mnt_group_id) 1351 set_mnt_shared(mnt); 1352 1353 mnt->mnt.mnt_idmap = mnt_idmap_get(mnt_idmap(&old->mnt)); 1354 1355 setup_mnt(mnt, root); 1356 1357 if (flag & CL_PRIVATE) // we are done with it 1358 return mnt; 1359 1360 if (peers(mnt, old)) 1361 list_add(&mnt->mnt_share, &old->mnt_share); 1362 1363 if ((flag & CL_SLAVE) && old->mnt_group_id) { 1364 hlist_add_head(&mnt->mnt_slave, &old->mnt_slave_list); 1365 mnt->mnt_master = old; 1366 } else if (IS_MNT_SLAVE(old)) { 1367 hlist_add_behind(&mnt->mnt_slave, &old->mnt_slave); 1368 mnt->mnt_master = old->mnt_master; 1369 } 1370 return mnt; 1371 1372 out_free: 1373 mnt_free_id(mnt); 1374 free_vfsmnt(mnt); 1375 return ERR_PTR(err); 1376 } 1377 1378 static void cleanup_mnt(struct mount *mnt) 1379 { 1380 /* 1381 * The warning here probably indicates that somebody messed 1382 * up a mnt_want/drop_write() pair. If this happens, the 1383 * filesystem was probably unable to make r/w->r/o transitions. 1384 * The locking used to deal with mnt_count decrement provides barriers, 1385 * so mnt_get_writers() below is safe. 1386 */ 1387 WARN_ON(mnt_get_writers(mnt)); 1388 if (unlikely(mnt->mnt_pins.first)) 1389 mnt_pin_kill(mnt); 1390 fsnotify_vfsmount_delete(&mnt->mnt); 1391 dput(mnt->mnt.mnt_root); 1392 deactivate_super(mnt->mnt.mnt_sb); 1393 mnt_free_id(mnt); 1394 call_rcu(&mnt->mnt_rcu, delayed_free_vfsmnt); 1395 } 1396 1397 static void __cleanup_mnt(struct rcu_head *head) 1398 { 1399 cleanup_mnt(container_of(head, struct mount, mnt_rcu)); 1400 } 1401 1402 static LLIST_HEAD(delayed_mntput_list); 1403 static void delayed_mntput(struct work_struct *unused) 1404 { 1405 struct llist_node *node = llist_del_all(&delayed_mntput_list); 1406 struct mount *m, *t; 1407 1408 llist_for_each_entry_safe(m, t, node, mnt_llist) 1409 cleanup_mnt(m); 1410 } 1411 static DECLARE_DELAYED_WORK(delayed_mntput_work, delayed_mntput); 1412 1413 static void noinline mntput_no_expire_slowpath(struct mount *mnt) 1414 { 1415 struct mnt_cover *cover; 1416 struct hlist_node *n; 1417 LIST_HEAD(list); 1418 int count; 1419 1420 VFS_BUG_ON(mnt->mnt_ns); 1421 lock_mount_hash(); 1422 /* 1423 * make sure that if __legitimize_mnt() has not seen us grab 1424 * mount_lock, we'll see their refcount increment here. 1425 */ 1426 smp_mb(); 1427 mnt_dec_count(mnt); 1428 count = mnt_get_count(mnt); 1429 if (count != 0) { 1430 WARN_ON(count < 0); 1431 rcu_read_unlock(); 1432 unlock_mount_hash(); 1433 return; 1434 } 1435 if (unlikely(mnt->mnt.mnt_flags & MNT_DOOMED)) { 1436 rcu_read_unlock(); 1437 unlock_mount_hash(); 1438 return; 1439 } 1440 mnt->mnt.mnt_flags |= MNT_DOOMED; 1441 rcu_read_unlock(); 1442 1443 mnt_del_instance(mnt); 1444 if (unlikely(!list_empty(&mnt->mnt_expire))) 1445 list_del(&mnt->mnt_expire); 1446 1447 /* nothing stays attached to an unmounted mount */ 1448 VFS_WARN_ON_ONCE(!list_empty(&mnt->mnt_mounts)); 1449 hlist_for_each_entry_safe(cover, n, &mnt->mnt_covers, node) 1450 drop_cover(cover, &list); 1451 unlock_mount_hash(); 1452 shrink_dentry_list(&list); 1453 1454 if (likely(!(mnt->mnt.mnt_flags & MNT_INTERNAL))) { 1455 struct task_struct *task = current; 1456 if (likely(!(task->flags & PF_KTHREAD))) { 1457 init_task_work(&mnt->mnt_rcu, __cleanup_mnt); 1458 if (!task_work_add(task, &mnt->mnt_rcu, TWA_RESUME)) 1459 return; 1460 } 1461 if (llist_add(&mnt->mnt_llist, &delayed_mntput_list)) 1462 schedule_delayed_work(&delayed_mntput_work, 1); 1463 return; 1464 } 1465 cleanup_mnt(mnt); 1466 } 1467 1468 static void mntput_no_expire(struct mount *mnt) 1469 { 1470 rcu_read_lock(); 1471 if (likely(READ_ONCE(mnt->mnt_ns))) { 1472 /* 1473 * Since we don't do lock_mount_hash() here, 1474 * ->mnt_ns can change under us. However, if it's 1475 * non-NULL, then there's a reference that won't 1476 * be dropped until after an RCU delay done after 1477 * turning ->mnt_ns NULL. So if we observe it 1478 * non-NULL under rcu_read_lock(), the reference 1479 * we are dropping is not the final one. 1480 */ 1481 smp_wmb(); /* pairs with the smp_mb() in mnt_get_count() */ 1482 mnt_dec_count(mnt); 1483 rcu_read_unlock(); 1484 return; 1485 } 1486 mntput_no_expire_slowpath(mnt); 1487 } 1488 1489 void mntput(struct vfsmount *mnt) 1490 { 1491 if (mnt) { 1492 struct mount *m = real_mount(mnt); 1493 /* avoid cacheline pingpong */ 1494 if (unlikely(m->mnt_expiry_mark)) 1495 WRITE_ONCE(m->mnt_expiry_mark, 0); 1496 mntput_no_expire(m); 1497 } 1498 } 1499 EXPORT_SYMBOL(mntput); 1500 1501 struct vfsmount *mntget(struct vfsmount *mnt) 1502 { 1503 if (mnt) 1504 mnt_inc_count(real_mount(mnt)); 1505 return mnt; 1506 } 1507 EXPORT_SYMBOL(mntget); 1508 1509 /* 1510 * Make a mount point inaccessible to new lookups. 1511 * Because there may still be current users, the caller MUST WAIT 1512 * for an RCU grace period before destroying the mount point. 1513 */ 1514 void mnt_make_shortterm(struct vfsmount *mnt) 1515 { 1516 if (mnt) 1517 WRITE_ONCE(real_mount(mnt)->mnt_ns, NULL); 1518 } 1519 1520 /** 1521 * path_is_mountpoint() - Check if path is a mount in the current namespace. 1522 * @path: path to check 1523 * 1524 * d_mountpoint() can only be used reliably to establish if a dentry is 1525 * not mounted in any namespace and that common case is handled inline. 1526 * d_mountpoint() isn't aware of the possibility there may be multiple 1527 * mounts using a given dentry in a different namespace. This function 1528 * checks if the passed in path is a mountpoint rather than the dentry 1529 * alone. 1530 */ 1531 bool path_is_mountpoint(const struct path *path) 1532 { 1533 unsigned seq; 1534 bool res; 1535 1536 if (!d_mountpoint(path->dentry)) 1537 return false; 1538 1539 rcu_read_lock(); 1540 do { 1541 seq = read_seqbegin(&mount_lock); 1542 res = __path_is_mountpoint(path); 1543 } while (read_seqretry(&mount_lock, seq)); 1544 rcu_read_unlock(); 1545 1546 return res; 1547 } 1548 EXPORT_SYMBOL(path_is_mountpoint); 1549 1550 struct vfsmount *mnt_clone_internal(const struct path *path) 1551 { 1552 struct mount *p; 1553 p = clone_mnt(real_mount(path->mnt), path->dentry, CL_PRIVATE); 1554 if (IS_ERR(p)) 1555 return ERR_CAST(p); 1556 p->mnt.mnt_flags |= MNT_INTERNAL; 1557 return &p->mnt; 1558 } 1559 1560 /* 1561 * Returns the mount which either has the specified mnt_id, or has the next 1562 * smallest id afer the specified one. 1563 */ 1564 static struct mount *mnt_find_id_at(struct mnt_namespace *ns, u64 mnt_id) 1565 { 1566 struct rb_node *node = ns->mounts.rb_node; 1567 struct mount *ret = NULL; 1568 1569 while (node) { 1570 struct mount *m = node_to_mount(node); 1571 1572 if (mnt_id <= m->mnt_id_unique) { 1573 ret = node_to_mount(node); 1574 if (mnt_id == m->mnt_id_unique) 1575 break; 1576 node = node->rb_left; 1577 } else { 1578 node = node->rb_right; 1579 } 1580 } 1581 return ret; 1582 } 1583 1584 /* 1585 * Returns the mount which either has the specified mnt_id, or has the next 1586 * greater id before the specified one. 1587 */ 1588 static struct mount *mnt_find_id_at_reverse(struct mnt_namespace *ns, u64 mnt_id) 1589 { 1590 struct rb_node *node = ns->mounts.rb_node; 1591 struct mount *ret = NULL; 1592 1593 while (node) { 1594 struct mount *m = node_to_mount(node); 1595 1596 if (mnt_id >= m->mnt_id_unique) { 1597 ret = node_to_mount(node); 1598 if (mnt_id == m->mnt_id_unique) 1599 break; 1600 node = node->rb_right; 1601 } else { 1602 node = node->rb_left; 1603 } 1604 } 1605 return ret; 1606 } 1607 1608 #ifdef CONFIG_PROC_FS 1609 1610 /* iterator; we want it to have access to namespace_sem, thus here... */ 1611 static void *m_start(struct seq_file *m, loff_t *pos) 1612 { 1613 struct proc_mounts *p = m->private; 1614 struct mount *mnt; 1615 1616 down_read(&namespace_sem); 1617 1618 mnt = mnt_find_id_at(p->ns, *pos); 1619 if (mnt) 1620 *pos = mnt->mnt_id_unique; 1621 return mnt; 1622 } 1623 1624 static void *m_next(struct seq_file *m, void *v, loff_t *pos) 1625 { 1626 struct mount *mnt = v; 1627 struct rb_node *node = rb_next(&mnt->mnt_node); 1628 1629 if (node) { 1630 struct mount *next = node_to_mount(node); 1631 *pos = next->mnt_id_unique; 1632 return next; 1633 } 1634 1635 /* 1636 * No more mounts. Set pos past current mount's ID so that if 1637 * iteration restarts, mnt_find_id_at() returns NULL. 1638 */ 1639 *pos = mnt->mnt_id_unique + 1; 1640 return NULL; 1641 } 1642 1643 static void m_stop(struct seq_file *m, void *v) 1644 { 1645 up_read(&namespace_sem); 1646 } 1647 1648 static int m_show(struct seq_file *m, void *v) 1649 { 1650 struct proc_mounts *p = m->private; 1651 struct mount *r = v; 1652 return p->show(m, &r->mnt); 1653 } 1654 1655 const struct seq_operations mounts_op = { 1656 .start = m_start, 1657 .next = m_next, 1658 .stop = m_stop, 1659 .show = m_show, 1660 }; 1661 1662 #endif /* CONFIG_PROC_FS */ 1663 1664 /** 1665 * may_umount_tree - check if a mount tree is busy 1666 * @m: root of mount tree 1667 * 1668 * This is called to check if a tree of mounts has any 1669 * open files, pwds, chroots or sub mounts that are 1670 * busy. 1671 */ 1672 int may_umount_tree(struct vfsmount *m) 1673 { 1674 struct mount *mnt = real_mount(m); 1675 bool busy = false; 1676 1677 /* write lock needed for mnt_get_count */ 1678 lock_mount_hash(); 1679 for (struct mount *p = mnt; p; p = next_mnt(p, mnt)) { 1680 if (mnt_get_count(p) > (p == mnt ? 2 : 1)) { 1681 busy = true; 1682 break; 1683 } 1684 } 1685 unlock_mount_hash(); 1686 1687 return !busy; 1688 } 1689 1690 EXPORT_SYMBOL(may_umount_tree); 1691 1692 /** 1693 * may_umount - check if a mount point is busy 1694 * @mnt: root of mount 1695 * 1696 * This is called to check if a mount point has any 1697 * open files, pwds, chroots or sub mounts. If the 1698 * mount has sub mounts this will return busy 1699 * regardless of whether the sub mounts are busy. 1700 * 1701 * Doesn't take quota and stuff into account. IOW, in some cases it will 1702 * give false negatives. The main reason why it's here is that we need 1703 * a non-destructive way to look for easily umountable filesystems. 1704 */ 1705 int may_umount(struct vfsmount *mnt) 1706 { 1707 int ret = 1; 1708 down_read(&namespace_sem); 1709 lock_mount_hash(); 1710 if (propagate_mount_busy(real_mount(mnt), 2)) 1711 ret = 0; 1712 unlock_mount_hash(); 1713 up_read(&namespace_sem); 1714 return ret; 1715 } 1716 1717 EXPORT_SYMBOL(may_umount); 1718 1719 #ifdef CONFIG_FSNOTIFY 1720 static void mnt_notify(struct mount *p) 1721 { 1722 if (!p->prev_ns && p->mnt_ns) { 1723 fsnotify_mnt_attach(p->mnt_ns, &p->mnt); 1724 } else if (p->prev_ns && !p->mnt_ns) { 1725 fsnotify_mnt_detach(p->prev_ns, &p->mnt); 1726 } else if (p->prev_ns == p->mnt_ns) { 1727 fsnotify_mnt_move(p->mnt_ns, &p->mnt); 1728 } else { 1729 fsnotify_mnt_detach(p->prev_ns, &p->mnt); 1730 fsnotify_mnt_attach(p->mnt_ns, &p->mnt); 1731 } 1732 p->prev_ns = p->mnt_ns; 1733 } 1734 1735 static void notify_mnt_list(void) 1736 { 1737 struct mount *m, *tmp; 1738 /* 1739 * Notify about mounts that were added/reparented/detached/remain 1740 * connected after unmount. 1741 */ 1742 list_for_each_entry_safe(m, tmp, ¬ify_list, to_notify) { 1743 mnt_notify(m); 1744 list_del_init(&m->to_notify); 1745 } 1746 } 1747 1748 static bool need_notify_mnt_list(void) 1749 { 1750 return !list_empty(¬ify_list); 1751 } 1752 #else 1753 static void notify_mnt_list(void) 1754 { 1755 } 1756 1757 static bool need_notify_mnt_list(void) 1758 { 1759 return false; 1760 } 1761 #endif 1762 1763 static void free_mnt_ns(struct mnt_namespace *); 1764 static void namespace_unlock(void) 1765 { 1766 struct hlist_head head; 1767 struct hlist_node *p; 1768 struct mount *m; 1769 struct mnt_namespace *ns = emptied_ns; 1770 LIST_HEAD(list); 1771 1772 hlist_move_list(&unmounted, &head); 1773 list_splice_init(&ex_mountpoints, &list); 1774 emptied_ns = NULL; 1775 1776 if (need_notify_mnt_list()) { 1777 /* 1778 * No point blocking out concurrent readers while notifications 1779 * are sent. This will also allow statmount()/listmount() to run 1780 * concurrently. 1781 */ 1782 downgrade_write(&namespace_sem); 1783 notify_mnt_list(); 1784 up_read(&namespace_sem); 1785 } else { 1786 up_write(&namespace_sem); 1787 } 1788 if (unlikely(ns)) { 1789 /* Make sure we notice when we leak mounts. */ 1790 VFS_WARN_ON_ONCE(!mnt_ns_empty(ns)); 1791 free_mnt_ns(ns); 1792 } 1793 1794 shrink_dentry_list(&list); 1795 1796 if (likely(hlist_empty(&head))) 1797 return; 1798 1799 synchronize_rcu_expedited(); 1800 1801 hlist_for_each_entry_safe(m, p, &head, mnt_umount) { 1802 hlist_del(&m->mnt_umount); 1803 mntput(&m->mnt); 1804 } 1805 } 1806 1807 static inline void namespace_lock(void) 1808 { 1809 down_write(&namespace_sem); 1810 } 1811 1812 enum umount_tree_flags { 1813 UMOUNT_SYNC = 1, 1814 UMOUNT_PROPAGATE = 2, 1815 UMOUNT_COVER = 4, 1816 }; 1817 1818 /* Do we need to leave the mountpoint on the parent covered? */ 1819 static bool needs_cover(struct mount *mnt, enum umount_tree_flags how) 1820 { 1821 if (how & UMOUNT_SYNC) 1822 return false; 1823 if (!(mnt->mnt_parent->mnt.mnt_flags & MNT_UMOUNT)) 1824 return false; 1825 return (how & UMOUNT_COVER) || IS_MNT_LOCKED(mnt); 1826 } 1827 1828 /* 1829 * mount_lock must be held 1830 * namespace_sem must be held for write 1831 */ 1832 static void umount_tree(struct mount *mnt, enum umount_tree_flags how) 1833 { 1834 LIST_HEAD(tmp_list); 1835 struct mount *p; 1836 1837 if (how & UMOUNT_PROPAGATE) 1838 propagate_mount_unlock(mnt); 1839 1840 /* Gather the mounts to umount */ 1841 for (p = mnt; p; p = next_mnt(p, mnt)) { 1842 /* A mount is unmounted once. */ 1843 VFS_WARN_ON_ONCE(p->mnt.mnt_flags & MNT_UMOUNT); 1844 p->mnt.mnt_flags |= MNT_UMOUNT; 1845 if (mnt_ns_attached(p)) 1846 move_from_ns(p); 1847 list_add_tail(&p->mnt_list, &tmp_list); 1848 } 1849 1850 /* Hide the mounts from mnt_mounts */ 1851 list_for_each_entry(p, &tmp_list, mnt_list) { 1852 list_del_init(&p->mnt_child); 1853 } 1854 1855 /* Add propagated mounts to the tmp_list */ 1856 if (how & UMOUNT_PROPAGATE) 1857 propagate_umount(&tmp_list); 1858 1859 bulk_make_private(&tmp_list); 1860 1861 while (!list_empty(&tmp_list)) { 1862 struct mnt_namespace *ns; 1863 p = list_first_entry(&tmp_list, struct mount, mnt_list); 1864 list_del_init(&p->mnt_expire); 1865 list_del_init(&p->mnt_list); 1866 ns = p->mnt_ns; 1867 if (ns) { 1868 ns->nr_mounts--; 1869 __touch_mnt_namespace(ns); 1870 } 1871 WRITE_ONCE(p->mnt_ns, NULL); 1872 if (how & UMOUNT_SYNC) 1873 p->mnt.mnt_flags |= MNT_SYNC_UMOUNT; 1874 1875 if (mnt_has_parent(p)) { 1876 if (needs_cover(p, how)) 1877 leave_cover(p); 1878 umount_mnt(p); 1879 } 1880 hlist_add_head(&p->mnt_umount, &unmounted); 1881 1882 /* 1883 * At this point p->mnt_ns is NULL, notification will be queued 1884 * only if 1885 * 1886 * - p->prev_ns is non-NULL *and* 1887 * - p->prev_ns->n_fsnotify_marks is non-NULL 1888 * 1889 * This will preclude queuing the mount if this is a cleanup 1890 * after a failed copy_tree() or destruction of an anonymous 1891 * namespace, etc. 1892 */ 1893 mnt_notify_add(p); 1894 } 1895 } 1896 1897 static void shrink_submounts(struct mount *mnt); 1898 1899 static int do_umount_root(struct super_block *sb) 1900 { 1901 int ret = 0; 1902 1903 down_write(&sb->s_umount); 1904 if (!sb_rdonly(sb)) { 1905 struct fs_context *fc; 1906 1907 fc = fs_context_for_reconfigure(sb->s_root, SB_RDONLY, 1908 SB_RDONLY); 1909 if (IS_ERR(fc)) { 1910 ret = PTR_ERR(fc); 1911 } else { 1912 ret = parse_monolithic_mount_data(fc, NULL); 1913 if (!ret) 1914 ret = reconfigure_super(fc); 1915 put_fs_context(fc); 1916 } 1917 } 1918 up_write(&sb->s_umount); 1919 return ret; 1920 } 1921 1922 static int do_umount(struct mount *mnt, int flags) 1923 { 1924 struct super_block *sb = mnt->mnt.mnt_sb; 1925 int retval; 1926 1927 retval = security_sb_umount(&mnt->mnt, flags); 1928 if (retval) 1929 return retval; 1930 1931 /* 1932 * Allow userspace to request a mountpoint be expired rather than 1933 * unmounting unconditionally. Unmount only happens if: 1934 * (1) the mark is already set (the mark is cleared by mntput()) 1935 * (2) the usage count == 1 [parent vfsmount] + 1 [sys_umount] 1936 */ 1937 if (flags & MNT_EXPIRE) { 1938 if (&mnt->mnt == current->fs->root.mnt || 1939 flags & (MNT_FORCE | MNT_DETACH)) 1940 return -EINVAL; 1941 1942 /* 1943 * probably don't strictly need the lock here if we examined 1944 * all race cases, but it's a slowpath. 1945 */ 1946 lock_mount_hash(); 1947 if (!list_empty(&mnt->mnt_mounts) || mnt_get_count(mnt) != 2) { 1948 unlock_mount_hash(); 1949 return -EBUSY; 1950 } 1951 unlock_mount_hash(); 1952 1953 if (!xchg(&mnt->mnt_expiry_mark, 1)) 1954 return -EAGAIN; 1955 } 1956 1957 /* 1958 * If we may have to abort operations to get out of this 1959 * mount, and they will themselves hold resources we must 1960 * allow the fs to do things. In the Unix tradition of 1961 * 'Gee thats tricky lets do it in userspace' the umount_begin 1962 * might fail to complete on the first run through as other tasks 1963 * must return, and the like. Thats for the mount program to worry 1964 * about for the moment. 1965 */ 1966 1967 if (flags & MNT_FORCE && sb->s_op->umount_begin) { 1968 sb->s_op->umount_begin(sb); 1969 } 1970 1971 /* 1972 * No sense to grab the lock for this test, but test itself looks 1973 * somewhat bogus. Suggestions for better replacement? 1974 * Ho-hum... In principle, we might treat that as umount + switch 1975 * to rootfs. GC would eventually take care of the old vfsmount. 1976 * Actually it makes sense, especially if rootfs would contain a 1977 * /reboot - static binary that would close all descriptors and 1978 * call reboot(9). Then init(8) could umount root and exec /reboot. 1979 */ 1980 if (&mnt->mnt == current->fs->root.mnt && !(flags & MNT_DETACH)) { 1981 /* 1982 * Special case for "unmounting" root ... 1983 * we just try to remount it readonly. 1984 */ 1985 if (!ns_capable(sb->s_user_ns, CAP_SYS_ADMIN)) 1986 return -EPERM; 1987 return do_umount_root(sb); 1988 } 1989 1990 namespace_lock(); 1991 lock_mount_hash(); 1992 1993 /* Repeat the earlier racy checks, now that we are holding the locks */ 1994 retval = -EINVAL; 1995 if (!check_mnt(mnt)) 1996 goto out; 1997 1998 if (mnt->mnt.mnt_flags & MNT_LOCKED) 1999 goto out; 2000 2001 if (!mnt_has_parent(mnt)) /* not the absolute root */ 2002 goto out; 2003 2004 event++; 2005 if (flags & MNT_DETACH) { 2006 umount_tree(mnt, UMOUNT_PROPAGATE); 2007 retval = 0; 2008 } else { 2009 smp_mb(); // paired with __legitimize_mnt() 2010 shrink_submounts(mnt); 2011 retval = -EBUSY; 2012 if (!propagate_mount_busy(mnt, 2)) { 2013 umount_tree(mnt, UMOUNT_PROPAGATE|UMOUNT_SYNC); 2014 retval = 0; 2015 } 2016 } 2017 out: 2018 unlock_mount_hash(); 2019 namespace_unlock(); 2020 return retval; 2021 } 2022 2023 /* 2024 * __detach_mounts - lazily unmount all mounts on the specified dentry 2025 * 2026 * During unlink, rmdir, and d_drop it is possible to loose the path 2027 * to an existing mountpoint, and wind up leaking the mount. 2028 * detach_mounts allows lazily unmounting those mounts instead of 2029 * leaking them. 2030 * 2031 * The dentry is unhashed before the mounts go so that no lookup finds 2032 * what they covered. The caller removes it for good afterwards. 2033 * 2034 * The caller may hold dentry->d_inode->i_rwsem. 2035 */ 2036 void __detach_mounts(struct dentry *dentry) 2037 { 2038 struct pinned_mountpoint mp = {}; 2039 struct mnt_cover *cover; 2040 struct hlist_node *n; 2041 struct mount *mnt; 2042 2043 guard(namespace_excl)(); 2044 guard(mount_writer)(); 2045 2046 if (!lookup_mountpoint(dentry, &mp)) 2047 return; 2048 2049 /* the name goes first, what covered it goes second */ 2050 d_drop(dentry); 2051 event++; 2052 while (mp.node.next) { 2053 mnt = hlist_entry(mp.node.next, struct mount, mnt_mp_list); 2054 umount_tree(mnt, UMOUNT_COVER); 2055 } 2056 /* the dentry goes away, so do the covers left behind on it */ 2057 hlist_for_each_entry_safe(cover, n, &mp.mp->m_covers, pin) 2058 drop_cover(cover, &ex_mountpoints); 2059 unpin_mountpoint(&mp); 2060 } 2061 2062 /* 2063 * Is the caller allowed to modify his namespace? 2064 */ 2065 bool may_mount(void) 2066 { 2067 return ns_capable(current->nsproxy->mnt_ns->user_ns, CAP_SYS_ADMIN); 2068 } 2069 2070 static void warn_mandlock(void) 2071 { 2072 pr_warn_once("=======================================================\n" 2073 "WARNING: The mand mount option has been deprecated and\n" 2074 " and is ignored by this kernel. Remove the mand\n" 2075 " option from the mount to silence this warning.\n" 2076 "=======================================================\n"); 2077 } 2078 2079 static int can_umount(const struct path *path, int flags) 2080 { 2081 struct mount *mnt = real_mount(path->mnt); 2082 struct super_block *sb = path->dentry->d_sb; 2083 2084 if (!may_mount()) 2085 return -EPERM; 2086 if (!path_mounted(path)) 2087 return -EINVAL; 2088 if (!check_mnt(mnt)) 2089 return -EINVAL; 2090 if (mnt->mnt.mnt_flags & MNT_LOCKED) /* Check optimistically */ 2091 return -EINVAL; 2092 if (flags & MNT_FORCE && !ns_capable(sb->s_user_ns, CAP_SYS_ADMIN)) 2093 return -EPERM; 2094 return 0; 2095 } 2096 2097 // caller is responsible for flags being sane 2098 int path_umount(const struct path *path, int flags) 2099 { 2100 struct mount *mnt = real_mount(path->mnt); 2101 int ret; 2102 2103 ret = can_umount(path, flags); 2104 if (!ret) 2105 ret = do_umount(mnt, flags); 2106 2107 /* we mustn't call path_put() as that would clear mnt_expiry_mark */ 2108 dput(path->dentry); 2109 mntput_no_expire(mnt); 2110 return ret; 2111 } 2112 2113 static int ksys_umount(char __user *name, int flags) 2114 { 2115 int lookup_flags = LOOKUP_MOUNTPOINT; 2116 struct path path; 2117 int ret; 2118 2119 // basic validity checks done first 2120 if (flags & ~(MNT_FORCE | MNT_DETACH | MNT_EXPIRE | UMOUNT_NOFOLLOW)) 2121 return -EINVAL; 2122 2123 if (!(flags & UMOUNT_NOFOLLOW)) 2124 lookup_flags |= LOOKUP_FOLLOW; 2125 ret = user_path_at(AT_FDCWD, name, lookup_flags, &path); 2126 if (ret) 2127 return ret; 2128 return path_umount(&path, flags); 2129 } 2130 2131 SYSCALL_DEFINE2(umount, char __user *, name, int, flags) 2132 { 2133 return ksys_umount(name, flags); 2134 } 2135 2136 #ifdef __ARCH_WANT_SYS_OLDUMOUNT 2137 2138 /* 2139 * The 2.0 compatible umount. No flags. 2140 */ 2141 SYSCALL_DEFINE1(oldumount, char __user *, name) 2142 { 2143 return ksys_umount(name, 0); 2144 } 2145 2146 #endif 2147 2148 static bool is_mnt_ns_file(struct dentry *dentry) 2149 { 2150 struct ns_common *ns; 2151 2152 /* Is this a proxy for a mount namespace? */ 2153 if (dentry->d_op != &ns_dentry_operations) 2154 return false; 2155 2156 ns = d_inode(dentry)->i_private; 2157 2158 return ns->ops == &mntns_operations; 2159 } 2160 2161 struct ns_common *from_mnt_ns(struct mnt_namespace *mnt) 2162 { 2163 return &mnt->ns; 2164 } 2165 2166 struct mnt_namespace *get_sequential_mnt_ns(struct mnt_namespace *mntns, bool previous) 2167 { 2168 struct ns_common *ns; 2169 2170 guard(rcu)(); 2171 2172 for (;;) { 2173 ns = ns_tree_adjoined_rcu(mntns, previous); 2174 if (IS_ERR(ns)) 2175 return ERR_CAST(ns); 2176 2177 mntns = to_mnt_ns(ns); 2178 2179 /* 2180 * The last passive reference count is put with RCU 2181 * delay so accessing the mount namespace is not just 2182 * safe but all relevant members are still valid. 2183 */ 2184 if (!ns_capable_noaudit(mntns->user_ns, CAP_SYS_ADMIN)) 2185 continue; 2186 2187 /* 2188 * We need an active reference count as we're persisting 2189 * the mount namespace and it might already be on its 2190 * deathbed. 2191 */ 2192 if (!ns_ref_get(mntns)) 2193 continue; 2194 2195 return mntns; 2196 } 2197 } 2198 2199 struct mnt_namespace *mnt_ns_from_dentry(struct dentry *dentry) 2200 { 2201 if (!is_mnt_ns_file(dentry)) 2202 return NULL; 2203 2204 return to_mnt_ns(get_proc_ns(dentry->d_inode)); 2205 } 2206 2207 static bool mnt_ns_loop(struct dentry *dentry) 2208 { 2209 /* Could bind mounting the mount namespace inode cause a 2210 * mount namespace loop? 2211 */ 2212 struct mnt_namespace *mnt_ns = mnt_ns_from_dentry(dentry); 2213 2214 if (!mnt_ns) 2215 return false; 2216 2217 return current->nsproxy->mnt_ns->ns.ns_id >= mnt_ns->ns.ns_id; 2218 } 2219 2220 struct mount *copy_tree(struct mount *src_root, struct dentry *dentry, 2221 int flag) 2222 { 2223 struct mount *res, *src_parent, *src_root_child, *src_mnt, 2224 *dst_parent, *dst_mnt; 2225 2226 if (!(flag & CL_COPY_UNBINDABLE) && IS_MNT_UNBINDABLE(src_root)) 2227 return ERR_PTR(-EINVAL); 2228 2229 if (!(flag & CL_COPY_MNT_NS_FILE) && is_mnt_ns_file(dentry)) 2230 return ERR_PTR(-EINVAL); 2231 2232 res = dst_mnt = clone_mnt(src_root, dentry, flag); 2233 if (IS_ERR(dst_mnt)) 2234 return dst_mnt; 2235 2236 src_parent = src_root; 2237 2238 list_for_each_entry(src_root_child, &src_root->mnt_mounts, mnt_child) { 2239 if (!is_subdir(src_root_child->mnt_mountpoint, dentry)) 2240 continue; 2241 2242 for (src_mnt = src_root_child; src_mnt; 2243 src_mnt = next_mnt(src_mnt, src_root_child)) { 2244 if (!(flag & CL_COPY_UNBINDABLE) && 2245 IS_MNT_UNBINDABLE(src_mnt)) { 2246 if (src_mnt->mnt.mnt_flags & MNT_LOCKED) { 2247 /* Both unbindable and locked. */ 2248 dst_mnt = ERR_PTR(-EPERM); 2249 goto out; 2250 } else { 2251 src_mnt = skip_mnt_tree(src_mnt); 2252 continue; 2253 } 2254 } 2255 if (!(flag & CL_COPY_MNT_NS_FILE) && 2256 is_mnt_ns_file(src_mnt->mnt.mnt_root)) { 2257 src_mnt = skip_mnt_tree(src_mnt); 2258 continue; 2259 } 2260 while (src_parent != src_mnt->mnt_parent) { 2261 src_parent = src_parent->mnt_parent; 2262 dst_mnt = dst_mnt->mnt_parent; 2263 } 2264 2265 src_parent = src_mnt; 2266 dst_parent = dst_mnt; 2267 dst_mnt = clone_mnt(src_mnt, src_mnt->mnt.mnt_root, flag); 2268 if (IS_ERR(dst_mnt)) 2269 goto out; 2270 lock_mount_hash(); 2271 if (src_mnt->mnt.mnt_flags & MNT_LOCKED) 2272 dst_mnt->mnt.mnt_flags |= MNT_LOCKED; 2273 if (unlikely(flag & CL_EXPIRE)) { 2274 /* stick the duplicate mount on the same expiry 2275 * list as the original if that was on one */ 2276 if (!list_empty(&src_mnt->mnt_expire)) 2277 list_add(&dst_mnt->mnt_expire, 2278 &src_mnt->mnt_expire); 2279 } 2280 attach_mnt(dst_mnt, dst_parent, src_parent->mnt_mp); 2281 unlock_mount_hash(); 2282 } 2283 } 2284 return res; 2285 2286 out: 2287 if (res) { 2288 lock_mount_hash(); 2289 umount_tree(res, UMOUNT_SYNC); 2290 unlock_mount_hash(); 2291 } 2292 return dst_mnt; 2293 } 2294 2295 static inline bool extend_array(struct path **res, struct path **to_free, 2296 unsigned n, unsigned *count, unsigned new_count) 2297 { 2298 struct path *p; 2299 2300 if (likely(n < *count)) 2301 return true; 2302 p = kmalloc_objs(struct path, new_count); 2303 if (p && *count) 2304 memcpy(p, *res, *count * sizeof(struct path)); 2305 *count = new_count; 2306 kfree(*to_free); 2307 *to_free = *res = p; 2308 return p; 2309 } 2310 2311 const struct path *collect_paths(const struct path *path, 2312 struct path *prealloc, unsigned count) 2313 { 2314 struct mount *root = real_mount(path->mnt); 2315 struct mount *child; 2316 struct path *res = prealloc, *to_free = NULL; 2317 unsigned n = 0; 2318 2319 guard(namespace_shared)(); 2320 2321 if (!check_mnt(root)) 2322 return ERR_PTR(-EINVAL); 2323 if (!extend_array(&res, &to_free, 0, &count, 32)) 2324 return ERR_PTR(-ENOMEM); 2325 res[n++] = *path; 2326 list_for_each_entry(child, &root->mnt_mounts, mnt_child) { 2327 if (!is_subdir(child->mnt_mountpoint, path->dentry)) 2328 continue; 2329 for (struct mount *m = child; m; m = next_mnt(m, child)) { 2330 if (!extend_array(&res, &to_free, n, &count, 2 * count)) 2331 return ERR_PTR(-ENOMEM); 2332 res[n].mnt = &m->mnt; 2333 res[n].dentry = m->mnt.mnt_root; 2334 n++; 2335 } 2336 } 2337 if (!extend_array(&res, &to_free, n, &count, count + 1)) 2338 return ERR_PTR(-ENOMEM); 2339 memset(res + n, 0, (count - n) * sizeof(struct path)); 2340 for (struct path *p = res; p->mnt; p++) 2341 path_get(p); 2342 return res; 2343 } 2344 2345 void drop_collected_paths(const struct path *paths, const struct path *prealloc) 2346 { 2347 for (const struct path *p = paths; p->mnt; p++) 2348 path_put(p); 2349 if (paths != prealloc) 2350 kfree(paths); 2351 } 2352 2353 static struct mnt_namespace *alloc_mnt_ns(struct user_namespace *, bool); 2354 2355 /* Consumes the caller's reference to @mnt. */ 2356 void dissolve_on_fput(struct vfsmount *mnt) 2357 { 2358 struct vfsmount *p __free(mntput) = mnt; 2359 struct mount *m = real_mount(p); 2360 2361 /* 2362 * m used to be the root of anon namespace; if it still is one, 2363 * we need to dissolve the mount tree and free that namespace. 2364 * Let's try to avoid taking namespace_sem if we can determine 2365 * that there's nothing to do without it - rcu_read_lock() is 2366 * enough to make anon_ns_root() memory-safe and once m has 2367 * left its namespace, it's no longer our concern, since it will 2368 * never become a root of anon ns again. 2369 */ 2370 2371 scoped_guard(rcu) { 2372 if (!anon_ns_root(m)) 2373 return; 2374 } 2375 2376 scoped_guard(namespace_excl) { 2377 if (!anon_ns_root(m)) 2378 return; 2379 2380 emptied_ns = m->mnt_ns; 2381 lock_mount_hash(); 2382 umount_tree(m, UMOUNT_COVER); 2383 unlock_mount_hash(); 2384 mntput(no_free_ptr(p)); 2385 } 2386 } 2387 2388 /* locks: namespace_shared && pinned(mnt) || mount_locked_reader */ 2389 bool has_locked_children(struct mount *mnt, struct dentry *dentry) 2390 { 2391 struct mount *child; 2392 2393 list_for_each_entry(child, &mnt->mnt_mounts, mnt_child) { 2394 if (!is_subdir(child->mnt_mountpoint, dentry)) 2395 continue; 2396 2397 if (child->mnt.mnt_flags & MNT_LOCKED) 2398 return true; 2399 } 2400 return false; 2401 } 2402 2403 /* locks: namespace_shared && pinned(mnt) || mount_locked_reader */ 2404 static bool __has_children(struct mount *mnt, struct dentry *dentry) 2405 { 2406 struct mount *child; 2407 2408 list_for_each_entry(child, &mnt->mnt_mounts, mnt_child) { 2409 if (is_subdir(child->mnt_mountpoint, dentry)) 2410 return true; 2411 } 2412 return false; 2413 } 2414 2415 /* 2416 * Check that there aren't references to earlier/same mount namespaces in the 2417 * specified subtree. Such references can act as pins for mount namespaces 2418 * that aren't checked by the mount-cycle checking code, thereby allowing 2419 * cycles to be made. 2420 * 2421 * locks: mount_locked_reader || namespace_shared && pinned(subtree) 2422 */ 2423 static bool check_for_nsfs_mounts(struct mount *subtree) 2424 { 2425 for (struct mount *p = subtree; p; p = next_mnt(p, subtree)) 2426 if (mnt_ns_loop(p->mnt.mnt_root)) 2427 return false; 2428 return true; 2429 } 2430 2431 /** 2432 * clone_private_mount - create a private clone of a path 2433 * @path: path to clone 2434 * 2435 * This creates a new vfsmount, which will be the clone of @path. The new mount 2436 * will not be attached anywhere in the namespace and will be private (i.e. 2437 * changes to the originating mount won't be propagated into this). 2438 * 2439 * This assumes caller has called or done the equivalent of may_mount(). 2440 * 2441 * Release with mntput(). 2442 */ 2443 struct vfsmount *clone_private_mount(const struct path *path) 2444 { 2445 struct mount *old_mnt = real_mount(path->mnt); 2446 struct mount *new_mnt; 2447 2448 guard(namespace_shared)(); 2449 2450 if (IS_MNT_UNBINDABLE(old_mnt)) 2451 return ERR_PTR(-EINVAL); 2452 2453 /* 2454 * Make sure the source mount is acceptable. 2455 * Anything mounted in our mount namespace is allowed. 2456 * Otherwise, it must be the root of an anonymous mount 2457 * namespace, and we need to make sure no namespace 2458 * loops get created. 2459 */ 2460 if (!check_mnt(old_mnt)) { 2461 if (!anon_ns_root(old_mnt)) 2462 return ERR_PTR(-EINVAL); 2463 2464 if (!check_for_nsfs_mounts(old_mnt)) 2465 return ERR_PTR(-EINVAL); 2466 } 2467 2468 if (!ns_capable(old_mnt->mnt_ns->user_ns, CAP_SYS_ADMIN)) 2469 return ERR_PTR(-EPERM); 2470 2471 if (has_locked_children(old_mnt, path->dentry)) 2472 return ERR_PTR(-EINVAL); 2473 2474 new_mnt = clone_mnt(old_mnt, path->dentry, CL_PRIVATE); 2475 if (IS_ERR(new_mnt)) 2476 return ERR_PTR(-EINVAL); 2477 2478 /* Longterm mount to be removed by kern_unmount*() */ 2479 new_mnt->mnt_ns = MNT_NS_INTERNAL; 2480 return &new_mnt->mnt; 2481 } 2482 EXPORT_SYMBOL_GPL(clone_private_mount); 2483 2484 static void lock_mnt_tree(struct mount *mnt) 2485 { 2486 struct mount *p; 2487 2488 for (p = mnt; p; p = next_mnt(p, mnt)) { 2489 int flags = p->mnt.mnt_flags; 2490 /* Don't allow unprivileged users to change mount flags */ 2491 flags |= MNT_LOCK_ATIME; 2492 2493 if (flags & MNT_READONLY) 2494 flags |= MNT_LOCK_READONLY; 2495 2496 if (flags & MNT_NODEV) 2497 flags |= MNT_LOCK_NODEV; 2498 2499 if (flags & MNT_NOSUID) 2500 flags |= MNT_LOCK_NOSUID; 2501 2502 if (flags & MNT_NOEXEC) 2503 flags |= MNT_LOCK_NOEXEC; 2504 /* Don't allow unprivileged users to reveal what is under a mount */ 2505 if (list_empty(&p->mnt_expire) && p != mnt) 2506 flags |= MNT_LOCKED; 2507 p->mnt.mnt_flags = flags; 2508 } 2509 } 2510 2511 static void cleanup_group_ids(struct mount *mnt, struct mount *end) 2512 { 2513 struct mount *p; 2514 2515 for (p = mnt; p != end; p = next_mnt(p, mnt)) { 2516 if (p->mnt_group_id && !IS_MNT_SHARED(p)) 2517 mnt_release_group_id(p); 2518 } 2519 } 2520 2521 static int invent_group_ids(struct mount *mnt, bool recurse) 2522 { 2523 struct mount *p; 2524 2525 for (p = mnt; p; p = recurse ? next_mnt(p, mnt) : NULL) { 2526 if (!p->mnt_group_id) { 2527 int err = mnt_alloc_group_id(p); 2528 if (err) { 2529 cleanup_group_ids(mnt, p); 2530 return err; 2531 } 2532 } 2533 } 2534 2535 return 0; 2536 } 2537 2538 int count_mounts(struct mnt_namespace *ns, struct mount *mnt) 2539 { 2540 unsigned int max = READ_ONCE(sysctl_mount_max); 2541 unsigned int mounts = 0; 2542 struct mount *p; 2543 2544 if (ns->nr_mounts >= max) 2545 return -ENOSPC; 2546 max -= ns->nr_mounts; 2547 if (ns->pending_mounts >= max) 2548 return -ENOSPC; 2549 max -= ns->pending_mounts; 2550 2551 for (p = mnt; p; p = next_mnt(p, mnt)) 2552 mounts++; 2553 2554 if (mounts > max) 2555 return -ENOSPC; 2556 2557 ns->pending_mounts += mounts; 2558 return 0; 2559 } 2560 2561 enum mnt_tree_flags_t { 2562 MNT_TREE_BENEATH = BIT(0), 2563 MNT_TREE_PROPAGATION = BIT(1), 2564 }; 2565 2566 /** 2567 * attach_recursive_mnt - attach a source mount tree 2568 * @source_mnt: mount tree to be attached 2569 * @dest: the context for mounting at the place where the tree should go 2570 * 2571 * NOTE: in the table below explains the semantics when a source mount 2572 * of a given type is attached to a destination mount of a given type. 2573 * --------------------------------------------------------------------------- 2574 * | BIND MOUNT OPERATION | 2575 * |************************************************************************** 2576 * | source-->| shared | private | slave | unbindable | 2577 * | dest | | | | | 2578 * | | | | | | | 2579 * | v | | | | | 2580 * |************************************************************************** 2581 * | shared | shared (++) | shared (+) | shared(+++)| invalid | 2582 * | | | | | | 2583 * |non-shared| shared (+) | private | slave (*) | invalid | 2584 * *************************************************************************** 2585 * A bind operation clones the source mount and mounts the clone on the 2586 * destination mount. 2587 * 2588 * (++) the cloned mount is propagated to all the mounts in the propagation 2589 * tree of the destination mount and the cloned mount is added to 2590 * the peer group of the source mount. 2591 * (+) the cloned mount is created under the destination mount and is marked 2592 * as shared. The cloned mount is added to the peer group of the source 2593 * mount. 2594 * (+++) the mount is propagated to all the mounts in the propagation tree 2595 * of the destination mount and the cloned mount is made slave 2596 * of the same master as that of the source mount. The cloned mount 2597 * is marked as 'shared and slave'. 2598 * (*) the cloned mount is made a slave of the same master as that of the 2599 * source mount. 2600 * 2601 * --------------------------------------------------------------------------- 2602 * | MOVE MOUNT OPERATION | 2603 * |************************************************************************** 2604 * | source-->| shared | private | slave | unbindable | 2605 * | dest | | | | | 2606 * | | | | | | | 2607 * | v | | | | | 2608 * |************************************************************************** 2609 * | shared | shared (+) | shared (+) | shared(+++) | invalid | 2610 * | | | | | | 2611 * |non-shared| shared (+*) | private | slave (*) | unbindable | 2612 * *************************************************************************** 2613 * 2614 * (+) the mount is moved to the destination. And is then propagated to 2615 * all the mounts in the propagation tree of the destination mount. 2616 * (+*) the mount is moved to the destination. 2617 * (+++) the mount is moved to the destination and is then propagated to 2618 * all the mounts belonging to the destination mount's propagation tree. 2619 * the mount is marked as 'shared and slave'. 2620 * (*) the mount continues to be a slave at the new location. 2621 * 2622 * if the source mount is a tree, the operations explained above is 2623 * applied to each mount in the tree. 2624 * Must be called without spinlocks held, since this function can sleep 2625 * in allocations. 2626 * 2627 * Context: The function expects namespace_lock() to be held. 2628 * Return: If @source_mnt was successfully attached 0 is returned. 2629 * Otherwise a negative error code is returned. 2630 */ 2631 static int attach_recursive_mnt(struct mount *source_mnt, 2632 const struct pinned_mountpoint *dest) 2633 { 2634 struct mount *dest_mnt = dest->parent; 2635 struct mountpoint *dest_mp = dest->mp; 2636 HLIST_HEAD(tree_list); 2637 struct mnt_namespace *ns = dest_mnt->mnt_ns; 2638 struct user_namespace *user_ns = ns->user_ns; 2639 struct pinned_mountpoint root = {}; 2640 struct mountpoint *shorter = NULL; 2641 struct mount *child, *p; 2642 struct mount *top; 2643 struct hlist_node *n; 2644 int err = 0; 2645 bool moving = mnt_has_parent(source_mnt); 2646 2647 /* 2648 * A caller in an unprivileged mount namespaces may trigger an 2649 * automount and propagate locked mounts into privileged mount 2650 * namespaces. Take ownership from the target mount namespace. 2651 * It's equivalent for everything but the automount case. 2652 * 2653 * Detached trees in anonymous mount namespaces by be handed 2654 * over via SCM_RIGHTS or inherited in other ways on purpose 2655 * the attaching task's mount namespace is authoritative, not 2656 * the creator of the detached tree. 2657 */ 2658 if (is_anon_ns(ns)) 2659 user_ns = current->nsproxy->mnt_ns->user_ns; 2660 else 2661 user_ns = ns->user_ns; 2662 2663 /* 2664 * Preallocate a mountpoint in case the new mounts need to be 2665 * mounted beneath mounts on the same mountpoint. 2666 */ 2667 for (top = source_mnt; ; top = top->overmount) { 2668 if (!shorter && is_mnt_ns_file(top->mnt.mnt_root)) 2669 shorter = top->mnt_mp; 2670 if (likely(!top->overmount)) 2671 break; 2672 } 2673 err = get_mountpoint(top->mnt.mnt_root, &root); 2674 if (err) 2675 return err; 2676 2677 /* Is there space to add these mounts to the mount namespace? */ 2678 if (!moving) { 2679 err = count_mounts(ns, source_mnt); 2680 if (err) 2681 goto out; 2682 } 2683 2684 if (IS_MNT_SHARED(dest_mnt)) { 2685 err = invent_group_ids(source_mnt, true); 2686 if (err) 2687 goto out; 2688 err = propagate_mnt(dest_mnt, dest_mp, source_mnt, &tree_list); 2689 } 2690 lock_mount_hash(); 2691 if (err) 2692 goto out_cleanup_ids; 2693 2694 if (IS_MNT_SHARED(dest_mnt)) { 2695 for (p = source_mnt; p; p = next_mnt(p, source_mnt)) 2696 set_mnt_shared(p); 2697 } 2698 2699 if (moving) { 2700 umount_mnt(source_mnt); 2701 mnt_notify_add(source_mnt); 2702 /* if the mount is moved, it should no longer be expired 2703 * automatically */ 2704 list_del_init(&source_mnt->mnt_expire); 2705 } else { 2706 if (source_mnt->mnt_ns) { 2707 /* move from anon - the caller will destroy */ 2708 emptied_ns = source_mnt->mnt_ns; 2709 for (p = source_mnt; p; p = next_mnt(p, source_mnt)) 2710 move_from_ns(p); 2711 } 2712 } 2713 2714 mnt_set_mountpoint(dest_mnt, dest_mp, source_mnt); 2715 /* 2716 * Now the original copy is in the same state as the secondaries - 2717 * its root attached to mountpoint, but not hashed and all mounts 2718 * in it are either in our namespace or in no namespace at all. 2719 * Add the original to the list of copies and deal with the 2720 * rest of work for all of them uniformly. 2721 */ 2722 hlist_add_head(&source_mnt->mnt_hash, &tree_list); 2723 2724 hlist_for_each_entry_safe(child, n, &tree_list, mnt_hash) { 2725 struct mount *q; 2726 hlist_del_init(&child->mnt_hash); 2727 /* Notice when we are propagating across user namespaces */ 2728 if (child->mnt_parent->mnt_ns->user_ns != user_ns) 2729 lock_mnt_tree(child); 2730 q = __lookup_mnt(&child->mnt_parent->mnt, 2731 child->mnt_mountpoint); 2732 commit_tree(child); 2733 if (q) { 2734 struct mount *r = topmost_overmount(child); 2735 struct mountpoint *mp = root.mp; 2736 2737 if (unlikely(shorter) && child != source_mnt) 2738 mp = shorter; 2739 /* 2740 * If @q was locked it was meant to hide 2741 * whatever was under it. Let @child take over 2742 * that job and lock it. If @child is the mount 2743 * the caller placed we can then unlock @q: 2744 * nothing another namespace does removes it 2745 * again. A propagated copy goes away when the 2746 * mounter of the original unmounts it, so @q 2747 * keeps its lock. 2748 */ 2749 if (IS_MNT_LOCKED(q)) { 2750 child->mnt.mnt_flags |= MNT_LOCKED; 2751 if (child == source_mnt) 2752 q->mnt.mnt_flags &= ~MNT_LOCKED; 2753 } 2754 mnt_change_mountpoint(r, mp, q); 2755 } 2756 } 2757 unpin_mountpoint(&root); 2758 unlock_mount_hash(); 2759 2760 return 0; 2761 2762 out_cleanup_ids: 2763 while (!hlist_empty(&tree_list)) { 2764 child = hlist_entry(tree_list.first, struct mount, mnt_hash); 2765 child->mnt_parent->mnt_ns->pending_mounts = 0; 2766 umount_tree(child, UMOUNT_SYNC); 2767 } 2768 unlock_mount_hash(); 2769 cleanup_group_ids(source_mnt, NULL); 2770 out: 2771 ns->pending_mounts = 0; 2772 2773 read_seqlock_excl(&mount_lock); 2774 unpin_mountpoint(&root); 2775 read_sequnlock_excl(&mount_lock); 2776 2777 return err; 2778 } 2779 2780 static inline struct mount *where_to_mount(const struct path *path, 2781 struct dentry **dentry, 2782 bool beneath) 2783 { 2784 struct mount *m; 2785 2786 if (unlikely(beneath)) { 2787 m = topmost_overmount(real_mount(path->mnt)); 2788 *dentry = m->mnt_mountpoint; 2789 return m->mnt_parent; 2790 } 2791 m = __lookup_mnt(path->mnt, path->dentry); 2792 if (unlikely(m)) { 2793 m = topmost_overmount(m); 2794 *dentry = m->mnt.mnt_root; 2795 return m; 2796 } 2797 *dentry = path->dentry; 2798 return real_mount(path->mnt); 2799 } 2800 2801 /** 2802 * do_lock_mount - acquire environment for mounting 2803 * @path: target path 2804 * @res: context to set up 2805 * @beneath: whether the intention is to mount beneath @path 2806 * 2807 * To mount something at given location, we need 2808 * namespace_sem locked exclusive 2809 * inode of dentry we are mounting on locked exclusive 2810 * struct mountpoint for that dentry 2811 * struct mount we are mounting on 2812 * 2813 * Results are stored in caller-supplied context (pinned_mountpoint); 2814 * on success we have res->parent and res->mp pointing to parent and 2815 * mountpoint respectively and res->node inserted into the ->m_list 2816 * of the mountpoint, making sure the mountpoint won't disappear. 2817 * On failure we have res->parent set to ERR_PTR(-E...), res->mp 2818 * left NULL, res->node - empty. 2819 * In case of success do_lock_mount returns with locks acquired (in 2820 * proper order - inode lock nests outside of namespace_sem). 2821 * 2822 * Request to mount on overmounted location is treated as "mount on 2823 * top of whatever's overmounting it"; request to mount beneath 2824 * a location - "mount immediately beneath the topmost mount at that 2825 * place". 2826 * 2827 * In all cases the location must not have been unmounted and the 2828 * chosen mountpoint must be allowed to be mounted on. For "beneath" 2829 * case we also require the location to be at the root of a mount 2830 * that has something mounted on top of it (i.e. has an overmount). 2831 */ 2832 static void do_lock_mount(const struct path *path, 2833 struct pinned_mountpoint *res, 2834 bool beneath) 2835 { 2836 int err; 2837 2838 if (unlikely(beneath) && !path_mounted(path)) { 2839 res->parent = ERR_PTR(-EINVAL); 2840 return; 2841 } 2842 2843 do { 2844 struct dentry *dentry, *d; 2845 struct mount *m, *n; 2846 2847 scoped_guard(mount_locked_reader) { 2848 m = where_to_mount(path, &dentry, beneath); 2849 /* sticky, so it takes no locks to refuse it */ 2850 if (unlikely(cant_mount(dentry))) { 2851 res->parent = ERR_PTR(-ENOENT); 2852 return; 2853 } 2854 if (&m->mnt != path->mnt) { 2855 mntget(&m->mnt); 2856 dget(dentry); 2857 } 2858 } 2859 2860 inode_lock(dentry->d_inode); 2861 namespace_lock(); 2862 2863 // check if the chain of mounts (if any) has changed. 2864 scoped_guard(mount_locked_reader) 2865 n = where_to_mount(path, &d, beneath); 2866 2867 if (unlikely(n != m || dentry != d)) 2868 err = -EAGAIN; // something moved, retry 2869 else if (unlikely(cant_mount(dentry) || !is_mounted(path->mnt))) 2870 err = -ENOENT; // not to be mounted on 2871 else if (beneath && &m->mnt == path->mnt && !m->overmount) 2872 err = -EINVAL; 2873 else 2874 err = get_mountpoint(dentry, res); 2875 2876 if (unlikely(err)) { 2877 res->parent = ERR_PTR(err); 2878 namespace_unlock(); 2879 inode_unlock(dentry->d_inode); 2880 } else { 2881 res->parent = m; 2882 } 2883 /* 2884 * Drop the temporary references. This is subtle - on success 2885 * we are doing that under namespace_sem, which would normally 2886 * be forbidden. However, in that case we are guaranteed that 2887 * refcounts won't reach zero, since we know that path->mnt 2888 * is mounted and thus all mounts reachable from it are pinned 2889 * and stable, along with their mountpoints and roots. 2890 */ 2891 if (&m->mnt != path->mnt) { 2892 dput(dentry); 2893 mntput(&m->mnt); 2894 } 2895 } while (err == -EAGAIN); 2896 } 2897 2898 static void __unlock_mount(struct pinned_mountpoint *m) 2899 { 2900 inode_unlock(m->mp->m_dentry->d_inode); 2901 read_seqlock_excl(&mount_lock); 2902 unpin_mountpoint(m); 2903 read_sequnlock_excl(&mount_lock); 2904 namespace_unlock(); 2905 } 2906 2907 static inline void unlock_mount(struct pinned_mountpoint *m) 2908 { 2909 if (!IS_ERR(m->parent)) 2910 __unlock_mount(m); 2911 } 2912 2913 static void lock_mount_exact(const struct path *path, 2914 struct pinned_mountpoint *mp, bool copy_mount, 2915 unsigned int copy_flags); 2916 2917 #define LOCK_MOUNT_MAYBE_BENEATH(mp, path, beneath) \ 2918 struct pinned_mountpoint mp __cleanup(unlock_mount) = {}; \ 2919 do_lock_mount((path), &mp, (beneath)) 2920 #define LOCK_MOUNT(mp, path) LOCK_MOUNT_MAYBE_BENEATH(mp, (path), false) 2921 #define LOCK_MOUNT_EXACT(mp, path) \ 2922 struct pinned_mountpoint mp __cleanup(unlock_mount) = {}; \ 2923 lock_mount_exact((path), &mp, false, 0) 2924 #define LOCK_MOUNT_EXACT_COPY(mp, path, copy_flags) \ 2925 struct pinned_mountpoint mp __cleanup(unlock_mount) = {}; \ 2926 lock_mount_exact((path), &mp, true, (copy_flags)) 2927 2928 static int graft_tree(struct mount *mnt, const struct pinned_mountpoint *mp) 2929 { 2930 if (mnt->mnt.mnt_sb->s_flags & SB_NOUSER) 2931 return -EINVAL; 2932 2933 if (d_is_dir(mp->mp->m_dentry) != 2934 d_is_dir(mnt->mnt.mnt_root)) 2935 return -ENOTDIR; 2936 2937 return attach_recursive_mnt(mnt, mp); 2938 } 2939 2940 static int may_change_propagation(const struct mount *m) 2941 { 2942 struct mnt_namespace *ns = m->mnt_ns; 2943 2944 // it must be mounted in some namespace 2945 if (IS_ERR_OR_NULL(ns)) // is_mounted() 2946 return -EINVAL; 2947 // and the caller must be admin in userns of that namespace 2948 if (!ns_capable(ns->user_ns, CAP_SYS_ADMIN)) 2949 return -EPERM; 2950 return 0; 2951 } 2952 2953 /* 2954 * Sanity check the flags to change_mnt_propagation. 2955 */ 2956 2957 static int flags_to_propagation_type(int ms_flags) 2958 { 2959 int type = ms_flags & ~(MS_REC | MS_SILENT); 2960 2961 /* Fail if any non-propagation flags are set */ 2962 if (type & ~(MS_SHARED | MS_PRIVATE | MS_SLAVE | MS_UNBINDABLE)) 2963 return 0; 2964 /* Only one propagation flag should be set */ 2965 if (!is_power_of_2(type)) 2966 return 0; 2967 return type; 2968 } 2969 2970 /* 2971 * recursively change the type of the mountpoint. 2972 */ 2973 static int do_change_type(const struct path *path, int ms_flags) 2974 { 2975 struct mount *m; 2976 struct mount *mnt = real_mount(path->mnt); 2977 int recurse = ms_flags & MS_REC; 2978 int type; 2979 int err; 2980 2981 if (!path_mounted(path)) 2982 return -EINVAL; 2983 2984 type = flags_to_propagation_type(ms_flags); 2985 if (!type) 2986 return -EINVAL; 2987 2988 guard(namespace_excl)(); 2989 2990 err = may_change_propagation(mnt); 2991 if (err) 2992 return err; 2993 2994 if (type == MS_SHARED) { 2995 err = invent_group_ids(mnt, recurse); 2996 if (err) 2997 return err; 2998 } 2999 3000 for (m = mnt; m; m = (recurse ? next_mnt(m, mnt) : NULL)) 3001 change_mnt_propagation(m, type); 3002 3003 guard(mount_locked_reader)(); 3004 touch_mnt_namespace(mnt->mnt_ns); 3005 3006 return 0; 3007 } 3008 3009 /* may_copy_tree() - check if a mount tree can be copied 3010 * @path: path to the mount tree to be copied 3011 * 3012 * This helper checks if the caller may copy the mount tree starting 3013 * from @path->mnt. The caller may copy the mount tree under the 3014 * following circumstances: 3015 * 3016 * (1) The caller is located in the mount namespace of the mount tree. 3017 * This also implies that the mount does not belong to an anonymous 3018 * mount namespace. 3019 * (2) The caller tries to copy an nfs mount referring to a mount 3020 * namespace, i.e., the caller is trying to copy a mount namespace 3021 * entry from nsfs. 3022 * (3) The caller tries to copy a pidfs mount referring to a pidfd. 3023 * (4) The caller is trying to copy a mount tree that belongs to an 3024 * anonymous mount namespace. 3025 * 3026 * For that to be safe, this helper enforces that the origin mount 3027 * namespace the anonymous mount namespace was created from is the 3028 * same as the caller's mount namespace by comparing the sequence 3029 * numbers. 3030 * 3031 * This is not strictly necessary. The current semantics of the new 3032 * mount api enforce that the caller must be located in the same 3033 * mount namespace as the mount tree it interacts with. Using the 3034 * origin sequence number preserves these semantics even for 3035 * anonymous mount namespaces. However, one could envision extending 3036 * the api to directly operate across mount namespace if needed. 3037 * 3038 * The ownership of a non-anonymous mount namespace such as the 3039 * caller's cannot change. 3040 * => We know that the caller's mount namespace is stable. 3041 * 3042 * If the origin sequence number of the anonymous mount namespace is 3043 * the same as the sequence number of the caller's mount namespace. 3044 * => The owning namespaces are the same. 3045 * 3046 * ==> The earlier capability check on the owning namespace of the 3047 * caller's mount namespace ensures that the caller has the 3048 * ability to copy the mount tree. 3049 * 3050 * Returns true if the mount tree can be copied, false otherwise. 3051 */ 3052 static inline bool may_copy_tree(const struct path *path) 3053 { 3054 struct mount *mnt = real_mount(path->mnt); 3055 const struct dentry_operations *d_op; 3056 3057 if (check_mnt(mnt)) 3058 return true; 3059 3060 d_op = path->dentry->d_op; 3061 if (d_op == &ns_dentry_operations) 3062 return true; 3063 3064 if (d_op == &pidfs_dentry_operations) 3065 return true; 3066 3067 if (!is_mounted(path->mnt)) 3068 return false; 3069 3070 return check_anonymous_mnt(mnt); 3071 } 3072 3073 static struct mount *__do_loopback(const struct path *old_path, 3074 bool recurse, unsigned int copy_flags) 3075 { 3076 struct mount *old = real_mount(old_path->mnt); 3077 3078 if (IS_MNT_UNBINDABLE(old)) 3079 return ERR_PTR(-EINVAL); 3080 3081 if (!may_copy_tree(old_path)) 3082 return ERR_PTR(-EINVAL); 3083 3084 /* a pseudo dentry is freed without an RCU delay, no walk may find it */ 3085 if (old_path->dentry->d_flags & DCACHE_NORCU) 3086 return ERR_PTR(-EINVAL); 3087 3088 if (recurse && !old->mnt_ns) 3089 return ERR_PTR(-EINVAL); 3090 3091 if (!recurse && has_locked_children(old, old_path->dentry)) 3092 return ERR_PTR(-EINVAL); 3093 3094 if (recurse) 3095 return copy_tree(old, old_path->dentry, copy_flags); 3096 3097 return clone_mnt(old, old_path->dentry, copy_flags); 3098 } 3099 3100 /* 3101 * do loopback mount. 3102 */ 3103 static int do_loopback(const struct path *path, const char *old_name, 3104 int recurse) 3105 { 3106 struct path old_path __free(path_put) = {}; 3107 struct mount *mnt = NULL; 3108 int err; 3109 3110 if (!old_name || !*old_name) 3111 return -EINVAL; 3112 err = kern_path(old_name, LOOKUP_FOLLOW|LOOKUP_AUTOMOUNT, &old_path); 3113 if (err) 3114 return err; 3115 3116 if (mnt_ns_loop(old_path.dentry)) 3117 return -EINVAL; 3118 3119 LOCK_MOUNT(mp, path); 3120 if (IS_ERR(mp.parent)) 3121 return PTR_ERR(mp.parent); 3122 3123 if (!check_mnt(mp.parent)) 3124 return -EINVAL; 3125 3126 mnt = __do_loopback(&old_path, recurse, CL_COPY_MNT_NS_FILE); 3127 if (IS_ERR(mnt)) 3128 return PTR_ERR(mnt); 3129 3130 /* the copy may carry mount namespace files from below the source */ 3131 if (recurse && !check_for_nsfs_mounts(mnt)) 3132 err = -EINVAL; 3133 else 3134 err = graft_tree(mnt, &mp); 3135 if (err) { 3136 lock_mount_hash(); 3137 umount_tree(mnt, UMOUNT_SYNC); 3138 unlock_mount_hash(); 3139 } 3140 return err; 3141 } 3142 3143 static struct mnt_namespace *get_detached_copy(const struct path *path, unsigned int flags) 3144 { 3145 struct mnt_namespace *ns, *mnt_ns = current->nsproxy->mnt_ns, *src_mnt_ns; 3146 struct user_namespace *user_ns = mnt_ns->user_ns; 3147 struct mount *mnt, *p; 3148 3149 ns = alloc_mnt_ns(user_ns, true); 3150 if (IS_ERR(ns)) 3151 return ns; 3152 3153 guard(namespace_excl)(); 3154 3155 /* 3156 * Record the sequence number of the source mount namespace. 3157 * This needs to hold namespace_sem to ensure that the mount 3158 * doesn't get attached. 3159 */ 3160 if (is_mounted(path->mnt)) { 3161 src_mnt_ns = real_mount(path->mnt)->mnt_ns; 3162 if (is_anon_ns(src_mnt_ns)) 3163 ns->seq_origin = src_mnt_ns->seq_origin; 3164 else 3165 ns->seq_origin = src_mnt_ns->ns.ns_id; 3166 } 3167 3168 mnt = __do_loopback(path, (flags & AT_RECURSIVE), CL_COPY_MNT_NS_FILE); 3169 if (IS_ERR(mnt)) { 3170 emptied_ns = ns; 3171 return ERR_CAST(mnt); 3172 } 3173 3174 for (p = mnt; p; p = next_mnt(p, mnt)) { 3175 mnt_add_to_ns(ns, p); 3176 ns->nr_mounts++; 3177 } 3178 ns->root = mnt; 3179 return ns; 3180 } 3181 3182 static struct file *open_detached_copy(struct path *path, unsigned int flags) 3183 { 3184 struct mnt_namespace *ns = get_detached_copy(path, flags); 3185 struct file *file; 3186 3187 if (IS_ERR(ns)) 3188 return ERR_CAST(ns); 3189 3190 mntput(path->mnt); 3191 path->mnt = mntget(&ns->root->mnt); 3192 file = dentry_open(path, O_PATH, current_cred()); 3193 if (IS_ERR(file)) 3194 dissolve_on_fput(no_free_ptr(path->mnt)); 3195 else 3196 file->f_mode |= FMODE_NEED_UNMOUNT; 3197 return file; 3198 } 3199 3200 enum mount_copy_flags_t { 3201 MOUNT_COPY_RECURSIVE = (1 << 0), 3202 MOUNT_COPY_NEW = (1 << 1), 3203 }; 3204 3205 static struct mnt_namespace *create_new_namespace(struct path *path, 3206 enum mount_copy_flags_t flags) 3207 { 3208 struct mnt_namespace *ns = current->nsproxy->mnt_ns; 3209 struct user_namespace *user_ns = current_user_ns(); 3210 struct mnt_namespace *new_ns; 3211 struct mount *new_ns_root, *old_ns_root; 3212 struct path to_path; 3213 struct mount *mnt; 3214 unsigned int copy_flags = 0; 3215 bool locked = false, recurse = flags & MOUNT_COPY_RECURSIVE; 3216 bool foreign = user_ns != ns->user_ns; 3217 3218 if (unlikely(!d_can_lookup(path->dentry))) 3219 return ERR_PTR(-ENOTDIR); 3220 3221 /* 3222 * Without privileges over the mount namespace the copy is made from 3223 * nothing mounted below @path may be left out. It would reveal what 3224 * it covers. That's what unshare() gives such a caller as well. 3225 */ 3226 if (foreign) 3227 copy_flags |= CL_SLAVE | CL_COPY_UNBINDABLE; 3228 3229 new_ns = alloc_mnt_ns(user_ns, false); 3230 if (IS_ERR(new_ns)) 3231 return ERR_CAST(new_ns); 3232 3233 old_ns_root = ns->root; 3234 to_path.mnt = &old_ns_root->mnt; 3235 to_path.dentry = old_ns_root->mnt.mnt_root; 3236 3237 VFS_WARN_ON_ONCE(old_ns_root->mnt.mnt_sb->s_type != &nullfs_fs_type); 3238 3239 LOCK_MOUNT_EXACT_COPY(mp, &to_path, copy_flags); 3240 if (IS_ERR(mp.parent)) { 3241 free_mnt_ns(new_ns); 3242 return ERR_CAST(mp.parent); 3243 } 3244 new_ns_root = mp.parent; 3245 3246 /* 3247 * If the real rootfs had a locked mount on top of it somewhere 3248 * in the stack, lock the new mount tree as well so it can't be 3249 * exposed. 3250 */ 3251 mnt = old_ns_root; 3252 while (mnt->overmount) { 3253 mnt = mnt->overmount; 3254 if (mnt->mnt.mnt_flags & MNT_LOCKED) 3255 locked = true; 3256 } 3257 3258 /* 3259 * We don't emulate unshare()ing a mount namespace. We stick to 3260 * the restrictions of creating detached bind-mounts. It has a 3261 * lot saner and simpler semantics. A caller without privileges 3262 * over the mount namespace can't leave out any child though. 3263 */ 3264 if (flags & MOUNT_COPY_NEW) 3265 mnt = clone_mnt(real_mount(path->mnt), path->dentry, copy_flags); 3266 else if (foreign && !recurse && 3267 __has_children(real_mount(path->mnt), path->dentry)) 3268 mnt = ERR_PTR(-EINVAL); 3269 else 3270 mnt = __do_loopback(path, recurse, copy_flags); 3271 scoped_guard(mount_writer) { 3272 if (IS_ERR(mnt)) { 3273 emptied_ns = new_ns; 3274 umount_tree(new_ns_root, 0); 3275 return ERR_CAST(mnt); 3276 } 3277 3278 if (locked) 3279 mnt->mnt.mnt_flags |= MNT_LOCKED; 3280 /* 3281 * now mount the detached tree on top of the copy 3282 * of the real rootfs we created. 3283 */ 3284 attach_mnt(mnt, new_ns_root, mp.mp); 3285 if (foreign) 3286 lock_mnt_tree(new_ns_root); 3287 } 3288 3289 for (mnt = new_ns_root; mnt; mnt = next_mnt(mnt, new_ns_root)) { 3290 mnt_add_to_ns(new_ns, mnt); 3291 new_ns->nr_mounts++; 3292 } 3293 3294 new_ns->root = new_ns_root; 3295 ns_tree_add_raw(new_ns); 3296 return new_ns; 3297 } 3298 3299 static struct file *open_new_namespace(struct path *path, 3300 enum mount_copy_flags_t flags) 3301 { 3302 struct mnt_namespace *new_ns; 3303 3304 new_ns = create_new_namespace(path, flags); 3305 if (IS_ERR(new_ns)) 3306 return ERR_CAST(new_ns); 3307 return open_namespace_file(to_ns_common(new_ns)); 3308 } 3309 3310 static struct file *vfs_open_tree(int dfd, const char __user *filename, unsigned int flags) 3311 { 3312 int ret; 3313 struct path path __free(path_put) = {}; 3314 int lookup_flags = LOOKUP_AUTOMOUNT | LOOKUP_FOLLOW; 3315 3316 BUILD_BUG_ON(OPEN_TREE_CLOEXEC != O_CLOEXEC); 3317 3318 if (flags & ~(AT_EMPTY_PATH | AT_NO_AUTOMOUNT | AT_RECURSIVE | 3319 AT_SYMLINK_NOFOLLOW | OPEN_TREE_CLONE | 3320 OPEN_TREE_CLOEXEC | OPEN_TREE_NAMESPACE)) 3321 return ERR_PTR(-EINVAL); 3322 3323 if ((flags & (AT_RECURSIVE | OPEN_TREE_CLONE | OPEN_TREE_NAMESPACE)) == 3324 AT_RECURSIVE) 3325 return ERR_PTR(-EINVAL); 3326 3327 if (hweight32(flags & (OPEN_TREE_CLONE | OPEN_TREE_NAMESPACE)) > 1) 3328 return ERR_PTR(-EINVAL); 3329 3330 if (flags & AT_NO_AUTOMOUNT) 3331 lookup_flags &= ~LOOKUP_AUTOMOUNT; 3332 if (flags & AT_SYMLINK_NOFOLLOW) 3333 lookup_flags &= ~LOOKUP_FOLLOW; 3334 3335 /* 3336 * If we create a new mount namespace with the cloned mount tree we 3337 * just care about being privileged over our current user namespace. 3338 * The new mount namespace will be owned by it. 3339 */ 3340 if ((flags & OPEN_TREE_NAMESPACE) && 3341 !ns_capable(current_user_ns(), CAP_SYS_ADMIN)) 3342 return ERR_PTR(-EPERM); 3343 3344 if ((flags & OPEN_TREE_CLONE) && !may_mount()) 3345 return ERR_PTR(-EPERM); 3346 3347 CLASS(filename_uflags, name)(filename, flags); 3348 ret = filename_lookup(dfd, name, lookup_flags, &path, NULL); 3349 if (unlikely(ret)) 3350 return ERR_PTR(ret); 3351 3352 if (flags & OPEN_TREE_NAMESPACE) 3353 return open_new_namespace(&path, (flags & AT_RECURSIVE) ? MOUNT_COPY_RECURSIVE : 0); 3354 3355 if (flags & OPEN_TREE_CLONE) 3356 return open_detached_copy(&path, flags); 3357 3358 return dentry_open(&path, O_PATH, current_cred()); 3359 } 3360 3361 SYSCALL_DEFINE3(open_tree, int, dfd, const char __user *, filename, unsigned, flags) 3362 { 3363 return FD_ADD(flags, vfs_open_tree(dfd, filename, flags)); 3364 } 3365 3366 /* 3367 * Don't allow locked mount flags to be cleared. 3368 * 3369 * No locks need to be held here while testing the various MNT_LOCK 3370 * flags because those flags can never be cleared once they are set. 3371 */ 3372 static bool can_change_locked_flags(struct mount *mnt, unsigned int mnt_flags) 3373 { 3374 unsigned int fl = mnt->mnt.mnt_flags; 3375 3376 if ((fl & MNT_LOCK_READONLY) && 3377 !(mnt_flags & MNT_READONLY)) 3378 return false; 3379 3380 if ((fl & MNT_LOCK_NODEV) && 3381 !(mnt_flags & MNT_NODEV)) 3382 return false; 3383 3384 if ((fl & MNT_LOCK_NOSUID) && 3385 !(mnt_flags & MNT_NOSUID)) 3386 return false; 3387 3388 if ((fl & MNT_LOCK_NOEXEC) && 3389 !(mnt_flags & MNT_NOEXEC)) 3390 return false; 3391 3392 if ((fl & MNT_LOCK_ATIME) && 3393 ((fl & MNT_ATIME_MASK) != (mnt_flags & MNT_ATIME_MASK))) 3394 return false; 3395 3396 return true; 3397 } 3398 3399 static int change_mount_ro_state(struct mount *mnt, unsigned int mnt_flags) 3400 { 3401 bool readonly_request = (mnt_flags & MNT_READONLY); 3402 3403 if (readonly_request == __mnt_is_readonly(&mnt->mnt)) 3404 return 0; 3405 3406 if (readonly_request) 3407 return mnt_make_readonly(mnt); 3408 3409 mnt->mnt.mnt_flags &= ~MNT_READONLY; 3410 return 0; 3411 } 3412 3413 static void set_mount_attributes(struct mount *mnt, unsigned int mnt_flags) 3414 { 3415 mnt_flags |= mnt->mnt.mnt_flags & ~MNT_USER_SETTABLE_MASK; 3416 mnt->mnt.mnt_flags = mnt_flags; 3417 touch_mnt_namespace(mnt->mnt_ns); 3418 } 3419 3420 static void mnt_warn_timestamp_expiry(const struct path *mountpoint, 3421 struct vfsmount *mnt) 3422 { 3423 struct super_block *sb = mnt->mnt_sb; 3424 3425 if (!__mnt_is_readonly(mnt) && 3426 (!(sb->s_iflags & SB_I_TS_EXPIRY_WARNED)) && 3427 (ktime_get_real_seconds() + TIME_UPTIME_SEC_MAX > sb->s_time_max)) { 3428 char *buf, *mntpath; 3429 3430 buf = __getname(); 3431 if (buf) 3432 mntpath = d_path(mountpoint, buf, PATH_MAX); 3433 else 3434 mntpath = ERR_PTR(-ENOMEM); 3435 if (IS_ERR(mntpath)) 3436 mntpath = "(unknown)"; 3437 3438 pr_warn("%s filesystem being %s at %s supports timestamps until %ptTd (0x%llx)\n", 3439 sb->s_type->name, 3440 is_mounted(mnt) ? "remounted" : "mounted", 3441 mntpath, &sb->s_time_max, 3442 (unsigned long long)sb->s_time_max); 3443 3444 sb->s_iflags |= SB_I_TS_EXPIRY_WARNED; 3445 __putname(buf); 3446 } 3447 } 3448 3449 /* 3450 * Handle reconfiguration of the mountpoint only without alteration of the 3451 * superblock it refers to. This is triggered by specifying MS_REMOUNT|MS_BIND 3452 * to mount(2). 3453 */ 3454 static int do_reconfigure_mnt(const struct path *path, unsigned int mnt_flags) 3455 { 3456 struct super_block *sb = path->mnt->mnt_sb; 3457 struct mount *mnt = real_mount(path->mnt); 3458 int ret; 3459 3460 if (!check_mnt(mnt)) 3461 return -EINVAL; 3462 3463 if (!path_mounted(path)) 3464 return -EINVAL; 3465 3466 if (!can_change_locked_flags(mnt, mnt_flags)) 3467 return -EPERM; 3468 3469 /* 3470 * We're only checking whether the superblock is read-only not 3471 * changing it, so only take down_read(&sb->s_umount). 3472 */ 3473 down_read(&sb->s_umount); 3474 lock_mount_hash(); 3475 ret = change_mount_ro_state(mnt, mnt_flags); 3476 if (ret == 0) 3477 set_mount_attributes(mnt, mnt_flags); 3478 unlock_mount_hash(); 3479 up_read(&sb->s_umount); 3480 3481 mnt_warn_timestamp_expiry(path, &mnt->mnt); 3482 3483 return ret; 3484 } 3485 3486 /* 3487 * change filesystem flags. dir should be a physical root of filesystem. 3488 * If you've mounted a non-root directory somewhere and want to do remount 3489 * on it - tough luck. 3490 */ 3491 static int do_remount(const struct path *path, int sb_flags, 3492 int mnt_flags, void *data) 3493 { 3494 int err; 3495 struct super_block *sb = path->mnt->mnt_sb; 3496 struct mount *mnt = real_mount(path->mnt); 3497 struct fs_context *fc; 3498 3499 if (!check_mnt(mnt)) 3500 return -EINVAL; 3501 3502 if (!path_mounted(path)) 3503 return -EINVAL; 3504 3505 if (!can_change_locked_flags(mnt, mnt_flags)) 3506 return -EPERM; 3507 3508 fc = fs_context_for_reconfigure(path->dentry, sb_flags, MS_RMT_MASK); 3509 if (IS_ERR(fc)) 3510 return PTR_ERR(fc); 3511 3512 /* 3513 * Indicate to the filesystem that the remount request is coming 3514 * from the legacy mount system call. 3515 */ 3516 fc->oldapi = true; 3517 3518 err = parse_monolithic_mount_data(fc, data); 3519 if (!err) { 3520 down_write(&sb->s_umount); 3521 err = -EPERM; 3522 if (ns_capable(sb->s_user_ns, CAP_SYS_ADMIN)) { 3523 err = reconfigure_super(fc); 3524 if (!err) { 3525 lock_mount_hash(); 3526 set_mount_attributes(mnt, mnt_flags); 3527 unlock_mount_hash(); 3528 } 3529 } 3530 up_write(&sb->s_umount); 3531 } 3532 3533 mnt_warn_timestamp_expiry(path, &mnt->mnt); 3534 3535 put_fs_context(fc); 3536 return err; 3537 } 3538 3539 static inline int tree_contains_unbindable(struct mount *mnt) 3540 { 3541 struct mount *p; 3542 for (p = mnt; p; p = next_mnt(p, mnt)) { 3543 if (IS_MNT_UNBINDABLE(p)) 3544 return 1; 3545 } 3546 return 0; 3547 } 3548 3549 static int do_set_group(const struct path *from_path, const struct path *to_path) 3550 { 3551 struct mount *from = real_mount(from_path->mnt); 3552 struct mount *to = real_mount(to_path->mnt); 3553 int err; 3554 3555 guard(namespace_excl)(); 3556 3557 err = may_change_propagation(from); 3558 if (err) 3559 return err; 3560 err = may_change_propagation(to); 3561 if (err) 3562 return err; 3563 3564 /* To and From paths should be mount roots */ 3565 if (!path_mounted(from_path)) 3566 return -EINVAL; 3567 if (!path_mounted(to_path)) 3568 return -EINVAL; 3569 3570 /* Setting sharing groups is only allowed across same superblock */ 3571 if (from->mnt.mnt_sb != to->mnt.mnt_sb) 3572 return -EINVAL; 3573 3574 /* From mount root should be wider than To mount root */ 3575 if (!is_subdir(to->mnt.mnt_root, from->mnt.mnt_root)) 3576 return -EINVAL; 3577 3578 /* From mount should not have locked children in place of To's root */ 3579 if (has_locked_children(from, to->mnt.mnt_root)) 3580 return -EINVAL; 3581 3582 /* Setting sharing groups is only allowed on private mounts */ 3583 if (IS_MNT_SHARED(to) || IS_MNT_SLAVE(to) || IS_MNT_UNBINDABLE(to)) 3584 return -EINVAL; 3585 3586 /* From should not be private */ 3587 if (!IS_MNT_SHARED(from) && !IS_MNT_SLAVE(from)) 3588 return -EINVAL; 3589 3590 if (IS_MNT_SLAVE(from)) { 3591 hlist_add_behind(&to->mnt_slave, &from->mnt_slave); 3592 to->mnt_master = from->mnt_master; 3593 } 3594 3595 if (IS_MNT_SHARED(from)) { 3596 to->mnt_group_id = from->mnt_group_id; 3597 list_add(&to->mnt_share, &from->mnt_share); 3598 set_mnt_shared(to); 3599 } 3600 3601 guard(mount_locked_reader)(); 3602 touch_mnt_namespace(to->mnt_ns); 3603 3604 return 0; 3605 } 3606 3607 /** 3608 * path_overmounted - check if path is overmounted 3609 * @path: path to check 3610 * 3611 * Check if path is overmounted, i.e., if there's a mount on top of 3612 * @path->mnt with @path->dentry as mountpoint. 3613 * 3614 * Context: namespace_sem must be held at least shared. 3615 * MUST NOT be called under lock_mount_hash() (there one should just 3616 * call __lookup_mnt() and check if it returns NULL). 3617 * Return: If path is overmounted true is returned, false if not. 3618 */ 3619 static inline bool path_overmounted(const struct path *path) 3620 { 3621 unsigned seq = read_seqbegin(&mount_lock); 3622 bool no_child; 3623 3624 rcu_read_lock(); 3625 no_child = !__lookup_mnt(path->mnt, path->dentry); 3626 rcu_read_unlock(); 3627 if (need_seqretry(&mount_lock, seq)) { 3628 read_seqlock_excl(&mount_lock); 3629 no_child = !__lookup_mnt(path->mnt, path->dentry); 3630 read_sequnlock_excl(&mount_lock); 3631 } 3632 return unlikely(!no_child); 3633 } 3634 3635 /* 3636 * Check if there is a possibly empty chain of descent from p1 to p2. 3637 * Locks: namespace_sem (shared) or mount_lock (read_seqlock_excl). 3638 */ 3639 static bool mount_is_ancestor(const struct mount *p1, const struct mount *p2) 3640 { 3641 while (p2 != p1 && mnt_has_parent(p2)) 3642 p2 = p2->mnt_parent; 3643 return p2 == p1; 3644 } 3645 3646 /** 3647 * can_move_mount_beneath - check that we can mount beneath the top mount 3648 * @mnt_from: mount we are trying to move 3649 * @mnt_to: mount under which to mount 3650 * @mp: mountpoint of @mnt_to 3651 * 3652 * - Make sure that the caller can unmount the topmost mount ensuring 3653 * that the caller could reveal the underlying mountpoint. 3654 * - Ensure that nothing has been mounted on top of @mnt_from before we 3655 * grabbed @namespace_sem to avoid creating pointless shadow mounts. 3656 * - Prevent mounting beneath a mount if the propagation relationship 3657 * between the source mount, parent mount, and top mount would lead to 3658 * nonsensical mount trees. 3659 * 3660 * Context: This function expects namespace_lock() to be held. 3661 * Return: On success 0, and on error a negative error code is returned. 3662 */ 3663 static int can_move_mount_beneath(const struct mount *mnt_from, 3664 const struct mount *mnt_to, 3665 struct pinned_mountpoint *mp) 3666 { 3667 struct mount *parent_mnt_to = mnt_to->mnt_parent; 3668 3669 /* Avoid creating shadow mounts during mount propagation. */ 3670 if (mnt_from->overmount) 3671 return -EINVAL; 3672 3673 if (mount_is_ancestor(mnt_to, mnt_from)) 3674 return -EINVAL; 3675 3676 /* 3677 * If the parent mount propagates to the child mount this would 3678 * mean mounting @mnt_from on @mnt_to->mnt_parent and then 3679 * propagating a copy @c of @mnt_from on top of @mnt_to. This 3680 * defeats the whole purpose of mounting beneath another mount. 3681 */ 3682 if (propagation_would_overmount(parent_mnt_to, mnt_to, mp->mp)) 3683 return -EINVAL; 3684 3685 /* 3686 * If @mnt_to->mnt_parent propagates to @mnt_from this would 3687 * mean propagating a copy @c of @mnt_from on top of @mnt_from. 3688 * Afterwards @mnt_from would be mounted on top of 3689 * @mnt_to->mnt_parent and @mnt_to would be unmounted from 3690 * @mnt->mnt_parent and remounted on @mnt_from. But since @c is 3691 * already mounted on @mnt_from, @mnt_to would ultimately be 3692 * remounted on top of @c. Afterwards, @mnt_from would be 3693 * covered by a copy @c of @mnt_from and @c would be covered by 3694 * @mnt_from itself. This defeats the whole purpose of mounting 3695 * @mnt_from beneath @mnt_to. 3696 */ 3697 if (check_mnt(mnt_from) && 3698 propagation_would_overmount(parent_mnt_to, mnt_from, mp->mp)) 3699 return -EINVAL; 3700 3701 return 0; 3702 } 3703 3704 /* may_use_mount() - check if a mount tree can be used 3705 * @mnt: vfsmount to be used 3706 * 3707 * This helper checks if the caller may use the mount tree starting 3708 * from @path->mnt. The caller may use the mount tree under the 3709 * following circumstances: 3710 * 3711 * (1) The caller is located in the mount namespace of the mount tree. 3712 * This also implies that the mount does not belong to an anonymous 3713 * mount namespace. 3714 * (2) The caller is trying to use a mount tree that belongs to an 3715 * anonymous mount namespace. 3716 * 3717 * For that to be safe, this helper enforces that the origin mount 3718 * namespace the anonymous mount namespace was created from is the 3719 * same as the caller's mount namespace by comparing the sequence 3720 * numbers. 3721 * 3722 * The ownership of a non-anonymous mount namespace such as the 3723 * caller's cannot change. 3724 * => We know that the caller's mount namespace is stable. 3725 * 3726 * If the origin sequence number of the anonymous mount namespace is 3727 * the same as the sequence number of the caller's mount namespace. 3728 * => The owning namespaces are the same. 3729 * 3730 * ==> The earlier capability check on the owning namespace of the 3731 * caller's mount namespace ensures that the caller has the 3732 * ability to use the mount tree. 3733 * 3734 * Returns true if the mount tree can be used, false otherwise. 3735 */ 3736 static inline bool may_use_mount(struct mount *mnt) 3737 { 3738 if (check_mnt(mnt)) 3739 return true; 3740 3741 /* 3742 * Make sure that noone unmounted the target path or somehow 3743 * managed to get their hands on something purely kernel 3744 * internal. 3745 */ 3746 if (!is_mounted(&mnt->mnt)) 3747 return false; 3748 3749 return check_anonymous_mnt(mnt); 3750 } 3751 3752 static int do_move_mount(const struct path *old_path, 3753 const struct path *new_path, 3754 enum mnt_tree_flags_t flags) 3755 { 3756 struct mount *old = real_mount(old_path->mnt); 3757 int err; 3758 bool beneath = flags & MNT_TREE_BENEATH; 3759 3760 if (!path_mounted(old_path)) 3761 return -EINVAL; 3762 3763 if (d_is_dir(new_path->dentry) != d_is_dir(old_path->dentry)) 3764 return -EINVAL; 3765 3766 LOCK_MOUNT_MAYBE_BENEATH(mp, new_path, beneath); 3767 if (IS_ERR(mp.parent)) 3768 return PTR_ERR(mp.parent); 3769 3770 if (check_mnt(old)) { 3771 /* if the source is in our namespace... */ 3772 /* ... it should be detachable from parent */ 3773 if (!mnt_has_parent(old) || IS_MNT_LOCKED(old)) 3774 return -EINVAL; 3775 /* ... which should not be shared */ 3776 if (IS_MNT_SHARED(old->mnt_parent)) 3777 return -EINVAL; 3778 /* ... and the target should be in our namespace */ 3779 if (!check_mnt(mp.parent)) 3780 return -EINVAL; 3781 } else { 3782 /* 3783 * otherwise the source must be the root of some anon namespace. 3784 */ 3785 if (!anon_ns_root(old)) 3786 return -EINVAL; 3787 /* 3788 * Bail out early if the target is within the same namespace - 3789 * subsequent checks would've rejected that, but they lose 3790 * some corner cases if we check it early. 3791 */ 3792 if (old->mnt_ns == mp.parent->mnt_ns) 3793 return -EINVAL; 3794 /* 3795 * Target should be either in our namespace or in an acceptable 3796 * anon namespace, sensu check_anonymous_mnt(). 3797 */ 3798 if (!may_use_mount(mp.parent)) 3799 return -EINVAL; 3800 } 3801 3802 if (beneath) { 3803 struct mount *over = real_mount(new_path->mnt); 3804 3805 if (mp.parent != over->mnt_parent) 3806 over = mp.parent->overmount; 3807 err = can_move_mount_beneath(old, over, &mp); 3808 if (err) 3809 return err; 3810 } 3811 3812 /* 3813 * Don't move a mount tree containing unbindable mounts to a destination 3814 * mount which is shared. 3815 */ 3816 if (IS_MNT_SHARED(mp.parent) && tree_contains_unbindable(old)) 3817 return -EINVAL; 3818 if (!check_for_nsfs_mounts(old)) 3819 return -ELOOP; 3820 if (mount_is_ancestor(old, mp.parent)) 3821 return -ELOOP; 3822 3823 return attach_recursive_mnt(old, &mp); 3824 } 3825 3826 static int do_move_mount_old(const struct path *path, const char *old_name) 3827 { 3828 struct path old_path __free(path_put) = {}; 3829 int err; 3830 3831 if (!old_name || !*old_name) 3832 return -EINVAL; 3833 3834 err = kern_path(old_name, LOOKUP_FOLLOW, &old_path); 3835 if (err) 3836 return err; 3837 3838 return do_move_mount(&old_path, path, 0); 3839 } 3840 3841 /* 3842 * add a mount into a namespace's mount tree 3843 */ 3844 static int do_add_mount(struct mount *newmnt, const struct pinned_mountpoint *mp, 3845 int mnt_flags) 3846 { 3847 struct mount *parent = mp->parent; 3848 3849 if (IS_ERR(parent)) 3850 return PTR_ERR(parent); 3851 3852 mnt_flags &= ~MNT_INTERNAL_FLAGS; 3853 3854 if (unlikely(!check_mnt(parent))) { 3855 /* that's acceptable only for automounts done in private ns */ 3856 if (!(mnt_flags & MNT_SHRINKABLE)) 3857 return -EINVAL; 3858 /* ... and for those we'd better have mountpoint still alive */ 3859 if (!is_mounted(&parent->mnt)) 3860 return -EINVAL; 3861 } 3862 3863 /* Refuse the same filesystem on the same mount point */ 3864 if (parent->mnt.mnt_sb == newmnt->mnt.mnt_sb && 3865 parent->mnt.mnt_root == mp->mp->m_dentry) 3866 return -EBUSY; 3867 3868 if (d_is_symlink(newmnt->mnt.mnt_root)) 3869 return -EINVAL; 3870 3871 newmnt->mnt.mnt_flags = mnt_flags; 3872 return graft_tree(newmnt, mp); 3873 } 3874 3875 static bool mount_too_revealing(const struct super_block *sb, int *new_mnt_flags); 3876 3877 /* 3878 * Create a new mount using a superblock configuration and request it 3879 * be added to the namespace tree. 3880 */ 3881 static int do_new_mount_fc(struct fs_context *fc, const struct path *mountpoint, 3882 unsigned int mnt_flags) 3883 { 3884 struct super_block *sb; 3885 struct vfsmount *mnt __free(mntput) = fc_mount(fc); 3886 int error; 3887 3888 if (IS_ERR(mnt)) 3889 return PTR_ERR(mnt); 3890 3891 sb = fc->root->d_sb; 3892 error = security_sb_kern_mount(sb); 3893 if (unlikely(error)) 3894 return error; 3895 3896 if (unlikely(mount_too_revealing(sb, &mnt_flags))) { 3897 errorfcp(fc, "VFS", "Mount too revealing"); 3898 return -EPERM; 3899 } 3900 3901 mnt_warn_timestamp_expiry(mountpoint, mnt); 3902 3903 LOCK_MOUNT(mp, mountpoint); 3904 error = do_add_mount(real_mount(mnt), &mp, mnt_flags); 3905 if (!error) 3906 retain_and_null_ptr(mnt); // consumed on success 3907 return error; 3908 } 3909 3910 /* 3911 * create a new mount for userspace and request it to be added into the 3912 * namespace's tree 3913 */ 3914 static int do_new_mount(const struct path *path, const char *fstype, 3915 int sb_flags, int mnt_flags, 3916 const char *name, void *data) 3917 { 3918 struct file_system_type *type; 3919 struct fs_context *fc; 3920 const char *subtype = NULL; 3921 int err = 0; 3922 3923 if (!fstype) 3924 return -EINVAL; 3925 3926 type = get_fs_type(fstype); 3927 if (!type) 3928 return -ENODEV; 3929 3930 if (type->fs_flags & FS_HAS_SUBTYPE) { 3931 subtype = strchr(fstype, '.'); 3932 if (subtype) { 3933 subtype++; 3934 if (!*subtype) { 3935 put_filesystem(type); 3936 return -EINVAL; 3937 } 3938 } 3939 } 3940 3941 fc = fs_context_for_mount(type, sb_flags); 3942 put_filesystem(type); 3943 if (IS_ERR(fc)) 3944 return PTR_ERR(fc); 3945 3946 /* 3947 * Indicate to the filesystem that the mount request is coming 3948 * from the legacy mount system call. 3949 */ 3950 fc->oldapi = true; 3951 3952 if (subtype) 3953 err = vfs_parse_fs_string(fc, "subtype", subtype); 3954 if (!err && name) 3955 err = vfs_parse_fs_string(fc, "source", name); 3956 if (!err) 3957 err = parse_monolithic_mount_data(fc, data); 3958 if (!err && !mount_capable(fc)) 3959 err = -EPERM; 3960 if (!err) 3961 err = do_new_mount_fc(fc, path, mnt_flags); 3962 3963 put_fs_context(fc); 3964 return err; 3965 } 3966 3967 static void lock_mount_exact(const struct path *path, 3968 struct pinned_mountpoint *mp, bool copy_mount, 3969 unsigned int copy_flags) 3970 { 3971 struct dentry *dentry = path->dentry; 3972 int err; 3973 3974 /* Assert that inode_lock() locked the correct inode. */ 3975 VFS_WARN_ON_ONCE(copy_mount && !path_mounted(path)); 3976 3977 inode_lock(dentry->d_inode); 3978 namespace_lock(); 3979 if (unlikely(cant_mount(dentry))) 3980 err = -ENOENT; 3981 else if (!copy_mount && path_overmounted(path)) 3982 err = -EBUSY; 3983 else 3984 err = get_mountpoint(dentry, mp); 3985 if (unlikely(err)) { 3986 namespace_unlock(); 3987 inode_unlock(dentry->d_inode); 3988 mp->parent = ERR_PTR(err); 3989 return; 3990 } 3991 3992 if (copy_mount) 3993 mp->parent = clone_mnt(real_mount(path->mnt), dentry, copy_flags); 3994 else 3995 mp->parent = real_mount(path->mnt); 3996 if (unlikely(IS_ERR(mp->parent))) 3997 __unlock_mount(mp); 3998 } 3999 4000 int finish_automount(struct vfsmount *__m, const struct path *path) 4001 { 4002 struct vfsmount *m __free(mntput) = __m; 4003 struct mount *mnt; 4004 int err; 4005 4006 if (!m) 4007 return 0; 4008 if (IS_ERR(m)) 4009 return PTR_ERR(m); 4010 4011 mnt = real_mount(m); 4012 4013 if (m->mnt_root == path->dentry) 4014 return -ELOOP; 4015 4016 /* 4017 * we don't want to use LOCK_MOUNT() - in this case finding something 4018 * that overmounts our mountpoint to be means "quitely drop what we've 4019 * got", not "try to mount it on top". 4020 */ 4021 LOCK_MOUNT_EXACT(mp, path); 4022 if (mp.parent == ERR_PTR(-EBUSY)) 4023 return 0; 4024 4025 err = do_add_mount(mnt, &mp, path->mnt->mnt_flags | MNT_SHRINKABLE); 4026 if (likely(!err)) 4027 retain_and_null_ptr(m); 4028 return err; 4029 } 4030 4031 /** 4032 * mnt_set_expiry - Put a mount on an expiration list 4033 * @mnt: The mount to list. 4034 * @expiry_list: The list to add the mount to. 4035 */ 4036 void mnt_set_expiry(struct vfsmount *mnt, struct list_head *expiry_list) 4037 { 4038 guard(mount_locked_reader)(); 4039 list_add_tail(&real_mount(mnt)->mnt_expire, expiry_list); 4040 } 4041 EXPORT_SYMBOL(mnt_set_expiry); 4042 4043 /* 4044 * process a list of expirable mountpoints with the intent of discarding any 4045 * mountpoints that aren't in use and haven't been touched since last we came 4046 * here 4047 */ 4048 void mark_mounts_for_expiry(struct list_head *mounts) 4049 { 4050 struct mount *mnt, *next; 4051 LIST_HEAD(graveyard); 4052 4053 if (list_empty(mounts)) 4054 return; 4055 4056 guard(namespace_excl)(); 4057 guard(mount_writer)(); 4058 4059 /* extract from the expiration list every vfsmount that matches the 4060 * following criteria: 4061 * - already mounted 4062 * - only referenced by its parent vfsmount 4063 * - still marked for expiry (marked on the last call here; marks are 4064 * cleared by mntput()) 4065 */ 4066 list_for_each_entry_safe(mnt, next, mounts, mnt_expire) { 4067 if (!is_mounted(&mnt->mnt)) 4068 continue; 4069 /* lock_mnt_tree() leaves expirable mounts alone */ 4070 VFS_WARN_ON_ONCE(IS_MNT_LOCKED(mnt)); 4071 if (!xchg(&mnt->mnt_expiry_mark, 1) || 4072 propagate_mount_busy(mnt, 1)) 4073 continue; 4074 list_move(&mnt->mnt_expire, &graveyard); 4075 } 4076 while (!list_empty(&graveyard)) { 4077 mnt = list_first_entry(&graveyard, struct mount, mnt_expire); 4078 /* an earlier umount_tree() may have moved a busy mount here */ 4079 if (propagate_mount_busy(mnt, 1)) { 4080 list_move(&mnt->mnt_expire, mounts); 4081 continue; 4082 } 4083 touch_mnt_namespace(mnt->mnt_ns); 4084 umount_tree(mnt, UMOUNT_PROPAGATE|UMOUNT_SYNC); 4085 } 4086 } 4087 4088 EXPORT_SYMBOL_GPL(mark_mounts_for_expiry); 4089 4090 /* 4091 * Unmount @mnt if it's a shrinkable mount without children that nobody uses. 4092 * 4093 * mount_lock must be held for write 4094 */ 4095 static bool shrink_submount(struct mount *mnt) 4096 { 4097 /* not the kernel's to remove either */ 4098 if (IS_MNT_LOCKED(mnt) || propagate_mount_busy(mnt, 1)) 4099 return false; 4100 touch_mnt_namespace(mnt->mnt_ns); 4101 umount_tree(mnt, UMOUNT_PROPAGATE|UMOUNT_SYNC); 4102 return true; 4103 } 4104 4105 /* 4106 * Ripoff of 'select_parent()' 4107 * 4108 * unmount the shrinkable submounts of @parent that aren't busy, children 4109 * before their parent, and say whether anything went 4110 * 4111 * The cursor into the children of @this_parent survives the umount of a 4112 * child mount without child mounts. The mounts that get umounted together with 4113 * it are located under receiving mounts of @this_parent and never under 4114 * @this_parent itself. The one exception is @this_parent getting unmounted 4115 * then the walk starts over. 4116 */ 4117 static bool __shrink_submounts(struct mount *parent) 4118 { 4119 struct mount *this_parent = parent; 4120 struct list_head *next; 4121 bool shrunk = false; 4122 4123 repeat: 4124 next = this_parent->mnt_mounts.next; 4125 resume: 4126 while (next != &this_parent->mnt_mounts) { 4127 struct list_head *tmp = next; 4128 struct mount *mnt = list_entry(tmp, struct mount, mnt_child); 4129 4130 next = tmp->next; 4131 if (!(mnt->mnt.mnt_flags & MNT_SHRINKABLE)) 4132 continue; 4133 /* 4134 * Descend a level if the d_mounts list is non-empty. 4135 */ 4136 if (!list_empty(&mnt->mnt_mounts)) { 4137 this_parent = mnt; 4138 goto repeat; 4139 } 4140 if (!shrink_submount(mnt)) 4141 continue; 4142 shrunk = true; 4143 if (unlikely(this_parent->mnt.mnt_flags & MNT_UMOUNT)) 4144 return true; 4145 } 4146 /* 4147 * All done at this level ... ascend and resume the search 4148 */ 4149 if (this_parent != parent) { 4150 struct mount *mnt = this_parent; 4151 4152 next = mnt->mnt_child.next; 4153 this_parent = mnt->mnt_parent; 4154 /* its children are gone, maybe it can go as well */ 4155 if (shrink_submount(mnt)) { 4156 shrunk = true; 4157 if (unlikely(this_parent->mnt.mnt_flags & MNT_UMOUNT)) 4158 return true; 4159 } 4160 goto resume; 4161 } 4162 return shrunk; 4163 } 4164 4165 /* 4166 * unmount the shrinkable submounts of @mnt that aren't busy 4167 * 4168 * The busy check and the umount of a mount are adjacent. An umount can 4169 * still empty or move a mount in a part of the tree that was walked 4170 * already, so walk again until nothing goes. 4171 * 4172 * mount_lock must be held for write 4173 */ 4174 static void shrink_submounts(struct mount *mnt) 4175 { 4176 for (;;) { 4177 if (!__shrink_submounts(mnt)) 4178 break; 4179 } 4180 } 4181 4182 static void *copy_mount_options(const void __user * data) 4183 { 4184 char *copy; 4185 unsigned left, offset; 4186 4187 if (!data) 4188 return NULL; 4189 4190 copy = kmalloc(PAGE_SIZE, GFP_KERNEL); 4191 if (!copy) 4192 return ERR_PTR(-ENOMEM); 4193 4194 left = copy_from_user(copy, data, PAGE_SIZE); 4195 4196 /* 4197 * Not all architectures have an exact copy_from_user(). Resort to 4198 * byte at a time. 4199 */ 4200 offset = PAGE_SIZE - left; 4201 while (left) { 4202 char c; 4203 if (get_user(c, (const char __user *)data + offset)) 4204 break; 4205 copy[offset] = c; 4206 left--; 4207 offset++; 4208 } 4209 4210 if (left == PAGE_SIZE) { 4211 kfree(copy); 4212 return ERR_PTR(-EFAULT); 4213 } 4214 4215 return copy; 4216 } 4217 4218 static char *copy_mount_string(const void __user *data) 4219 { 4220 return data ? strndup_user(data, PATH_MAX) : NULL; 4221 } 4222 4223 /* 4224 * Flags is a 32-bit value that allows up to 31 non-fs dependent flags to 4225 * be given to the mount() call (ie: read-only, no-dev, no-suid etc). 4226 * 4227 * data is a (void *) that can point to any structure up to 4228 * PAGE_SIZE-1 bytes, which can contain arbitrary fs-dependent 4229 * information (or be NULL). 4230 * 4231 * Pre-0.97 versions of mount() didn't have a flags word. 4232 * When the flags word was introduced its top half was required 4233 * to have the magic value 0xC0ED, and this remained so until 2.4.0-test9. 4234 * Therefore, if this magic number is present, it carries no information 4235 * and must be discarded. 4236 */ 4237 int path_mount(const char *dev_name, const struct path *path, 4238 const char *type_page, unsigned long flags, void *data_page) 4239 { 4240 unsigned int mnt_flags = 0, sb_flags; 4241 int ret; 4242 4243 /* Discard magic */ 4244 if ((flags & MS_MGC_MSK) == MS_MGC_VAL) 4245 flags &= ~MS_MGC_MSK; 4246 4247 /* Basic sanity checks */ 4248 if (data_page) 4249 ((char *)data_page)[PAGE_SIZE - 1] = 0; 4250 4251 if (flags & MS_NOUSER) 4252 return -EINVAL; 4253 4254 ret = security_sb_mount(dev_name, path, type_page, flags, data_page); 4255 if (ret) 4256 return ret; 4257 if (!may_mount()) 4258 return -EPERM; 4259 if (flags & SB_MANDLOCK) 4260 warn_mandlock(); 4261 4262 /* Default to relatime unless overriden */ 4263 if (!(flags & MS_NOATIME)) 4264 mnt_flags |= MNT_RELATIME; 4265 4266 /* Separate the per-mountpoint flags */ 4267 if (flags & MS_NOSUID) 4268 mnt_flags |= MNT_NOSUID; 4269 if (flags & MS_NODEV) 4270 mnt_flags |= MNT_NODEV; 4271 if (flags & MS_NOEXEC) 4272 mnt_flags |= MNT_NOEXEC; 4273 if (flags & MS_NOATIME) 4274 mnt_flags |= MNT_NOATIME; 4275 if (flags & MS_NODIRATIME) 4276 mnt_flags |= MNT_NODIRATIME; 4277 if (flags & MS_STRICTATIME) 4278 mnt_flags &= ~(MNT_RELATIME | MNT_NOATIME); 4279 if (flags & MS_RDONLY) 4280 mnt_flags |= MNT_READONLY; 4281 if (flags & MS_NOSYMFOLLOW) 4282 mnt_flags |= MNT_NOSYMFOLLOW; 4283 4284 /* The default atime for remount is preservation */ 4285 if ((flags & MS_REMOUNT) && 4286 ((flags & (MS_NOATIME | MS_NODIRATIME | MS_RELATIME | 4287 MS_STRICTATIME)) == 0)) { 4288 mnt_flags &= ~MNT_ATIME_MASK; 4289 mnt_flags |= path->mnt->mnt_flags & MNT_ATIME_MASK; 4290 } 4291 4292 sb_flags = flags & (SB_RDONLY | 4293 SB_SYNCHRONOUS | 4294 SB_MANDLOCK | 4295 SB_DIRSYNC | 4296 SB_SILENT | 4297 SB_POSIXACL | 4298 SB_LAZYTIME | 4299 SB_I_VERSION); 4300 4301 if ((flags & (MS_REMOUNT | MS_BIND)) == (MS_REMOUNT | MS_BIND)) 4302 return do_reconfigure_mnt(path, mnt_flags); 4303 if (flags & MS_REMOUNT) 4304 return do_remount(path, sb_flags, mnt_flags, data_page); 4305 if (flags & MS_BIND) 4306 return do_loopback(path, dev_name, flags & MS_REC); 4307 if (flags & (MS_SHARED | MS_PRIVATE | MS_SLAVE | MS_UNBINDABLE)) 4308 return do_change_type(path, flags); 4309 if (flags & MS_MOVE) 4310 return do_move_mount_old(path, dev_name); 4311 4312 return do_new_mount(path, type_page, sb_flags, mnt_flags, dev_name, 4313 data_page); 4314 } 4315 4316 int do_mount(const char *dev_name, const char __user *dir_name, 4317 const char *type_page, unsigned long flags, void *data_page) 4318 { 4319 struct path path __free(path_put) = {}; 4320 int ret; 4321 4322 ret = user_path_at(AT_FDCWD, dir_name, LOOKUP_FOLLOW, &path); 4323 if (ret) 4324 return ret; 4325 return path_mount(dev_name, &path, type_page, flags, data_page); 4326 } 4327 4328 static struct ucounts *inc_mnt_namespaces(struct user_namespace *ns) 4329 { 4330 return inc_ucount(ns, current_euid(), UCOUNT_MNT_NAMESPACES); 4331 } 4332 4333 static void dec_mnt_namespaces(struct ucounts *ucounts) 4334 { 4335 dec_ucount(ucounts, UCOUNT_MNT_NAMESPACES); 4336 } 4337 4338 static void free_mnt_ns(struct mnt_namespace *ns) 4339 { 4340 if (!is_anon_ns(ns)) 4341 ns_common_free(ns); 4342 dec_mnt_namespaces(ns->ucounts); 4343 /* the last active reference is gone, no mark can show up anymore */ 4344 fsnotify_mntns_delete(ns); 4345 mnt_ns_tree_remove(ns); 4346 } 4347 4348 static struct mnt_namespace *alloc_mnt_ns(struct user_namespace *user_ns, bool anon) 4349 { 4350 struct mnt_namespace *new_ns; 4351 struct ucounts *ucounts; 4352 int ret; 4353 4354 ucounts = inc_mnt_namespaces(user_ns); 4355 if (!ucounts) 4356 return ERR_PTR(-ENOSPC); 4357 4358 new_ns = kzalloc_obj(struct mnt_namespace, GFP_KERNEL_ACCOUNT); 4359 if (!new_ns) { 4360 dec_mnt_namespaces(ucounts); 4361 return ERR_PTR(-ENOMEM); 4362 } 4363 4364 if (anon) 4365 ret = ns_common_init_inum(new_ns, MNT_NS_ANON_INO); 4366 else 4367 ret = ns_common_init(new_ns); 4368 if (ret) { 4369 kfree(new_ns); 4370 dec_mnt_namespaces(ucounts); 4371 return ERR_PTR(ret); 4372 } 4373 ns_tree_gen_id(new_ns); 4374 4375 new_ns->is_anon = anon; 4376 refcount_set(&new_ns->passive, 1); 4377 new_ns->mounts = RB_ROOT; 4378 init_waitqueue_head(&new_ns->poll); 4379 new_ns->user_ns = get_user_ns(user_ns); 4380 new_ns->ucounts = ucounts; 4381 return new_ns; 4382 } 4383 4384 __latent_entropy 4385 struct mnt_namespace *copy_mnt_ns(u64 flags, struct mnt_namespace *ns, 4386 struct user_namespace *user_ns, struct fs_struct *new_fs) 4387 { 4388 struct mnt_namespace *new_ns; 4389 struct path old_root __free(path_put) = {}; 4390 struct path old_pwd __free(path_put) = {}; 4391 struct mount *p, *q; 4392 struct mount *old; 4393 struct mount *new; 4394 int copy_flags; 4395 4396 BUG_ON(!ns); 4397 4398 if (likely(!(flags & CLONE_NEWNS))) { 4399 get_mnt_ns(ns); 4400 return ns; 4401 } 4402 4403 old = ns->root; 4404 4405 new_ns = alloc_mnt_ns(user_ns, false); 4406 if (IS_ERR(new_ns)) 4407 return new_ns; 4408 4409 guard(namespace_excl)(); 4410 4411 if (flags & CLONE_EMPTY_MNTNS) 4412 copy_flags = 0; 4413 else 4414 copy_flags = CL_COPY_UNBINDABLE | CL_EXPIRE; 4415 if (user_ns != ns->user_ns) 4416 copy_flags |= CL_SLAVE; 4417 4418 if (flags & CLONE_EMPTY_MNTNS) 4419 new = clone_mnt(old, old->mnt.mnt_root, copy_flags); 4420 else 4421 new = copy_tree(old, old->mnt.mnt_root, copy_flags); 4422 if (IS_ERR(new)) { 4423 emptied_ns = new_ns; 4424 return ERR_CAST(new); 4425 } 4426 if (user_ns != ns->user_ns) { 4427 guard(mount_writer)(); 4428 lock_mnt_tree(new); 4429 } 4430 new_ns->root = new; 4431 4432 if (flags & CLONE_EMPTY_MNTNS) { 4433 /* 4434 * Empty mount namespace: only the root mount exists. 4435 * Reset root and pwd to the cloned mount's root dentry. 4436 */ 4437 if (new_fs) { 4438 old_root = new_fs->root; 4439 old_pwd = new_fs->pwd; 4440 4441 new_fs->root.mnt = mntget(&new->mnt); 4442 new_fs->root.dentry = dget(new->mnt.mnt_root); 4443 4444 new_fs->pwd.mnt = mntget(&new->mnt); 4445 new_fs->pwd.dentry = dget(new->mnt.mnt_root); 4446 } 4447 mnt_add_to_ns(new_ns, new); 4448 new_ns->nr_mounts++; 4449 } else { 4450 /* 4451 * Full copy: walk old and new trees in parallel, switching 4452 * the tsk->fs->* elements and marking new vfsmounts as 4453 * belonging to new namespace. We have already acquired a 4454 * private fs_struct, so tsk->fs->lock is not needed. 4455 */ 4456 p = old; 4457 q = new; 4458 while (p) { 4459 mnt_add_to_ns(new_ns, q); 4460 new_ns->nr_mounts++; 4461 if (new_fs) { 4462 if (&p->mnt == new_fs->root.mnt) { 4463 old_root.mnt = new_fs->root.mnt; 4464 new_fs->root.mnt = mntget(&q->mnt); 4465 } 4466 if (&p->mnt == new_fs->pwd.mnt) { 4467 old_pwd.mnt = new_fs->pwd.mnt; 4468 new_fs->pwd.mnt = mntget(&q->mnt); 4469 } 4470 } 4471 p = next_mnt(p, old); 4472 q = next_mnt(q, new); 4473 if (!q) 4474 break; 4475 // an mntns binding we'd skipped? 4476 while (p->mnt.mnt_root != q->mnt.mnt_root) 4477 p = next_mnt(skip_mnt_tree(p), old); 4478 } 4479 } 4480 ns_tree_add_raw(new_ns); 4481 return new_ns; 4482 } 4483 4484 struct dentry *mount_subtree(struct vfsmount *m, const char *name) 4485 { 4486 struct mount *mnt = real_mount(m); 4487 struct mnt_namespace *ns; 4488 struct super_block *s; 4489 struct path path; 4490 int err; 4491 4492 ns = alloc_mnt_ns(&init_user_ns, true); 4493 if (IS_ERR(ns)) { 4494 mntput(m); 4495 return ERR_CAST(ns); 4496 } 4497 ns->root = mnt; 4498 ns->nr_mounts++; 4499 mnt_add_to_ns(ns, mnt); 4500 4501 err = vfs_path_lookup(m->mnt_root, m, 4502 name, LOOKUP_FOLLOW|LOOKUP_AUTOMOUNT, &path); 4503 4504 put_mnt_ns(ns); 4505 4506 if (err) 4507 return ERR_PTR(err); 4508 4509 /* trade a vfsmount reference for active sb one */ 4510 s = path.mnt->mnt_sb; 4511 atomic_inc(&s->s_active); 4512 mntput(path.mnt); 4513 /* lock the sucker */ 4514 down_write(&s->s_umount); 4515 /* ... and return the root of (sub)tree on it */ 4516 return path.dentry; 4517 } 4518 EXPORT_SYMBOL(mount_subtree); 4519 4520 SYSCALL_DEFINE5(mount, char __user *, dev_name, char __user *, dir_name, 4521 char __user *, type, unsigned long, flags, void __user *, data) 4522 { 4523 int ret; 4524 char *kernel_type; 4525 char *kernel_dev; 4526 void *options; 4527 4528 kernel_type = copy_mount_string(type); 4529 ret = PTR_ERR(kernel_type); 4530 if (IS_ERR(kernel_type)) 4531 goto out_type; 4532 4533 kernel_dev = copy_mount_string(dev_name); 4534 ret = PTR_ERR(kernel_dev); 4535 if (IS_ERR(kernel_dev)) 4536 goto out_dev; 4537 4538 options = copy_mount_options(data); 4539 ret = PTR_ERR(options); 4540 if (IS_ERR(options)) 4541 goto out_data; 4542 4543 ret = do_mount(kernel_dev, dir_name, kernel_type, flags, options); 4544 4545 kfree(options); 4546 out_data: 4547 kfree(kernel_dev); 4548 out_dev: 4549 kfree(kernel_type); 4550 out_type: 4551 return ret; 4552 } 4553 4554 #define FSMOUNT_VALID_FLAGS \ 4555 (MOUNT_ATTR_RDONLY | MOUNT_ATTR_NOSUID | MOUNT_ATTR_NODEV | \ 4556 MOUNT_ATTR_NOEXEC | MOUNT_ATTR__ATIME | MOUNT_ATTR_NODIRATIME | \ 4557 MOUNT_ATTR_NOSYMFOLLOW) 4558 4559 #define MOUNT_SETATTR_VALID_FLAGS (FSMOUNT_VALID_FLAGS | MOUNT_ATTR_IDMAP) 4560 4561 #define MOUNT_SETATTR_PROPAGATION_FLAGS \ 4562 (MS_UNBINDABLE | MS_PRIVATE | MS_SLAVE | MS_SHARED) 4563 4564 static unsigned int attr_flags_to_mnt_flags(u64 attr_flags) 4565 { 4566 unsigned int mnt_flags = 0; 4567 4568 if (attr_flags & MOUNT_ATTR_RDONLY) 4569 mnt_flags |= MNT_READONLY; 4570 if (attr_flags & MOUNT_ATTR_NOSUID) 4571 mnt_flags |= MNT_NOSUID; 4572 if (attr_flags & MOUNT_ATTR_NODEV) 4573 mnt_flags |= MNT_NODEV; 4574 if (attr_flags & MOUNT_ATTR_NOEXEC) 4575 mnt_flags |= MNT_NOEXEC; 4576 if (attr_flags & MOUNT_ATTR_NODIRATIME) 4577 mnt_flags |= MNT_NODIRATIME; 4578 if (attr_flags & MOUNT_ATTR_NOSYMFOLLOW) 4579 mnt_flags |= MNT_NOSYMFOLLOW; 4580 4581 return mnt_flags; 4582 } 4583 4584 /* 4585 * Create a kernel mount representation for a new, prepared superblock 4586 * (specified by fs_fd) and attach to an open_tree-like file descriptor. 4587 */ 4588 SYSCALL_DEFINE3(fsmount, int, fs_fd, unsigned int, flags, 4589 unsigned int, attr_flags) 4590 { 4591 struct path new_path __free(path_put) = {}; 4592 struct mnt_namespace *ns; 4593 struct fs_context *fc; 4594 struct vfsmount *new_mnt; 4595 struct mount *mnt; 4596 unsigned int mnt_flags = 0; 4597 long ret; 4598 4599 if ((flags & ~(FSMOUNT_CLOEXEC | FSMOUNT_NAMESPACE)) != 0) 4600 return -EINVAL; 4601 4602 if ((flags & FSMOUNT_NAMESPACE) && 4603 !ns_capable(current_user_ns(), CAP_SYS_ADMIN)) 4604 return -EPERM; 4605 4606 if (!(flags & FSMOUNT_NAMESPACE) && !may_mount()) 4607 return -EPERM; 4608 4609 if (attr_flags & ~FSMOUNT_VALID_FLAGS) 4610 return -EINVAL; 4611 4612 mnt_flags = attr_flags_to_mnt_flags(attr_flags); 4613 4614 switch (attr_flags & MOUNT_ATTR__ATIME) { 4615 case MOUNT_ATTR_STRICTATIME: 4616 break; 4617 case MOUNT_ATTR_NOATIME: 4618 mnt_flags |= MNT_NOATIME; 4619 break; 4620 case MOUNT_ATTR_RELATIME: 4621 mnt_flags |= MNT_RELATIME; 4622 break; 4623 default: 4624 return -EINVAL; 4625 } 4626 4627 CLASS(fd, f)(fs_fd); 4628 if (fd_empty(f)) 4629 return -EBADF; 4630 4631 if (fd_file(f)->f_op != &fscontext_fops) 4632 return -EINVAL; 4633 4634 fc = fd_file(f)->private_data; 4635 4636 ACQUIRE(mutex_intr, uapi_mutex)(&fc->uapi_mutex); 4637 ret = ACQUIRE_ERR(mutex_intr, &uapi_mutex); 4638 if (ret) 4639 return ret; 4640 4641 /* There must be a valid superblock or we can't mount it */ 4642 ret = -EINVAL; 4643 if (!fc->root) 4644 return ret; 4645 4646 ret = -EPERM; 4647 if (mount_too_revealing(fc->root->d_sb, &mnt_flags)) { 4648 errorfcp(fc, "VFS", "Mount too revealing"); 4649 return ret; 4650 } 4651 4652 ret = -EBUSY; 4653 if (fc->phase != FS_CONTEXT_AWAITING_MOUNT) 4654 return ret; 4655 4656 if (fc->sb_flags & SB_MANDLOCK) 4657 warn_mandlock(); 4658 4659 new_mnt = vfs_create_mount(fc); 4660 if (IS_ERR(new_mnt)) 4661 return PTR_ERR(new_mnt); 4662 if (new_mnt->mnt_sb->s_flags & SB_NOUSER) { 4663 mntput(new_mnt); 4664 return -EINVAL; 4665 } 4666 new_mnt->mnt_flags = mnt_flags; 4667 4668 new_path.dentry = dget(fc->root); 4669 new_path.mnt = new_mnt; 4670 4671 /* We've done the mount bit - now move the file context into more or 4672 * less the same state as if we'd done an fspick(). We don't want to 4673 * do any memory allocation or anything like that at this point as we 4674 * don't want to have to handle any errors incurred. 4675 */ 4676 vfs_clean_context(fc); 4677 4678 if (flags & FSMOUNT_NAMESPACE) 4679 return FD_ADD((flags & FSMOUNT_CLOEXEC) ? O_CLOEXEC : 0, 4680 open_new_namespace(&new_path, MOUNT_COPY_NEW)); 4681 4682 ns = alloc_mnt_ns(current->nsproxy->mnt_ns->user_ns, true); 4683 if (IS_ERR(ns)) 4684 return PTR_ERR(ns); 4685 mnt = real_mount(new_path.mnt); 4686 ns->root = mnt; 4687 ns->nr_mounts = 1; 4688 mnt_add_to_ns(ns, mnt); 4689 mntget(new_path.mnt); 4690 4691 FD_PREPARE(fdf, (flags & FSMOUNT_CLOEXEC) ? O_CLOEXEC : 0, 4692 dentry_open(&new_path, O_PATH, fc->cred)); 4693 if (fdf.err) { 4694 dissolve_on_fput(no_free_ptr(new_path.mnt)); 4695 return fdf.err; 4696 } 4697 4698 /* 4699 * Attach to an apparent O_PATH fd with a note that we 4700 * need to unmount it, not just simply put it. 4701 */ 4702 fd_prepare_file(fdf)->f_mode |= FMODE_NEED_UNMOUNT; 4703 return fd_publish(fdf); 4704 } 4705 4706 static inline int vfs_move_mount(const struct path *from_path, 4707 const struct path *to_path, 4708 enum mnt_tree_flags_t mflags) 4709 { 4710 int ret; 4711 4712 ret = security_move_mount(from_path, to_path); 4713 if (ret) 4714 return ret; 4715 4716 if (mflags & MNT_TREE_PROPAGATION) 4717 return do_set_group(from_path, to_path); 4718 4719 return do_move_mount(from_path, to_path, mflags); 4720 } 4721 4722 /* 4723 * Move a mount from one place to another. In combination with 4724 * fsopen()/fsmount() this is used to install a new mount and in combination 4725 * with open_tree(OPEN_TREE_CLONE [| AT_RECURSIVE]) it can be used to copy 4726 * a mount subtree. 4727 * 4728 * Note the flags value is a combination of MOVE_MOUNT_* flags. 4729 */ 4730 SYSCALL_DEFINE5(move_mount, 4731 int, from_dfd, const char __user *, from_pathname, 4732 int, to_dfd, const char __user *, to_pathname, 4733 unsigned int, flags) 4734 { 4735 struct path to_path __free(path_put) = {}; 4736 struct path from_path __free(path_put) = {}; 4737 unsigned int lflags, uflags; 4738 enum mnt_tree_flags_t mflags = 0; 4739 int ret = 0; 4740 4741 if (!may_mount()) 4742 return -EPERM; 4743 4744 if (flags & ~MOVE_MOUNT__MASK) 4745 return -EINVAL; 4746 4747 if ((flags & (MOVE_MOUNT_BENEATH | MOVE_MOUNT_SET_GROUP)) == 4748 (MOVE_MOUNT_BENEATH | MOVE_MOUNT_SET_GROUP)) 4749 return -EINVAL; 4750 4751 if (flags & MOVE_MOUNT_SET_GROUP) mflags |= MNT_TREE_PROPAGATION; 4752 if (flags & MOVE_MOUNT_BENEATH) mflags |= MNT_TREE_BENEATH; 4753 4754 uflags = 0; 4755 if (flags & MOVE_MOUNT_T_EMPTY_PATH) 4756 uflags = AT_EMPTY_PATH; 4757 4758 CLASS(filename_maybe_null,to_name)(to_pathname, uflags); 4759 if (!to_name && to_dfd >= 0) { 4760 CLASS(fd_raw, f_to)(to_dfd); 4761 if (fd_empty(f_to)) 4762 return -EBADF; 4763 4764 to_path = fd_file(f_to)->f_path; 4765 path_get(&to_path); 4766 } else { 4767 lflags = 0; 4768 if (flags & MOVE_MOUNT_T_SYMLINKS) 4769 lflags |= LOOKUP_FOLLOW; 4770 if (flags & MOVE_MOUNT_T_AUTOMOUNTS) 4771 lflags |= LOOKUP_AUTOMOUNT; 4772 ret = filename_lookup(to_dfd, to_name, lflags, &to_path, NULL); 4773 if (ret) 4774 return ret; 4775 } 4776 4777 uflags = 0; 4778 if (flags & MOVE_MOUNT_F_EMPTY_PATH) 4779 uflags = AT_EMPTY_PATH; 4780 4781 CLASS(filename_maybe_null,from_name)(from_pathname, uflags); 4782 if (!from_name && from_dfd >= 0) { 4783 CLASS(fd_raw, f_from)(from_dfd); 4784 if (fd_empty(f_from)) 4785 return -EBADF; 4786 4787 return vfs_move_mount(&fd_file(f_from)->f_path, &to_path, mflags); 4788 } 4789 4790 lflags = 0; 4791 if (flags & MOVE_MOUNT_F_SYMLINKS) 4792 lflags |= LOOKUP_FOLLOW; 4793 if (flags & MOVE_MOUNT_F_AUTOMOUNTS) 4794 lflags |= LOOKUP_AUTOMOUNT; 4795 ret = filename_lookup(from_dfd, from_name, lflags, &from_path, NULL); 4796 if (ret) 4797 return ret; 4798 4799 return vfs_move_mount(&from_path, &to_path, mflags); 4800 } 4801 4802 /* 4803 * Return true if path is reachable from root 4804 * 4805 * locks: mount_locked_reader || namespace_shared && is_mounted(mnt) 4806 */ 4807 bool is_path_reachable(struct mount *mnt, struct dentry *dentry, 4808 const struct path *root) 4809 { 4810 while (&mnt->mnt != root->mnt && mnt_has_parent(mnt)) { 4811 dentry = mnt->mnt_mountpoint; 4812 mnt = mnt->mnt_parent; 4813 } 4814 return &mnt->mnt == root->mnt && is_subdir(dentry, root->dentry); 4815 } 4816 4817 bool path_is_under(const struct path *path1, const struct path *path2) 4818 { 4819 guard(mount_locked_reader)(); 4820 return is_path_reachable(real_mount(path1->mnt), path1->dentry, path2); 4821 } 4822 EXPORT_SYMBOL(path_is_under); 4823 4824 int path_pivot_root(struct path *new, struct path *old) 4825 { 4826 struct path root __free(path_put) = {}; 4827 struct mount *new_mnt, *root_mnt, *old_mnt, *root_parent, *ex_parent; 4828 int error; 4829 4830 if (!may_mount()) 4831 return -EPERM; 4832 4833 error = security_sb_pivotroot(old, new); 4834 if (error) 4835 return error; 4836 4837 get_fs_root(current->fs, &root); 4838 4839 LOCK_MOUNT(old_mp, old); 4840 old_mnt = old_mp.parent; 4841 if (IS_ERR(old_mnt)) 4842 return PTR_ERR(old_mnt); 4843 4844 new_mnt = real_mount(new->mnt); 4845 root_mnt = real_mount(root.mnt); 4846 /* only a mounted mount has a parent that namespace_sem pins */ 4847 if (!check_mnt(root_mnt) || !check_mnt(new_mnt)) 4848 return -EINVAL; 4849 ex_parent = new_mnt->mnt_parent; 4850 root_parent = root_mnt->mnt_parent; 4851 if (IS_MNT_SHARED(old_mnt) || 4852 IS_MNT_SHARED(ex_parent) || 4853 IS_MNT_SHARED(root_parent)) 4854 return -EINVAL; 4855 if (new_mnt->mnt.mnt_flags & MNT_LOCKED) 4856 return -EINVAL; 4857 if (d_unlinked(new->dentry)) 4858 return -ENOENT; 4859 if (new_mnt == root_mnt || old_mnt == root_mnt) 4860 return -EBUSY; /* loop, on the same file system */ 4861 if (!path_mounted(&root)) 4862 return -EINVAL; /* not a mountpoint */ 4863 if (!mnt_has_parent(root_mnt)) 4864 return -EINVAL; /* absolute root */ 4865 if (!path_mounted(new)) 4866 return -EINVAL; /* not a mountpoint */ 4867 if (!mnt_has_parent(new_mnt)) 4868 return -EINVAL; /* absolute root */ 4869 /* make sure we can reach put_old from new_root */ 4870 if (!is_path_reachable(old_mnt, old_mp.mp->m_dentry, new)) 4871 return -EINVAL; 4872 /* make certain new is below the root */ 4873 if (!is_path_reachable(new_mnt, new->dentry, &root)) 4874 return -EINVAL; 4875 lock_mount_hash(); 4876 umount_mnt(new_mnt); 4877 if (root_mnt->mnt.mnt_flags & MNT_LOCKED) { 4878 new_mnt->mnt.mnt_flags |= MNT_LOCKED; 4879 root_mnt->mnt.mnt_flags &= ~MNT_LOCKED; 4880 } 4881 /* mount new_root on / */ 4882 attach_mnt(new_mnt, root_parent, root_mnt->mnt_mp); 4883 umount_mnt(root_mnt); 4884 /* mount old root on put_old */ 4885 attach_mnt(root_mnt, old_mnt, old_mp.mp); 4886 touch_mnt_namespace(current->nsproxy->mnt_ns); 4887 /* A moved mount should not expire automatically */ 4888 list_del_init(&new_mnt->mnt_expire); 4889 unlock_mount_hash(); 4890 mnt_notify_add(root_mnt); 4891 mnt_notify_add(new_mnt); 4892 chroot_fs_refs(&root, new); 4893 return 0; 4894 } 4895 4896 /* 4897 * pivot_root Semantics: 4898 * Moves the root file system of the current process to the directory put_old, 4899 * makes new_root as the new root file system of the current process, and sets 4900 * root/cwd of all processes which had them on the current root to new_root. 4901 * 4902 * Restrictions: 4903 * The new_root and put_old must be directories, and must not be on the 4904 * same file system as the current process root. The put_old must be 4905 * underneath new_root, i.e. adding a non-zero number of /.. to the string 4906 * pointed to by put_old must yield the same directory as new_root. No other 4907 * file system may be mounted on put_old. After all, new_root is a mountpoint. 4908 * 4909 * The immutable nullfs filesystem is mounted as the true root of the VFS 4910 * hierarchy. The mutable rootfs (tmpfs/ramfs) is layered on top of this, 4911 * allowing pivot_root() to work normally from initramfs. 4912 * 4913 * Notes: 4914 * - we don't move root/cwd if they are not at the root (reason: if something 4915 * cared enough to change them, it's probably wrong to force them elsewhere) 4916 * - it's okay to pick a root that isn't the root of a file system, e.g. 4917 * /nfs/my_root where /nfs is the mount point. It must be a mountpoint, 4918 * though, so you may need to say mount --bind /nfs/my_root /nfs/my_root 4919 * first. 4920 */ 4921 SYSCALL_DEFINE2(pivot_root, const char __user *, new_root, 4922 const char __user *, put_old) 4923 { 4924 struct path new __free(path_put) = {}; 4925 struct path old __free(path_put) = {}; 4926 int error; 4927 4928 error = user_path_at(AT_FDCWD, new_root, 4929 LOOKUP_FOLLOW | LOOKUP_DIRECTORY, &new); 4930 if (error) 4931 return error; 4932 4933 error = user_path_at(AT_FDCWD, put_old, 4934 LOOKUP_FOLLOW | LOOKUP_DIRECTORY, &old); 4935 if (error) 4936 return error; 4937 4938 return path_pivot_root(&new, &old); 4939 } 4940 4941 static unsigned int recalc_flags(struct mount_kattr *kattr, struct mount *mnt) 4942 { 4943 unsigned int flags = mnt->mnt.mnt_flags; 4944 4945 /* flags to clear */ 4946 flags &= ~kattr->attr_clr; 4947 /* flags to raise */ 4948 flags |= kattr->attr_set; 4949 4950 return flags; 4951 } 4952 4953 static int can_idmap_mount(const struct mount_kattr *kattr, struct mount *mnt) 4954 { 4955 struct vfsmount *m = &mnt->mnt; 4956 struct user_namespace *fs_userns = m->mnt_sb->s_user_ns; 4957 4958 if (!kattr->mnt_idmap) 4959 return 0; 4960 4961 /* 4962 * Creating an idmapped mount with the filesystem wide idmapping 4963 * doesn't make sense so block that. We don't allow mushy semantics. 4964 */ 4965 if (kattr->mnt_userns == m->mnt_sb->s_user_ns) 4966 return -EINVAL; 4967 4968 /* 4969 * We only allow an mount to change it's idmapping if it has 4970 * never been accessible to userspace. 4971 */ 4972 if (!(kattr->kflags & MOUNT_KATTR_IDMAP_REPLACE) && is_idmapped_mnt(m)) 4973 return -EPERM; 4974 4975 /* The underlying filesystem doesn't support idmapped mounts yet. */ 4976 if (!(m->mnt_sb->s_type->fs_flags & FS_ALLOW_IDMAP)) 4977 return -EINVAL; 4978 4979 /* The filesystem has turned off idmapped mounts. */ 4980 if (m->mnt_sb->s_iflags & SB_I_NOIDMAP) 4981 return -EINVAL; 4982 4983 /* We're not controlling the superblock. */ 4984 if (!ns_capable(fs_userns, CAP_SYS_ADMIN)) 4985 return -EPERM; 4986 4987 /* Mount has already been visible in the filesystem hierarchy. */ 4988 if (!is_anon_ns(mnt->mnt_ns)) 4989 return -EINVAL; 4990 4991 return 0; 4992 } 4993 4994 /** 4995 * mnt_allow_writers() - check whether the attribute change allows writers 4996 * @kattr: the new mount attributes 4997 * @mnt: the mount to which @kattr will be applied 4998 * 4999 * Check whether thew new mount attributes in @kattr allow concurrent writers. 5000 * 5001 * Return: true if writers need to be held, false if not 5002 */ 5003 static inline bool mnt_allow_writers(const struct mount_kattr *kattr, 5004 const struct mount *mnt) 5005 { 5006 return (!(kattr->attr_set & MNT_READONLY) || 5007 (mnt->mnt.mnt_flags & MNT_READONLY)) && 5008 !kattr->mnt_idmap; 5009 } 5010 5011 static int mount_setattr_prepare(struct mount_kattr *kattr, struct mount *mnt) 5012 { 5013 struct mount *m; 5014 int err; 5015 5016 for (m = mnt; m; m = next_mnt(m, mnt)) { 5017 if (!can_change_locked_flags(m, recalc_flags(kattr, m))) { 5018 err = -EPERM; 5019 break; 5020 } 5021 5022 err = can_idmap_mount(kattr, m); 5023 if (err) 5024 break; 5025 5026 if (!mnt_allow_writers(kattr, m)) { 5027 err = mnt_hold_writers(m); 5028 if (err) { 5029 m = next_mnt(m, mnt); 5030 break; 5031 } 5032 } 5033 5034 if (!(kattr->kflags & MOUNT_KATTR_RECURSE)) 5035 return 0; 5036 } 5037 5038 if (err) { 5039 /* undo all mnt_hold_writers() we'd done */ 5040 for (struct mount *p = mnt; p != m; p = next_mnt(p, mnt)) 5041 mnt_unhold_writers(p); 5042 } 5043 return err; 5044 } 5045 5046 static void do_idmap_mount(const struct mount_kattr *kattr, struct mount *mnt) 5047 { 5048 struct mnt_idmap *old_idmap; 5049 5050 if (!kattr->mnt_idmap) 5051 return; 5052 5053 old_idmap = mnt_idmap(&mnt->mnt); 5054 5055 /* Pairs with smp_load_acquire() in mnt_idmap(). */ 5056 smp_store_release(&mnt->mnt.mnt_idmap, mnt_idmap_get(kattr->mnt_idmap)); 5057 mnt_idmap_put(old_idmap); 5058 } 5059 5060 static void mount_setattr_commit(struct mount_kattr *kattr, struct mount *mnt) 5061 { 5062 struct mount *m; 5063 5064 for (m = mnt; m; m = next_mnt(m, mnt)) { 5065 unsigned int flags; 5066 5067 do_idmap_mount(kattr, m); 5068 flags = recalc_flags(kattr, m); 5069 WRITE_ONCE(m->mnt.mnt_flags, flags); 5070 5071 /* If we had to hold writers unblock them. */ 5072 mnt_unhold_writers(m); 5073 5074 if (kattr->propagation) 5075 change_mnt_propagation(m, kattr->propagation); 5076 if (!(kattr->kflags & MOUNT_KATTR_RECURSE)) 5077 break; 5078 } 5079 touch_mnt_namespace(mnt->mnt_ns); 5080 } 5081 5082 static int do_mount_setattr(const struct path *path, struct mount_kattr *kattr) 5083 { 5084 struct mount *mnt = real_mount(path->mnt); 5085 int err = 0; 5086 5087 if (!path_mounted(path)) 5088 return -EINVAL; 5089 5090 if (kattr->mnt_userns) { 5091 struct mnt_idmap *mnt_idmap; 5092 5093 mnt_idmap = alloc_mnt_idmap(kattr->mnt_userns); 5094 if (IS_ERR(mnt_idmap)) 5095 return PTR_ERR(mnt_idmap); 5096 kattr->mnt_idmap = mnt_idmap; 5097 } 5098 5099 if (kattr->propagation) { 5100 /* 5101 * Only take namespace_lock() if we're actually changing 5102 * propagation. 5103 */ 5104 namespace_lock(); 5105 /* invent_group_ids() walks the tree, only walk a mounted one */ 5106 if (!anon_ns_root(mnt) && !check_mnt(mnt)) { 5107 namespace_unlock(); 5108 return -EINVAL; 5109 } 5110 if (kattr->propagation == MS_SHARED) { 5111 err = invent_group_ids(mnt, kattr->kflags & MOUNT_KATTR_RECURSE); 5112 if (err) { 5113 namespace_unlock(); 5114 return err; 5115 } 5116 } 5117 } 5118 5119 err = -EINVAL; 5120 lock_mount_hash(); 5121 5122 /* Checked under namespace_sem already if the propagation changes. */ 5123 if (!anon_ns_root(mnt) && !check_mnt(mnt)) 5124 goto out; 5125 5126 /* 5127 * First, we get the mount tree in a shape where we can change mount 5128 * properties without failure. If we succeeded to do so we commit all 5129 * changes and if we failed we clean up. 5130 */ 5131 err = mount_setattr_prepare(kattr, mnt); 5132 if (!err) 5133 mount_setattr_commit(kattr, mnt); 5134 5135 out: 5136 unlock_mount_hash(); 5137 5138 if (kattr->propagation) { 5139 if (err) 5140 cleanup_group_ids(mnt, NULL); 5141 namespace_unlock(); 5142 } 5143 5144 return err; 5145 } 5146 5147 static int build_mount_idmapped(const struct mount_attr *attr, size_t usize, 5148 struct mount_kattr *kattr) 5149 { 5150 struct ns_common *ns; 5151 struct user_namespace *mnt_userns; 5152 5153 if (!((attr->attr_set | attr->attr_clr) & MOUNT_ATTR_IDMAP)) 5154 return 0; 5155 5156 if (attr->attr_clr & MOUNT_ATTR_IDMAP) { 5157 /* 5158 * We can only remove an idmapping if it's never been 5159 * exposed to userspace. 5160 */ 5161 if (!(kattr->kflags & MOUNT_KATTR_IDMAP_REPLACE)) 5162 return -EINVAL; 5163 5164 /* 5165 * Removal of idmappings is equivalent to setting 5166 * nop_mnt_idmap. 5167 */ 5168 if (!(attr->attr_set & MOUNT_ATTR_IDMAP)) { 5169 kattr->mnt_idmap = &nop_mnt_idmap; 5170 return 0; 5171 } 5172 } 5173 5174 if (attr->userns_fd > INT_MAX) 5175 return -EINVAL; 5176 5177 CLASS(fd, f)(attr->userns_fd); 5178 if (fd_empty(f)) 5179 return -EBADF; 5180 5181 if (!proc_ns_file(fd_file(f))) 5182 return -EINVAL; 5183 5184 ns = get_proc_ns(file_inode(fd_file(f))); 5185 if (ns->ns_type != CLONE_NEWUSER) 5186 return -EINVAL; 5187 5188 /* 5189 * The initial idmapping cannot be used to create an idmapped 5190 * mount. We use the initial idmapping as an indicator of a mount 5191 * that is not idmapped. It can simply be passed into helpers that 5192 * are aware of idmapped mounts as a convenient shortcut. A user 5193 * can just create a dedicated identity mapping to achieve the same 5194 * result. 5195 */ 5196 mnt_userns = container_of(ns, struct user_namespace, ns); 5197 if (mnt_userns == &init_user_ns) 5198 return -EPERM; 5199 5200 /* We're not controlling the target namespace. */ 5201 if (!ns_capable(mnt_userns, CAP_SYS_ADMIN)) 5202 return -EPERM; 5203 5204 kattr->mnt_userns = get_user_ns(mnt_userns); 5205 return 0; 5206 } 5207 5208 static int build_mount_kattr(const struct mount_attr *attr, size_t usize, 5209 struct mount_kattr *kattr) 5210 { 5211 if (attr->propagation & ~MOUNT_SETATTR_PROPAGATION_FLAGS) 5212 return -EINVAL; 5213 if (hweight32(attr->propagation & MOUNT_SETATTR_PROPAGATION_FLAGS) > 1) 5214 return -EINVAL; 5215 kattr->propagation = attr->propagation; 5216 5217 if ((attr->attr_set | attr->attr_clr) & ~MOUNT_SETATTR_VALID_FLAGS) 5218 return -EINVAL; 5219 5220 kattr->attr_set = attr_flags_to_mnt_flags(attr->attr_set); 5221 kattr->attr_clr = attr_flags_to_mnt_flags(attr->attr_clr); 5222 5223 /* 5224 * Since the MOUNT_ATTR_<atime> values are an enum, not a bitmap, 5225 * users wanting to transition to a different atime setting cannot 5226 * simply specify the atime setting in @attr_set, but must also 5227 * specify MOUNT_ATTR__ATIME in the @attr_clr field. 5228 * So ensure that MOUNT_ATTR__ATIME can't be partially set in 5229 * @attr_clr and that @attr_set can't have any atime bits set if 5230 * MOUNT_ATTR__ATIME isn't set in @attr_clr. 5231 */ 5232 if (attr->attr_clr & MOUNT_ATTR__ATIME) { 5233 if ((attr->attr_clr & MOUNT_ATTR__ATIME) != MOUNT_ATTR__ATIME) 5234 return -EINVAL; 5235 5236 /* 5237 * Clear all previous time settings as they are mutually 5238 * exclusive. 5239 */ 5240 kattr->attr_clr |= MNT_RELATIME | MNT_NOATIME; 5241 switch (attr->attr_set & MOUNT_ATTR__ATIME) { 5242 case MOUNT_ATTR_RELATIME: 5243 kattr->attr_set |= MNT_RELATIME; 5244 break; 5245 case MOUNT_ATTR_NOATIME: 5246 kattr->attr_set |= MNT_NOATIME; 5247 break; 5248 case MOUNT_ATTR_STRICTATIME: 5249 break; 5250 default: 5251 return -EINVAL; 5252 } 5253 } else { 5254 if (attr->attr_set & MOUNT_ATTR__ATIME) 5255 return -EINVAL; 5256 } 5257 5258 return build_mount_idmapped(attr, usize, kattr); 5259 } 5260 5261 static void finish_mount_kattr(struct mount_kattr *kattr) 5262 { 5263 if (kattr->mnt_userns) { 5264 put_user_ns(kattr->mnt_userns); 5265 kattr->mnt_userns = NULL; 5266 } 5267 5268 if (kattr->mnt_idmap) 5269 mnt_idmap_put(kattr->mnt_idmap); 5270 } 5271 5272 static int wants_mount_setattr(struct mount_attr __user *uattr, size_t usize, 5273 struct mount_kattr *kattr) 5274 { 5275 int ret; 5276 struct mount_attr attr; 5277 5278 BUILD_BUG_ON(sizeof(struct mount_attr) != MOUNT_ATTR_SIZE_VER0); 5279 5280 if (unlikely(usize > PAGE_SIZE)) 5281 return -E2BIG; 5282 if (unlikely(usize < MOUNT_ATTR_SIZE_VER0)) 5283 return -EINVAL; 5284 5285 if (!may_mount()) 5286 return -EPERM; 5287 5288 ret = copy_struct_from_user(&attr, sizeof(attr), uattr, usize); 5289 if (ret) 5290 return ret; 5291 5292 /* Don't bother walking through the mounts if this is a nop. */ 5293 if (attr.attr_set == 0 && 5294 attr.attr_clr == 0 && 5295 attr.propagation == 0) 5296 return 0; /* Tell caller to not bother. */ 5297 5298 ret = build_mount_kattr(&attr, usize, kattr); 5299 if (ret < 0) 5300 return ret; 5301 5302 return 1; 5303 } 5304 5305 SYSCALL_DEFINE5(mount_setattr, int, dfd, const char __user *, path, 5306 unsigned int, flags, struct mount_attr __user *, uattr, 5307 size_t, usize) 5308 { 5309 int err; 5310 struct path target; 5311 struct mount_kattr kattr; 5312 unsigned int lookup_flags = LOOKUP_AUTOMOUNT | LOOKUP_FOLLOW; 5313 5314 if (flags & ~(AT_EMPTY_PATH | 5315 AT_RECURSIVE | 5316 AT_SYMLINK_NOFOLLOW | 5317 AT_NO_AUTOMOUNT)) 5318 return -EINVAL; 5319 5320 if (flags & AT_NO_AUTOMOUNT) 5321 lookup_flags &= ~LOOKUP_AUTOMOUNT; 5322 if (flags & AT_SYMLINK_NOFOLLOW) 5323 lookup_flags &= ~LOOKUP_FOLLOW; 5324 5325 kattr = (struct mount_kattr) { 5326 .lookup_flags = lookup_flags, 5327 }; 5328 5329 if (flags & AT_RECURSIVE) 5330 kattr.kflags |= MOUNT_KATTR_RECURSE; 5331 5332 err = wants_mount_setattr(uattr, usize, &kattr); 5333 if (err <= 0) 5334 return err; 5335 5336 CLASS(filename_uflags, name)(path, flags); 5337 err = filename_lookup(dfd, name, kattr.lookup_flags, &target, NULL); 5338 if (!err) { 5339 err = do_mount_setattr(&target, &kattr); 5340 path_put(&target); 5341 } 5342 finish_mount_kattr(&kattr); 5343 return err; 5344 } 5345 5346 SYSCALL_DEFINE5(open_tree_attr, int, dfd, const char __user *, filename, 5347 unsigned, flags, struct mount_attr __user *, uattr, 5348 size_t, usize) 5349 { 5350 if (!uattr && usize) 5351 return -EINVAL; 5352 5353 FD_PREPARE(fdf, flags, vfs_open_tree(dfd, filename, flags)); 5354 if (fdf.err) 5355 return fdf.err; 5356 5357 if (uattr) { 5358 struct mount_kattr kattr = {}; 5359 struct file *file = fd_prepare_file(fdf); 5360 int ret; 5361 5362 if (flags & OPEN_TREE_CLONE) 5363 kattr.kflags = MOUNT_KATTR_IDMAP_REPLACE; 5364 if (flags & AT_RECURSIVE) 5365 kattr.kflags |= MOUNT_KATTR_RECURSE; 5366 5367 ret = wants_mount_setattr(uattr, usize, &kattr); 5368 if (ret > 0) { 5369 ret = do_mount_setattr(&file->f_path, &kattr); 5370 finish_mount_kattr(&kattr); 5371 } 5372 if (ret) 5373 return ret; 5374 } 5375 5376 return fd_publish(fdf); 5377 } 5378 5379 int show_path(struct seq_file *m, struct dentry *root) 5380 { 5381 if (root->d_sb->s_op->show_path) 5382 return root->d_sb->s_op->show_path(m, root); 5383 5384 seq_dentry(m, root, " \t\n\\"); 5385 return 0; 5386 } 5387 5388 static struct vfsmount *lookup_mnt_in_ns(u64 id, struct mnt_namespace *ns) 5389 { 5390 struct mount *mnt = mnt_find_id_at(ns, id); 5391 5392 if (!mnt || mnt->mnt_id_unique != id) 5393 return NULL; 5394 5395 return &mnt->mnt; 5396 } 5397 5398 struct kstatmount { 5399 struct statmount __user *buf; 5400 size_t bufsize; 5401 struct vfsmount *mnt; 5402 struct mnt_idmap *idmap; 5403 u64 mask; 5404 struct path root; 5405 struct seq_file seq; 5406 5407 /* Must be last --ends in a flexible-array member. */ 5408 struct statmount sm; 5409 }; 5410 5411 static u64 mnt_to_attr_flags(struct vfsmount *mnt) 5412 { 5413 unsigned int mnt_flags = READ_ONCE(mnt->mnt_flags); 5414 u64 attr_flags = 0; 5415 5416 if (mnt_flags & MNT_READONLY) 5417 attr_flags |= MOUNT_ATTR_RDONLY; 5418 if (mnt_flags & MNT_NOSUID) 5419 attr_flags |= MOUNT_ATTR_NOSUID; 5420 if (mnt_flags & MNT_NODEV) 5421 attr_flags |= MOUNT_ATTR_NODEV; 5422 if (mnt_flags & MNT_NOEXEC) 5423 attr_flags |= MOUNT_ATTR_NOEXEC; 5424 if (mnt_flags & MNT_NODIRATIME) 5425 attr_flags |= MOUNT_ATTR_NODIRATIME; 5426 if (mnt_flags & MNT_NOSYMFOLLOW) 5427 attr_flags |= MOUNT_ATTR_NOSYMFOLLOW; 5428 5429 if (mnt_flags & MNT_NOATIME) 5430 attr_flags |= MOUNT_ATTR_NOATIME; 5431 else if (mnt_flags & MNT_RELATIME) 5432 attr_flags |= MOUNT_ATTR_RELATIME; 5433 else 5434 attr_flags |= MOUNT_ATTR_STRICTATIME; 5435 5436 if (is_idmapped_mnt(mnt)) 5437 attr_flags |= MOUNT_ATTR_IDMAP; 5438 5439 return attr_flags; 5440 } 5441 5442 static u64 mnt_to_propagation_flags(struct mount *m) 5443 { 5444 u64 propagation = 0; 5445 5446 if (IS_MNT_SHARED(m)) 5447 propagation |= MS_SHARED; 5448 if (IS_MNT_SLAVE(m)) 5449 propagation |= MS_SLAVE; 5450 if (IS_MNT_UNBINDABLE(m)) 5451 propagation |= MS_UNBINDABLE; 5452 if (!propagation) 5453 propagation |= MS_PRIVATE; 5454 5455 return propagation; 5456 } 5457 5458 u64 vfsmount_to_propagation_flags(struct vfsmount *mnt) 5459 { 5460 return mnt_to_propagation_flags(real_mount(mnt)); 5461 } 5462 EXPORT_SYMBOL_GPL(vfsmount_to_propagation_flags); 5463 5464 static void statmount_sb_basic(struct kstatmount *s) 5465 { 5466 struct super_block *sb = s->mnt->mnt_sb; 5467 5468 s->sm.mask |= STATMOUNT_SB_BASIC; 5469 s->sm.sb_dev_major = MAJOR(sb->s_dev); 5470 s->sm.sb_dev_minor = MINOR(sb->s_dev); 5471 s->sm.sb_magic = sb->s_magic; 5472 s->sm.sb_flags = sb->s_flags & (SB_RDONLY|SB_SYNCHRONOUS|SB_DIRSYNC|SB_LAZYTIME); 5473 } 5474 5475 static void statmount_mnt_parent(struct kstatmount *s, const struct mount *m) 5476 { 5477 s->sm.mnt_parent_id = m->mnt_parent->mnt_id_unique; 5478 s->sm.mnt_parent_id_old = m->mnt_parent->mnt_id; 5479 } 5480 5481 static void statmount_mnt_basic(struct kstatmount *s) 5482 { 5483 struct mount *m = real_mount(s->mnt); 5484 5485 s->sm.mask |= STATMOUNT_MNT_BASIC; 5486 s->sm.mnt_id = m->mnt_id_unique; 5487 s->sm.mnt_id_old = m->mnt_id; 5488 /* An unmounted mount is cut loose from its parent under mount_lock alone. */ 5489 if (likely(is_mounted(s->mnt))) 5490 statmount_mnt_parent(s, m); 5491 else 5492 scoped_guard(mount_locked_reader) 5493 statmount_mnt_parent(s, m); 5494 s->sm.mnt_attr = mnt_to_attr_flags(&m->mnt); 5495 s->sm.mnt_propagation = mnt_to_propagation_flags(m); 5496 s->sm.mnt_peer_group = m->mnt_group_id; 5497 s->sm.mnt_master = IS_MNT_SLAVE(m) ? m->mnt_master->mnt_group_id : 0; 5498 } 5499 5500 static void statmount_propagate_from(struct kstatmount *s) 5501 { 5502 struct mount *m = real_mount(s->mnt); 5503 5504 s->sm.mask |= STATMOUNT_PROPAGATE_FROM; 5505 if (IS_MNT_SLAVE(m)) 5506 s->sm.propagate_from = get_dominating_id(m, ¤t->fs->root); 5507 } 5508 5509 static int statmount_mnt_root(struct kstatmount *s, struct seq_file *seq) 5510 { 5511 int ret; 5512 size_t start = seq->count; 5513 5514 ret = show_path(seq, s->mnt->mnt_root); 5515 if (ret) 5516 return ret; 5517 5518 if (unlikely(seq_has_overflowed(seq))) 5519 return -EAGAIN; 5520 5521 /* 5522 * Unescape the result. It would be better if supplied string was not 5523 * escaped in the first place, but that's a pretty invasive change. 5524 */ 5525 seq->buf[seq->count] = '\0'; 5526 seq->count = start; 5527 seq_commit(seq, string_unescape_inplace(seq->buf + start, UNESCAPE_OCTAL)); 5528 return 0; 5529 } 5530 5531 static int statmount_mnt_point(struct kstatmount *s, struct seq_file *seq) 5532 { 5533 struct vfsmount *mnt = s->mnt; 5534 struct path mnt_path = { .dentry = mnt->mnt_root, .mnt = mnt }; 5535 int err; 5536 5537 err = seq_path_root(seq, &mnt_path, &s->root, ""); 5538 return err == SEQ_SKIP ? 0 : err; 5539 } 5540 5541 static int statmount_fs_type(struct kstatmount *s, struct seq_file *seq) 5542 { 5543 struct super_block *sb = s->mnt->mnt_sb; 5544 5545 seq_puts(seq, sb->s_type->name); 5546 return 0; 5547 } 5548 5549 static void statmount_fs_subtype(struct kstatmount *s, struct seq_file *seq) 5550 { 5551 struct super_block *sb = s->mnt->mnt_sb; 5552 5553 if (sb->s_subtype) 5554 seq_puts(seq, sb->s_subtype); 5555 } 5556 5557 static int statmount_sb_source(struct kstatmount *s, struct seq_file *seq) 5558 { 5559 struct super_block *sb = s->mnt->mnt_sb; 5560 struct mount *r = real_mount(s->mnt); 5561 5562 if (sb->s_op->show_devname) { 5563 size_t start = seq->count; 5564 int ret; 5565 5566 ret = sb->s_op->show_devname(seq, s->mnt->mnt_root); 5567 if (ret) 5568 return ret; 5569 5570 if (unlikely(seq_has_overflowed(seq))) 5571 return -EAGAIN; 5572 5573 /* Unescape the result */ 5574 seq->buf[seq->count] = '\0'; 5575 seq->count = start; 5576 seq_commit(seq, string_unescape_inplace(seq->buf + start, UNESCAPE_OCTAL)); 5577 } else { 5578 seq_puts(seq, r->mnt_devname); 5579 } 5580 return 0; 5581 } 5582 5583 static void statmount_mnt_ns_id(struct kstatmount *s, struct mnt_namespace *ns) 5584 { 5585 s->sm.mask |= STATMOUNT_MNT_NS_ID; 5586 s->sm.mnt_ns_id = ns->ns.ns_id; 5587 } 5588 5589 static int statmount_mnt_opts(struct kstatmount *s, struct seq_file *seq) 5590 { 5591 struct vfsmount *mnt = s->mnt; 5592 struct super_block *sb = mnt->mnt_sb; 5593 size_t start = seq->count; 5594 int err; 5595 5596 err = security_sb_show_options(seq, sb); 5597 if (err) 5598 return err; 5599 5600 if (sb->s_op->show_options) { 5601 err = sb->s_op->show_options(seq, mnt->mnt_root); 5602 if (err) 5603 return err; 5604 } 5605 5606 if (unlikely(seq_has_overflowed(seq))) 5607 return -EAGAIN; 5608 5609 if (seq->count == start) 5610 return 0; 5611 5612 /* skip leading comma */ 5613 memmove(seq->buf + start, seq->buf + start + 1, 5614 seq->count - start - 1); 5615 seq->count--; 5616 5617 return 0; 5618 } 5619 5620 static inline int statmount_opt_process(struct seq_file *seq, size_t start) 5621 { 5622 char *buf_end, *opt_end, *src, *dst; 5623 int count = 0; 5624 5625 if (unlikely(seq_has_overflowed(seq))) 5626 return -EAGAIN; 5627 5628 buf_end = seq->buf + seq->count; 5629 dst = seq->buf + start; 5630 src = dst + 1; /* skip initial comma */ 5631 5632 if (src >= buf_end) { 5633 seq->count = start; 5634 return 0; 5635 } 5636 5637 *buf_end = '\0'; 5638 for (; src < buf_end; src = opt_end + 1) { 5639 opt_end = strchrnul(src, ','); 5640 *opt_end = '\0'; 5641 dst += string_unescape(src, dst, 0, UNESCAPE_OCTAL) + 1; 5642 if (WARN_ON_ONCE(++count == INT_MAX)) 5643 return -EOVERFLOW; 5644 } 5645 seq->count = dst - 1 - seq->buf; 5646 return count; 5647 } 5648 5649 static int statmount_opt_array(struct kstatmount *s, struct seq_file *seq) 5650 { 5651 struct vfsmount *mnt = s->mnt; 5652 struct super_block *sb = mnt->mnt_sb; 5653 size_t start = seq->count; 5654 int err; 5655 5656 if (!sb->s_op->show_options) 5657 return 0; 5658 5659 err = sb->s_op->show_options(seq, mnt->mnt_root); 5660 if (err) 5661 return err; 5662 5663 err = statmount_opt_process(seq, start); 5664 if (err < 0) 5665 return err; 5666 5667 s->sm.opt_num = err; 5668 return 0; 5669 } 5670 5671 static int statmount_opt_sec_array(struct kstatmount *s, struct seq_file *seq) 5672 { 5673 struct vfsmount *mnt = s->mnt; 5674 struct super_block *sb = mnt->mnt_sb; 5675 size_t start = seq->count; 5676 int err; 5677 5678 err = security_sb_show_options(seq, sb); 5679 if (err) 5680 return err; 5681 5682 err = statmount_opt_process(seq, start); 5683 if (err < 0) 5684 return err; 5685 5686 s->sm.opt_sec_num = err; 5687 return 0; 5688 } 5689 5690 static inline int statmount_mnt_uidmap(struct kstatmount *s, struct seq_file *seq) 5691 { 5692 int ret; 5693 5694 ret = statmount_mnt_idmap(s->idmap, seq, true); 5695 if (ret < 0) 5696 return ret; 5697 5698 s->sm.mnt_uidmap_num = ret; 5699 /* 5700 * Always raise STATMOUNT_MNT_UIDMAP even if there are no valid 5701 * mappings. This allows userspace to distinguish between a 5702 * non-idmapped mount and an idmapped mount where none of the 5703 * individual mappings are valid in the caller's idmapping. 5704 */ 5705 if (is_valid_mnt_idmap(s->idmap)) 5706 s->sm.mask |= STATMOUNT_MNT_UIDMAP; 5707 return 0; 5708 } 5709 5710 static inline int statmount_mnt_gidmap(struct kstatmount *s, struct seq_file *seq) 5711 { 5712 int ret; 5713 5714 ret = statmount_mnt_idmap(s->idmap, seq, false); 5715 if (ret < 0) 5716 return ret; 5717 5718 s->sm.mnt_gidmap_num = ret; 5719 /* 5720 * Always raise STATMOUNT_MNT_GIDMAP even if there are no valid 5721 * mappings. This allows userspace to distinguish between a 5722 * non-idmapped mount and an idmapped mount where none of the 5723 * individual mappings are valid in the caller's idmapping. 5724 */ 5725 if (is_valid_mnt_idmap(s->idmap)) 5726 s->sm.mask |= STATMOUNT_MNT_GIDMAP; 5727 return 0; 5728 } 5729 5730 static int statmount_string(struct kstatmount *s, u64 flag) 5731 { 5732 int ret = 0; 5733 size_t kbufsize; 5734 struct seq_file *seq = &s->seq; 5735 struct statmount *sm = &s->sm; 5736 u32 start, *offp; 5737 5738 /* Reserve an empty string at the beginning for any unset offsets */ 5739 if (!seq->count) 5740 seq_putc(seq, 0); 5741 5742 start = seq->count; 5743 5744 switch (flag) { 5745 case STATMOUNT_FS_TYPE: 5746 offp = &sm->fs_type; 5747 ret = statmount_fs_type(s, seq); 5748 break; 5749 case STATMOUNT_MNT_ROOT: 5750 offp = &sm->mnt_root; 5751 ret = statmount_mnt_root(s, seq); 5752 break; 5753 case STATMOUNT_MNT_POINT: 5754 offp = &sm->mnt_point; 5755 ret = statmount_mnt_point(s, seq); 5756 break; 5757 case STATMOUNT_MNT_OPTS: 5758 offp = &sm->mnt_opts; 5759 ret = statmount_mnt_opts(s, seq); 5760 break; 5761 case STATMOUNT_OPT_ARRAY: 5762 offp = &sm->opt_array; 5763 ret = statmount_opt_array(s, seq); 5764 break; 5765 case STATMOUNT_OPT_SEC_ARRAY: 5766 offp = &sm->opt_sec_array; 5767 ret = statmount_opt_sec_array(s, seq); 5768 break; 5769 case STATMOUNT_FS_SUBTYPE: 5770 offp = &sm->fs_subtype; 5771 statmount_fs_subtype(s, seq); 5772 break; 5773 case STATMOUNT_SB_SOURCE: 5774 offp = &sm->sb_source; 5775 ret = statmount_sb_source(s, seq); 5776 break; 5777 case STATMOUNT_MNT_UIDMAP: 5778 offp = &sm->mnt_uidmap; 5779 ret = statmount_mnt_uidmap(s, seq); 5780 break; 5781 case STATMOUNT_MNT_GIDMAP: 5782 offp = &sm->mnt_gidmap; 5783 ret = statmount_mnt_gidmap(s, seq); 5784 break; 5785 default: 5786 WARN_ON_ONCE(true); 5787 return -EINVAL; 5788 } 5789 5790 /* 5791 * If nothing was emitted, return to avoid setting the flag 5792 * and terminating the buffer. 5793 */ 5794 if (seq->count == start) 5795 return ret; 5796 if (unlikely(check_add_overflow(sizeof(*sm), seq->count, &kbufsize))) 5797 return -EOVERFLOW; 5798 if (kbufsize >= s->bufsize) 5799 return -EOVERFLOW; 5800 5801 /* signal a retry */ 5802 if (unlikely(seq_has_overflowed(seq))) 5803 return -EAGAIN; 5804 5805 if (ret) 5806 return ret; 5807 5808 seq->buf[seq->count++] = '\0'; 5809 sm->mask |= flag; 5810 *offp = start; 5811 return 0; 5812 } 5813 5814 static int copy_statmount_to_user(struct kstatmount *s) 5815 { 5816 struct statmount *sm = &s->sm; 5817 struct seq_file *seq = &s->seq; 5818 char __user *str = ((char __user *)s->buf) + sizeof(*sm); 5819 size_t copysize = min_t(size_t, s->bufsize, sizeof(*sm)); 5820 5821 if (seq->count && copy_to_user(str, seq->buf, seq->count)) 5822 return -EFAULT; 5823 5824 /* Return the number of bytes copied to the buffer */ 5825 sm->size = copysize + seq->count; 5826 if (copy_to_user(s->buf, sm, copysize)) 5827 return -EFAULT; 5828 5829 return 0; 5830 } 5831 5832 static struct mount *listmnt_next(struct mount *curr, bool reverse) 5833 { 5834 struct rb_node *node; 5835 5836 if (reverse) 5837 node = rb_prev(&curr->mnt_node); 5838 else 5839 node = rb_next(&curr->mnt_node); 5840 5841 return node_to_mount(node); 5842 } 5843 5844 static int grab_requested_root(struct mnt_namespace *ns, struct path *root) 5845 { 5846 struct mount *first, *child; 5847 5848 rwsem_assert_held(&namespace_sem); 5849 5850 /* We're looking at our own ns, just use get_fs_root. */ 5851 if (ns == current->nsproxy->mnt_ns) { 5852 get_fs_root(current->fs, root); 5853 return 0; 5854 } 5855 5856 /* 5857 * We have to find the first mount in our ns and use that, however it 5858 * may not exist, so handle that properly. 5859 */ 5860 if (mnt_ns_empty(ns)) 5861 return -ENOENT; 5862 5863 first = ns->root; 5864 for (child = node_to_mount(ns->mnt_first_node); child; 5865 child = listmnt_next(child, false)) { 5866 if (child != first && child->mnt_parent == first) 5867 break; 5868 } 5869 if (!child) 5870 return -ENOENT; 5871 5872 root->mnt = mntget(&child->mnt); 5873 root->dentry = dget(root->mnt->mnt_root); 5874 return 0; 5875 } 5876 5877 /* This must be updated whenever a new flag is added */ 5878 #define STATMOUNT_SUPPORTED (STATMOUNT_SB_BASIC | \ 5879 STATMOUNT_MNT_BASIC | \ 5880 STATMOUNT_PROPAGATE_FROM | \ 5881 STATMOUNT_MNT_ROOT | \ 5882 STATMOUNT_MNT_POINT | \ 5883 STATMOUNT_FS_TYPE | \ 5884 STATMOUNT_MNT_NS_ID | \ 5885 STATMOUNT_MNT_OPTS | \ 5886 STATMOUNT_FS_SUBTYPE | \ 5887 STATMOUNT_SB_SOURCE | \ 5888 STATMOUNT_OPT_ARRAY | \ 5889 STATMOUNT_OPT_SEC_ARRAY | \ 5890 STATMOUNT_SUPPORTED_MASK | \ 5891 STATMOUNT_MNT_UIDMAP | \ 5892 STATMOUNT_MNT_GIDMAP) 5893 5894 /* locks: namespace_shared */ 5895 static int do_statmount(struct kstatmount *s, u64 mnt_id, u64 mnt_ns_id, 5896 struct file *mnt_file, struct mnt_namespace *ns) 5897 { 5898 int err; 5899 5900 if (mnt_file) { 5901 WARN_ON_ONCE(ns != NULL); 5902 5903 s->mnt = mnt_file->f_path.mnt; 5904 ns = real_mount(s->mnt)->mnt_ns; 5905 if (IS_ERR(ns)) 5906 return PTR_ERR(ns); 5907 if (!ns) 5908 /* 5909 * We can't set mount point and mnt_ns_id since we don't have a 5910 * ns for the mount. This can happen if the mount is unmounted 5911 * with MNT_DETACH. 5912 */ 5913 s->mask &= ~(STATMOUNT_MNT_POINT | STATMOUNT_MNT_NS_ID); 5914 } else { 5915 /* Has the namespace already been emptied? */ 5916 if (mnt_ns_id && mnt_ns_empty(ns)) 5917 return -ENOENT; 5918 5919 s->mnt = lookup_mnt_in_ns(mnt_id, ns); 5920 if (!s->mnt) 5921 return -ENOENT; 5922 } 5923 5924 if (ns) { 5925 err = grab_requested_root(ns, &s->root); 5926 if (err) 5927 return err; 5928 5929 if (!mnt_file) { 5930 struct mount *m; 5931 /* 5932 * Don't trigger audit denials. We just want to determine what 5933 * mounts to show users. 5934 */ 5935 m = real_mount(s->mnt); 5936 if (!is_path_reachable(m, m->mnt.mnt_root, &s->root) && 5937 !ns_capable_noaudit(ns->user_ns, CAP_SYS_ADMIN)) 5938 return -EPERM; 5939 } 5940 } 5941 5942 err = security_sb_statfs(s->mnt->mnt_root); 5943 if (err) 5944 return err; 5945 5946 /* 5947 * Note that mount properties in mnt->mnt_flags, mnt->mnt_idmap 5948 * can change concurrently as we only hold the read-side of the 5949 * namespace semaphore and mount properties may change with only 5950 * the mount lock held. 5951 * 5952 * We could sample the mount lock sequence counter to detect 5953 * those changes and retry. But it's not worth it. Worst that 5954 * happens is that the mnt->mnt_idmap pointer is already changed 5955 * while mnt->mnt_flags isn't or vica versa. So what. 5956 * 5957 * Both mnt->mnt_flags and mnt->mnt_idmap are set and retrieved 5958 * via READ_ONCE()/WRITE_ONCE() and guard against theoretical 5959 * torn read/write. That's all we care about right now. 5960 */ 5961 s->idmap = mnt_idmap(s->mnt); 5962 if (s->mask & STATMOUNT_MNT_BASIC) 5963 statmount_mnt_basic(s); 5964 5965 if (s->mask & STATMOUNT_SB_BASIC) 5966 statmount_sb_basic(s); 5967 5968 if (s->mask & STATMOUNT_PROPAGATE_FROM) 5969 statmount_propagate_from(s); 5970 5971 if (s->mask & STATMOUNT_FS_TYPE) 5972 err = statmount_string(s, STATMOUNT_FS_TYPE); 5973 5974 if (!err && s->mask & STATMOUNT_MNT_ROOT) 5975 err = statmount_string(s, STATMOUNT_MNT_ROOT); 5976 5977 if (!err && s->mask & STATMOUNT_MNT_POINT) 5978 err = statmount_string(s, STATMOUNT_MNT_POINT); 5979 5980 if (!err && s->mask & STATMOUNT_MNT_OPTS) 5981 err = statmount_string(s, STATMOUNT_MNT_OPTS); 5982 5983 if (!err && s->mask & STATMOUNT_OPT_ARRAY) 5984 err = statmount_string(s, STATMOUNT_OPT_ARRAY); 5985 5986 if (!err && s->mask & STATMOUNT_OPT_SEC_ARRAY) 5987 err = statmount_string(s, STATMOUNT_OPT_SEC_ARRAY); 5988 5989 if (!err && s->mask & STATMOUNT_FS_SUBTYPE) 5990 err = statmount_string(s, STATMOUNT_FS_SUBTYPE); 5991 5992 if (!err && s->mask & STATMOUNT_SB_SOURCE) 5993 err = statmount_string(s, STATMOUNT_SB_SOURCE); 5994 5995 if (!err && s->mask & STATMOUNT_MNT_UIDMAP) 5996 err = statmount_string(s, STATMOUNT_MNT_UIDMAP); 5997 5998 if (!err && s->mask & STATMOUNT_MNT_GIDMAP) 5999 err = statmount_string(s, STATMOUNT_MNT_GIDMAP); 6000 6001 if (!err && s->mask & STATMOUNT_MNT_NS_ID) 6002 statmount_mnt_ns_id(s, ns); 6003 6004 if (!err && s->mask & STATMOUNT_SUPPORTED_MASK) { 6005 s->sm.mask |= STATMOUNT_SUPPORTED_MASK; 6006 s->sm.supported_mask = STATMOUNT_SUPPORTED; 6007 } 6008 6009 if (err) 6010 return err; 6011 6012 /* Are there bits in the return mask not present in STATMOUNT_SUPPORTED? */ 6013 WARN_ON_ONCE(~STATMOUNT_SUPPORTED & s->sm.mask); 6014 6015 return 0; 6016 } 6017 6018 static inline bool retry_statmount(const long ret, size_t *seq_size) 6019 { 6020 if (likely(ret != -EAGAIN)) 6021 return false; 6022 if (unlikely(check_mul_overflow(*seq_size, 2, seq_size))) 6023 return false; 6024 if (unlikely(*seq_size > MAX_RW_COUNT)) 6025 return false; 6026 return true; 6027 } 6028 6029 #define STATMOUNT_STRING_REQ (STATMOUNT_MNT_ROOT | STATMOUNT_MNT_POINT | \ 6030 STATMOUNT_FS_TYPE | STATMOUNT_MNT_OPTS | \ 6031 STATMOUNT_FS_SUBTYPE | STATMOUNT_SB_SOURCE | \ 6032 STATMOUNT_OPT_ARRAY | STATMOUNT_OPT_SEC_ARRAY | \ 6033 STATMOUNT_MNT_UIDMAP | STATMOUNT_MNT_GIDMAP) 6034 6035 static int prepare_kstatmount(struct kstatmount *ks, struct mnt_id_req *kreq, 6036 struct statmount __user *buf, size_t bufsize, 6037 size_t seq_size) 6038 { 6039 if (!access_ok(buf, bufsize)) 6040 return -EFAULT; 6041 6042 memset(ks, 0, sizeof(*ks)); 6043 ks->mask = kreq->param; 6044 ks->buf = buf; 6045 ks->bufsize = bufsize; 6046 6047 if (ks->mask & STATMOUNT_STRING_REQ) { 6048 if (bufsize == sizeof(ks->sm)) 6049 return -EOVERFLOW; 6050 6051 ks->seq.buf = kvmalloc(seq_size, GFP_KERNEL_ACCOUNT); 6052 if (!ks->seq.buf) 6053 return -ENOMEM; 6054 6055 ks->seq.size = seq_size; 6056 } 6057 6058 return 0; 6059 } 6060 6061 static int copy_mnt_id_req(const struct mnt_id_req __user *req, 6062 struct mnt_id_req *kreq, unsigned int flags) 6063 { 6064 int ret; 6065 size_t usize; 6066 6067 BUILD_BUG_ON(sizeof(struct mnt_id_req) != MNT_ID_REQ_SIZE_VER1); 6068 6069 ret = get_user(usize, &req->size); 6070 if (ret) 6071 return -EFAULT; 6072 if (unlikely(usize > PAGE_SIZE)) 6073 return -E2BIG; 6074 if (unlikely(usize < MNT_ID_REQ_SIZE_VER0)) 6075 return -EINVAL; 6076 memset(kreq, 0, sizeof(*kreq)); 6077 ret = copy_struct_from_user(kreq, sizeof(*kreq), req, usize); 6078 if (ret) 6079 return ret; 6080 6081 if (flags & STATMOUNT_BY_FD) { 6082 if (kreq->mnt_id || kreq->mnt_ns_id) 6083 return -EINVAL; 6084 } else { 6085 if (kreq->mnt_ns_fd != 0 && kreq->mnt_ns_id) 6086 return -EINVAL; 6087 /* The first valid unique mount id is MNT_UNIQUE_ID_OFFSET + 1. */ 6088 if (kreq->mnt_id <= MNT_UNIQUE_ID_OFFSET) 6089 return -EINVAL; 6090 } 6091 return 0; 6092 } 6093 6094 /* 6095 * If the user requested a specific mount namespace id, look that up and return 6096 * that, or if not simply grab a passive reference on our mount namespace and 6097 * return that. 6098 */ 6099 static struct mnt_namespace *grab_requested_mnt_ns(const struct mnt_id_req *kreq) 6100 { 6101 struct mnt_namespace *mnt_ns; 6102 6103 if (kreq->mnt_ns_id) { 6104 mnt_ns = lookup_mnt_ns(kreq->mnt_ns_id); 6105 if (!mnt_ns) 6106 return ERR_PTR(-ENOENT); 6107 } else if (kreq->mnt_ns_fd) { 6108 struct ns_common *ns; 6109 6110 CLASS(fd, f)(kreq->mnt_ns_fd); 6111 if (fd_empty(f)) 6112 return ERR_PTR(-EBADF); 6113 6114 if (!proc_ns_file(fd_file(f))) 6115 return ERR_PTR(-EINVAL); 6116 6117 ns = get_proc_ns(file_inode(fd_file(f))); 6118 if (ns->ns_type != CLONE_NEWNS) 6119 return ERR_PTR(-EINVAL); 6120 6121 mnt_ns = to_mnt_ns(ns); 6122 refcount_inc(&mnt_ns->passive); 6123 } else { 6124 mnt_ns = current->nsproxy->mnt_ns; 6125 refcount_inc(&mnt_ns->passive); 6126 } 6127 6128 return mnt_ns; 6129 } 6130 6131 SYSCALL_DEFINE4(statmount, const struct mnt_id_req __user *, req, 6132 struct statmount __user *, buf, size_t, bufsize, 6133 unsigned int, flags) 6134 { 6135 struct mnt_namespace *ns __free(mnt_ns_release) = NULL; 6136 struct kstatmount *ks __free(kfree) = NULL; 6137 struct file *mnt_file __free(fput) = NULL; 6138 struct mnt_id_req kreq; 6139 /* We currently support retrieval of 3 strings. */ 6140 size_t seq_size = 3 * PATH_MAX; 6141 int ret; 6142 6143 if (flags & ~STATMOUNT_BY_FD) 6144 return -EINVAL; 6145 6146 ret = copy_mnt_id_req(req, &kreq, flags); 6147 if (ret) 6148 return ret; 6149 6150 if (flags & STATMOUNT_BY_FD) { 6151 mnt_file = fget_raw(kreq.mnt_fd); 6152 if (!mnt_file) 6153 return -EBADF; 6154 /* do_statmount sets ns in case of STATMOUNT_BY_FD */ 6155 } else { 6156 ns = grab_requested_mnt_ns(&kreq); 6157 if (IS_ERR(ns)) 6158 return PTR_ERR(ns); 6159 6160 if (kreq.mnt_ns_id && (ns != current->nsproxy->mnt_ns) && 6161 !ns_capable_noaudit(ns->user_ns, CAP_SYS_ADMIN)) 6162 return -EPERM; 6163 } 6164 6165 ks = kmalloc_obj(*ks, GFP_KERNEL_ACCOUNT); 6166 if (!ks) 6167 return -ENOMEM; 6168 6169 retry: 6170 ret = prepare_kstatmount(ks, &kreq, buf, bufsize, seq_size); 6171 if (ret) 6172 return ret; 6173 6174 scoped_guard(namespace_shared) 6175 ret = do_statmount(ks, kreq.mnt_id, kreq.mnt_ns_id, mnt_file, ns); 6176 6177 if (!ret) 6178 ret = copy_statmount_to_user(ks); 6179 kvfree(ks->seq.buf); 6180 path_put(&ks->root); 6181 if (retry_statmount(ret, &seq_size)) 6182 goto retry; 6183 return ret; 6184 } 6185 6186 struct klistmount { 6187 u64 last_mnt_id; 6188 u64 mnt_parent_id; 6189 u64 *kmnt_ids; 6190 u32 nr_mnt_ids; 6191 struct mnt_namespace *ns; 6192 struct path root; 6193 }; 6194 6195 /* locks: namespace_shared */ 6196 static ssize_t do_listmount(struct klistmount *kls, bool reverse) 6197 { 6198 struct mnt_namespace *ns = kls->ns; 6199 u64 mnt_parent_id = kls->mnt_parent_id; 6200 u64 last_mnt_id = kls->last_mnt_id; 6201 u64 *mnt_ids = kls->kmnt_ids; 6202 size_t nr_mnt_ids = kls->nr_mnt_ids; 6203 struct path orig; 6204 struct mount *r, *first; 6205 ssize_t ret; 6206 6207 rwsem_assert_held(&namespace_sem); 6208 6209 ret = grab_requested_root(ns, &kls->root); 6210 if (ret) 6211 return ret; 6212 6213 if (mnt_parent_id == LSMT_ROOT) { 6214 orig = kls->root; 6215 } else { 6216 orig.mnt = lookup_mnt_in_ns(mnt_parent_id, ns); 6217 if (!orig.mnt) 6218 return -ENOENT; 6219 orig.dentry = orig.mnt->mnt_root; 6220 } 6221 6222 /* 6223 * Don't trigger audit denials. We just want to determine what 6224 * mounts to show users. 6225 */ 6226 if (!is_path_reachable(real_mount(orig.mnt), orig.dentry, &kls->root) && 6227 !ns_capable_noaudit(ns->user_ns, CAP_SYS_ADMIN)) 6228 return -EPERM; 6229 6230 ret = security_sb_statfs(orig.dentry); 6231 if (ret) 6232 return ret; 6233 6234 if (!last_mnt_id) { 6235 if (reverse) 6236 first = node_to_mount(ns->mnt_last_node); 6237 else 6238 first = node_to_mount(ns->mnt_first_node); 6239 } else { 6240 if (reverse) 6241 first = mnt_find_id_at_reverse(ns, last_mnt_id - 1); 6242 else 6243 first = mnt_find_id_at(ns, last_mnt_id + 1); 6244 } 6245 6246 for (ret = 0, r = first; r && nr_mnt_ids; r = listmnt_next(r, reverse)) { 6247 if (r->mnt_id_unique == mnt_parent_id) 6248 continue; 6249 if (!is_path_reachable(r, r->mnt.mnt_root, &orig)) 6250 continue; 6251 *mnt_ids = r->mnt_id_unique; 6252 mnt_ids++; 6253 nr_mnt_ids--; 6254 ret++; 6255 } 6256 return ret; 6257 } 6258 6259 static void __free_klistmount_free(const struct klistmount *kls) 6260 { 6261 path_put(&kls->root); 6262 kvfree(kls->kmnt_ids); 6263 mnt_ns_release(kls->ns); 6264 } 6265 6266 static inline int prepare_klistmount(struct klistmount *kls, struct mnt_id_req *kreq, 6267 size_t nr_mnt_ids) 6268 { 6269 u64 last_mnt_id = kreq->param; 6270 struct mnt_namespace *ns; 6271 6272 /* The first valid unique mount id is MNT_UNIQUE_ID_OFFSET + 1. */ 6273 if (last_mnt_id != 0 && last_mnt_id <= MNT_UNIQUE_ID_OFFSET) 6274 return -EINVAL; 6275 6276 kls->last_mnt_id = last_mnt_id; 6277 6278 kls->nr_mnt_ids = nr_mnt_ids; 6279 kls->kmnt_ids = kvmalloc_array(nr_mnt_ids, sizeof(*kls->kmnt_ids), 6280 GFP_KERNEL_ACCOUNT); 6281 if (!kls->kmnt_ids) 6282 return -ENOMEM; 6283 6284 ns = grab_requested_mnt_ns(kreq); 6285 if (IS_ERR(ns)) 6286 return PTR_ERR(ns); 6287 kls->ns = ns; 6288 6289 kls->mnt_parent_id = kreq->mnt_id; 6290 return 0; 6291 } 6292 6293 SYSCALL_DEFINE4(listmount, const struct mnt_id_req __user *, req, 6294 u64 __user *, mnt_ids, size_t, nr_mnt_ids, unsigned int, flags) 6295 { 6296 struct klistmount kls __free(klistmount_free) = {}; 6297 const size_t maxcount = 1000000; 6298 struct mnt_id_req kreq; 6299 ssize_t ret; 6300 6301 if (flags & ~LISTMOUNT_REVERSE) 6302 return -EINVAL; 6303 6304 /* 6305 * If the mount namespace really has more than 1 million mounts the 6306 * caller must iterate over the mount namespace (and reconsider their 6307 * system design...). 6308 */ 6309 if (unlikely(nr_mnt_ids > maxcount)) 6310 return -EOVERFLOW; 6311 6312 if (!access_ok(mnt_ids, nr_mnt_ids * sizeof(*mnt_ids))) 6313 return -EFAULT; 6314 6315 ret = copy_mnt_id_req(req, &kreq, 0); 6316 if (ret) 6317 return ret; 6318 6319 ret = prepare_klistmount(&kls, &kreq, nr_mnt_ids); 6320 if (ret) 6321 return ret; 6322 6323 if (kreq.mnt_ns_id && (kls.ns != current->nsproxy->mnt_ns) && 6324 !ns_capable_noaudit(kls.ns->user_ns, CAP_SYS_ADMIN)) 6325 return -ENOENT; 6326 6327 /* 6328 * We only need to guard against mount topology changes as 6329 * listmount() doesn't care about any mount properties. 6330 */ 6331 scoped_guard(namespace_shared) 6332 ret = do_listmount(&kls, (flags & LISTMOUNT_REVERSE)); 6333 if (ret <= 0) 6334 return ret; 6335 6336 if (copy_to_user(mnt_ids, kls.kmnt_ids, ret * sizeof(*mnt_ids))) 6337 return -EFAULT; 6338 6339 return ret; 6340 } 6341 6342 struct mnt_namespace init_mnt_ns = { 6343 .ns = NS_COMMON_INIT(init_mnt_ns), 6344 .user_ns = &init_user_ns, 6345 .passive = REFCOUNT_INIT(1), 6346 .mounts = RB_ROOT, 6347 .poll = __WAIT_QUEUE_HEAD_INITIALIZER(init_mnt_ns.poll), 6348 }; 6349 6350 static void __init mount_rootfs_on_nullfs(struct vfsmount *mnt, 6351 struct vfsmount *nullfs_mnt) 6352 { 6353 struct path root = { 6354 .mnt = nullfs_mnt, 6355 .dentry = nullfs_mnt->mnt_root, 6356 }; 6357 6358 LOCK_MOUNT_EXACT(mp, &root); 6359 if (unlikely(IS_ERR(mp.parent))) 6360 panic("VFS: Failed to mount rootfs on nullfs"); 6361 scoped_guard(mount_writer) 6362 attach_mnt(real_mount(mnt), mp.parent, mp.mp); 6363 } 6364 6365 static struct vfsmount *__init knullfs_file_mount(void) 6366 { 6367 struct dentry *file; 6368 struct mount *mnt; 6369 6370 file = nullfs_new_file(knullfs->mnt_sb); 6371 if (IS_ERR(file)) 6372 return ERR_CAST(file); 6373 mnt = clone_mnt(real_mount(knullfs), file, CL_PRIVATE); 6374 dput(file); 6375 if (IS_ERR(mnt)) 6376 return ERR_CAST(mnt); 6377 mnt->mnt_ns = MNT_NS_INTERNAL; 6378 mnt->mnt.mnt_flags |= MNT_INTERNAL; 6379 mnt->mnt.mnt_flags &= ~MNT_READONLY; 6380 dont_mount(mnt->mnt.mnt_root); 6381 return &mnt->mnt; 6382 } 6383 6384 static void __init init_mount_tree(void) 6385 { 6386 struct vfsmount *mnt, *nullfs_mnt; 6387 struct mount *mnt_root; 6388 struct path root; 6389 6390 /* 6391 * We create three mounts: 6392 * 6393 * (1) nullfs with mount id 1 6394 * (2) mutable rootfs with mount id 2 6395 * (3) private nullfs for kthreads (SB_KERNMOUNT), kept in knullfs 6396 * (4) a second mount of (3) rooted on a regular file, kept in 6397 * knullfs_file 6398 * 6399 * with (2) mounted on top of (1). The init_task's root and pwd 6400 * are pointed at (3) so all kthreads start isolated in nullfs. 6401 * A lookup at the cover an unmounted mount left behind finds (3) 6402 * or (4), see __lookup_mnt(). 6403 */ 6404 nullfs_mnt = vfs_kern_mount(&nullfs_fs_type, 0, "nullfs", NULL); 6405 if (IS_ERR(nullfs_mnt)) 6406 panic("VFS: Failed to create nullfs"); 6407 6408 mnt = vfs_kern_mount(&rootfs_fs_type, 0, "rootfs", initramfs_options); 6409 if (IS_ERR(mnt)) 6410 panic("Can't create rootfs"); 6411 6412 VFS_WARN_ON_ONCE(real_mount(nullfs_mnt)->mnt_id != 1); 6413 VFS_WARN_ON_ONCE(real_mount(mnt)->mnt_id != 2); 6414 6415 /* The namespace root is the nullfs mnt. */ 6416 mnt_root = real_mount(nullfs_mnt); 6417 init_mnt_ns.root = mnt_root; 6418 6419 mount_rootfs_on_nullfs(mnt, nullfs_mnt); 6420 6421 pr_info("VFS: Finished mounting rootfs on nullfs\n"); 6422 6423 /* 6424 * We've dropped all locks here but that's fine. Not just are we 6425 * the only task that's running, there's no other mount 6426 * namespace in existence and the initial mount namespace is 6427 * completely empty until we add the mounts we just created. 6428 */ 6429 for (struct mount *p = mnt_root; p; p = next_mnt(p, mnt_root)) { 6430 mnt_add_to_ns(&init_mnt_ns, p); 6431 init_mnt_ns.nr_mounts++; 6432 } 6433 6434 knullfs = kern_mount(&nullfs_fs_type); 6435 if (IS_ERR(knullfs)) 6436 panic("VFS: Failed to create private nullfs instance"); 6437 /* nothing is ever mounted on the root of a kernel thread */ 6438 dont_mount(knullfs->mnt_root); 6439 /* and nothing is ever written through it */ 6440 knullfs->mnt_flags |= MNT_READONLY; 6441 knullfs_file = knullfs_file_mount(); 6442 if (IS_ERR(knullfs_file)) 6443 panic("VFS: Failed to create the nullfs file stand-in"); 6444 root.mnt = knullfs; 6445 root.dentry = knullfs->mnt_root; 6446 6447 init_task.nsproxy->mnt_ns = &init_mnt_ns; 6448 get_mnt_ns(&init_mnt_ns); 6449 set_fs_pwd(current->fs, &root); 6450 set_fs_root(current->fs, &root); 6451 6452 ns_tree_add(&init_mnt_ns); 6453 } 6454 6455 void __init mnt_init(void) 6456 { 6457 int err; 6458 6459 mnt_cache = kmem_cache_create("mnt_cache", sizeof(struct mount), 6460 0, SLAB_HWCACHE_ALIGN|SLAB_PANIC|SLAB_ACCOUNT, NULL); 6461 6462 mount_hashtable = alloc_large_system_hash("Mount-cache", 6463 sizeof(struct hlist_head), 6464 mhash_entries, 19, 6465 HASH_ZERO, 6466 &m_hash_shift, &m_hash_mask, 0, 0); 6467 mountpoint_hashtable = alloc_large_system_hash("Mountpoint-cache", 6468 sizeof(struct hlist_head), 6469 mphash_entries, 19, 6470 HASH_ZERO, 6471 &mp_hash_shift, &mp_hash_mask, 0, 0); 6472 6473 super_dev_init(); 6474 6475 kernfs_init(); 6476 6477 err = sysfs_init(); 6478 if (err) 6479 printk(KERN_WARNING "%s: sysfs_init error: %d\n", 6480 __func__, err); 6481 fs_kobj = kobject_create_and_add("fs", NULL); 6482 if (!fs_kobj) 6483 printk(KERN_WARNING "%s: kobj create error\n", __func__); 6484 shmem_init(); 6485 init_rootfs(); 6486 init_mount_tree(); 6487 failfs_init(); 6488 } 6489 6490 void put_mnt_ns(struct mnt_namespace *ns) 6491 { 6492 if (!ns_ref_put(ns)) 6493 return; 6494 guard(namespace_excl)(); 6495 emptied_ns = ns; 6496 guard(mount_writer)(); 6497 umount_tree(ns->root, 0); 6498 } 6499 6500 struct vfsmount *kern_mount(struct file_system_type *type) 6501 { 6502 struct vfsmount *mnt; 6503 mnt = vfs_kern_mount(type, SB_KERNMOUNT, type->name, NULL); 6504 if (!IS_ERR(mnt)) { 6505 /* 6506 * it is a longterm mount, don't release mnt until 6507 * we unmount before file sys is unregistered 6508 */ 6509 real_mount(mnt)->mnt_ns = MNT_NS_INTERNAL; 6510 } 6511 return mnt; 6512 } 6513 EXPORT_SYMBOL_GPL(kern_mount); 6514 6515 void kern_unmount(struct vfsmount *mnt) 6516 { 6517 /* release long term mount so mount point can be released */ 6518 if (!IS_ERR(mnt)) { 6519 mnt_make_shortterm(mnt); 6520 synchronize_rcu(); /* yecchhh... */ 6521 mntput(mnt); 6522 } 6523 } 6524 EXPORT_SYMBOL(kern_unmount); 6525 6526 void kern_unmount_array(struct vfsmount *mnt[], unsigned int num) 6527 { 6528 unsigned int i; 6529 6530 for (i = 0; i < num; i++) 6531 mnt_make_shortterm(mnt[i]); 6532 synchronize_rcu_expedited(); 6533 for (i = 0; i < num; i++) 6534 mntput(mnt[i]); 6535 } 6536 EXPORT_SYMBOL(kern_unmount_array); 6537 6538 bool our_mnt(struct vfsmount *mnt) 6539 { 6540 return check_mnt(real_mount(mnt)); 6541 } 6542 6543 bool current_chrooted(void) 6544 { 6545 /* Does the current process have a non-standard root */ 6546 struct path fs_root __free(path_put) = {}; 6547 struct mount *root; 6548 6549 get_fs_root(current->fs, &fs_root); 6550 6551 /* Find the namespace root */ 6552 6553 guard(mount_locked_reader)(); 6554 6555 root = topmost_overmount(current->nsproxy->mnt_ns->root); 6556 6557 return fs_root.mnt != &root->mnt || !path_mounted(&fs_root); 6558 } 6559 6560 static bool mnt_already_visible(struct mnt_namespace *ns, 6561 const struct super_block *sb, 6562 int *new_mnt_flags) 6563 { 6564 int new_flags = *new_mnt_flags; 6565 struct mount *mnt; 6566 6567 /* Don't acquire namespace semaphore without a good reason. */ 6568 if (hlist_empty(&ns->mnt_visible_mounts)) 6569 return false; 6570 6571 guard(namespace_shared)(); 6572 hlist_for_each_entry(mnt, &ns->mnt_visible_mounts, mnt_ns_visible) { 6573 const struct super_block *sb_visible = mnt->mnt.mnt_sb; 6574 struct mount *child; 6575 int mnt_flags; 6576 6577 if (sb_visible->s_type != sb->s_type) 6578 continue; 6579 6580 /* 6581 * Restricted variants are not compatible with anything, even 6582 * other restricted variants. 6583 */ 6584 if (sb_visible->s_iflags & SB_I_RESTRICTED_VARIANT) 6585 continue; 6586 6587 /* A local view of the mount flags */ 6588 mnt_flags = mnt->mnt.mnt_flags; 6589 6590 /* Don't miss readonly hidden in the superblock flags */ 6591 if (sb_rdonly(mnt->mnt.mnt_sb)) 6592 mnt_flags |= MNT_LOCK_READONLY; 6593 6594 /* Verify the mount flags are equal to or more permissive 6595 * than the proposed new mount. 6596 */ 6597 if ((mnt_flags & MNT_LOCK_READONLY) && 6598 !(new_flags & MNT_READONLY)) 6599 continue; 6600 if ((mnt_flags & MNT_LOCK_ATIME) && 6601 ((mnt_flags & MNT_ATIME_MASK) != (new_flags & MNT_ATIME_MASK))) 6602 continue; 6603 6604 /* This mount is not fully visible if there are any 6605 * locked child mounts that cover anything except for 6606 * empty directories. 6607 */ 6608 list_for_each_entry(child, &mnt->mnt_mounts, mnt_child) { 6609 struct inode *inode = child->mnt_mountpoint->d_inode; 6610 /* Only worry about locked mounts */ 6611 if (!(child->mnt.mnt_flags & MNT_LOCKED)) 6612 continue; 6613 /* Is the directory permanently empty? */ 6614 if (!is_empty_dir_inode(inode)) 6615 goto next; 6616 } 6617 /* Preserve the locked attributes */ 6618 *new_mnt_flags |= mnt_flags & (MNT_LOCK_READONLY | \ 6619 MNT_LOCK_ATIME); 6620 return true; 6621 next: ; 6622 } 6623 return false; 6624 } 6625 6626 static bool mount_too_revealing(const struct super_block *sb, int *new_mnt_flags) 6627 { 6628 const unsigned long required_iflags = SB_I_NOEXEC | SB_I_NODEV; 6629 struct mnt_namespace *ns = current->nsproxy->mnt_ns; 6630 unsigned long s_iflags; 6631 6632 if (ns->user_ns == &init_user_ns) 6633 return false; 6634 6635 /* Can this filesystem be too revealing? */ 6636 if (!(sb->s_type->fs_flags & FS_USERNS_MOUNT_RESTRICTED)) 6637 return false; 6638 6639 s_iflags = sb->s_iflags; 6640 if ((s_iflags & required_iflags) != required_iflags) { 6641 WARN_ONCE(1, "Expected s_iflags to contain 0x%lx\n", 6642 required_iflags); 6643 return true; 6644 } 6645 6646 /* 6647 * Restricted variants don't need an already visible mount because they 6648 * don't expose the full filesystem view. 6649 */ 6650 if (s_iflags & SB_I_RESTRICTED_VARIANT) 6651 return false; 6652 6653 return !mnt_already_visible(ns, sb, new_mnt_flags); 6654 } 6655 6656 bool mnt_may_suid(struct vfsmount *mnt) 6657 { 6658 /* 6659 * Foreign mounts (accessed via fchdir or through /proc 6660 * symlinks) are always treated as if they are nosuid. This 6661 * prevents namespaces from trusting potentially unsafe 6662 * suid/sgid bits, file caps, or security labels that originate 6663 * in other namespaces. 6664 */ 6665 return !(mnt->mnt_flags & MNT_NOSUID) && check_mnt(real_mount(mnt)) && 6666 current_in_userns(mnt->mnt_sb->s_user_ns); 6667 } 6668 6669 static struct ns_common *mntns_get(struct task_struct *task) 6670 { 6671 struct ns_common *ns = NULL; 6672 struct nsproxy *nsproxy; 6673 6674 task_lock(task); 6675 nsproxy = task->nsproxy; 6676 if (nsproxy) { 6677 ns = &nsproxy->mnt_ns->ns; 6678 get_mnt_ns(to_mnt_ns(ns)); 6679 } 6680 task_unlock(task); 6681 6682 return ns; 6683 } 6684 6685 static void mntns_put(struct ns_common *ns) 6686 { 6687 put_mnt_ns(to_mnt_ns(ns)); 6688 } 6689 6690 static int mntns_install(struct nsset *nsset, struct ns_common *ns) 6691 { 6692 struct nsproxy *nsproxy = nsset->nsproxy; 6693 struct fs_struct *fs = nsset->fs; 6694 struct mnt_namespace *mnt_ns = to_mnt_ns(ns), *old_mnt_ns; 6695 struct user_namespace *user_ns = nsset->cred->user_ns; 6696 struct path root; 6697 int err; 6698 6699 if (!ns_capable(mnt_ns->user_ns, CAP_SYS_ADMIN) || 6700 !ns_capable(user_ns, CAP_SYS_CHROOT) || 6701 !ns_capable(user_ns, CAP_SYS_ADMIN)) 6702 return -EPERM; 6703 6704 if (is_anon_ns(mnt_ns)) 6705 return -EINVAL; 6706 6707 if (fs->users != 1) 6708 return -EINVAL; 6709 6710 get_mnt_ns(mnt_ns); 6711 old_mnt_ns = nsproxy->mnt_ns; 6712 nsproxy->mnt_ns = mnt_ns; 6713 6714 /* Find the root */ 6715 err = vfs_path_lookup(mnt_ns->root->mnt.mnt_root, &mnt_ns->root->mnt, 6716 "/", LOOKUP_DOWN, &root); 6717 if (err) { 6718 /* revert to old namespace */ 6719 nsproxy->mnt_ns = old_mnt_ns; 6720 put_mnt_ns(mnt_ns); 6721 return err; 6722 } 6723 6724 put_mnt_ns(old_mnt_ns); 6725 6726 /* Update the pwd and root */ 6727 set_fs_pwd(fs, &root); 6728 set_fs_root(fs, &root); 6729 6730 path_put(&root); 6731 return 0; 6732 } 6733 6734 static struct user_namespace *mntns_owner(struct ns_common *ns) 6735 { 6736 return to_mnt_ns(ns)->user_ns; 6737 } 6738 6739 const struct proc_ns_operations mntns_operations = { 6740 .name = "mnt", 6741 .get = mntns_get, 6742 .put = mntns_put, 6743 .install = mntns_install, 6744 .owner = mntns_owner, 6745 }; 6746 6747 #ifdef CONFIG_SYSCTL 6748 static const struct ctl_table fs_namespace_sysctls[] = { 6749 { 6750 .procname = "mount-max", 6751 .data = &sysctl_mount_max, 6752 .maxlen = sizeof(unsigned int), 6753 .mode = 0644, 6754 .proc_handler = proc_dointvec_minmax, 6755 .extra1 = SYSCTL_ONE, 6756 }, 6757 }; 6758 6759 static int __init init_fs_namespace_sysctls(void) 6760 { 6761 register_sysctl_init("fs", fs_namespace_sysctls); 6762 return 0; 6763 } 6764 fs_initcall(init_fs_namespace_sysctls); 6765 6766 #endif /* CONFIG_SYSCTL */ 6767