xref: /linux/kernel/futex/core.c (revision f35e3b5784221654f9cdbd6222275a7fa203f6c3)
1 // SPDX-License-Identifier: GPL-2.0-or-later
2 /*
3  *  Fast Userspace Mutexes (which I call "Futexes!").
4  *  (C) Rusty Russell, IBM 2002
5  *
6  *  Generalized futexes, futex requeueing, misc fixes by Ingo Molnar
7  *  (C) Copyright 2003 Red Hat Inc, All Rights Reserved
8  *
9  *  Removed page pinning, fix privately mapped COW pages and other cleanups
10  *  (C) Copyright 2003, 2004 Jamie Lokier
11  *
12  *  Robust futex support started by Ingo Molnar
13  *  (C) Copyright 2006 Red Hat Inc, All Rights Reserved
14  *  Thanks to Thomas Gleixner for suggestions, analysis and fixes.
15  *
16  *  PI-futex support started by Ingo Molnar and Thomas Gleixner
17  *  Copyright (C) 2006 Red Hat, Inc., Ingo Molnar <mingo@redhat.com>
18  *  Copyright (C) 2006 Timesys Corp., Thomas Gleixner <tglx@timesys.com>
19  *
20  *  PRIVATE futexes by Eric Dumazet
21  *  Copyright (C) 2007 Eric Dumazet <dada1@cosmosbay.com>
22  *
23  *  Requeue-PI support by Darren Hart <dvhltc@us.ibm.com>
24  *  Copyright (C) IBM Corporation, 2009
25  *  Thanks to Thomas Gleixner for conceptual design and careful reviews.
26  *
27  *  Thanks to Ben LaHaise for yelling "hashed waitqueues" loudly
28  *  enough at me, Linus for the original (flawed) idea, Matthew
29  *  Kirkwood for proof-of-concept implementation.
30  *
31  *  "The futexes are also cursed."
32  *  "But they come in a choice of three flavours!"
33  */
34 #include <linux/compat.h>
35 #include <linux/debugfs.h>
36 #include <linux/fault-inject.h>
37 #include <linux/gfp.h>
38 #include <linux/jhash.h>
39 #include <linux/memblock.h>
40 #include <linux/mempolicy.h>
41 #include <linux/mmap_lock.h>
42 #include <linux/pagemap.h>
43 #include <linux/plist.h>
44 #include <linux/prctl.h>
45 #include <linux/rseq.h>
46 #include <linux/slab.h>
47 #include <linux/vmalloc.h>
48 #include <linux/kmemleak.h>
49 #include <linux/wait_bit.h>
50 
51 #include <vdso/futex.h>
52 
53 #include <asm/runtime-const.h>
54 
55 #include "futex.h"
56 #include "../locking/rtmutex_common.h"
57 
58 static u32 __futex_mask __ro_after_init;
59 static u32 __futex_shift __ro_after_init;
60 static struct futex_hash_bucket **__futex_queues __ro_after_init;
61 
62 static __always_inline struct futex_hash_bucket **futex_queues(void)
63 {
64 	return runtime_const_ptr(__futex_queues);
65 }
66 
67 struct futex_private_hash {
68 	int		state;
69 	unsigned int	hash_mask;
70 	struct rcu_head	rcu;
71 	void		*mm;
72 	bool		custom;
73 	struct futex_hash_bucket queues[];
74 };
75 
76 /*
77  * Fault injections for futexes.
78  */
79 #ifdef CONFIG_FAIL_FUTEX
80 
81 static struct {
82 	struct fault_attr attr;
83 
84 	bool ignore_private;
85 } fail_futex = {
86 	.attr = FAULT_ATTR_INITIALIZER,
87 	.ignore_private = false,
88 };
89 
90 static int __init setup_fail_futex(char *str)
91 {
92 	return setup_fault_attr(&fail_futex.attr, str);
93 }
94 __setup("fail_futex=", setup_fail_futex);
95 
96 bool should_fail_futex(bool fshared)
97 {
98 	if (fail_futex.ignore_private && !fshared)
99 		return false;
100 
101 	return should_fail(&fail_futex.attr, 1);
102 }
103 
104 #ifdef CONFIG_FAULT_INJECTION_DEBUG_FS
105 
106 static int __init fail_futex_debugfs(void)
107 {
108 	umode_t mode = S_IFREG | S_IRUSR | S_IWUSR;
109 	struct dentry *dir;
110 
111 	dir = fault_create_debugfs_attr("fail_futex", NULL,
112 					&fail_futex.attr);
113 	if (IS_ERR(dir))
114 		return PTR_ERR(dir);
115 
116 	debugfs_create_bool("ignore-private", mode, dir,
117 			    &fail_futex.ignore_private);
118 	return 0;
119 }
120 
121 late_initcall(fail_futex_debugfs);
122 
123 #endif /* CONFIG_FAULT_INJECTION_DEBUG_FS */
124 
125 #endif /* CONFIG_FAIL_FUTEX */
126 
127 static struct futex_hash_bucket *
128 __futex_hash(union futex_key *key, struct futex_private_hash *fph, struct futex_private_hash **fph_p);
129 
130 #ifdef CONFIG_FUTEX_PRIVATE_HASH
131 static bool futex_ref_get(struct futex_private_hash *fph);
132 static bool futex_ref_put(struct futex_private_hash *fph);
133 static bool futex_ref_is_dead(struct futex_private_hash *fph);
134 
135 enum { FR_PERCPU = 0, FR_ATOMIC };
136 
137 static bool futex_private_hash_get(struct futex_private_hash *fph)
138 {
139 	return futex_ref_get(fph);
140 }
141 
142 void futex_private_hash_put(struct futex_private_hash *fph)
143 {
144 	struct mm_struct *mm;
145 
146 	if (!fph)
147 		return;
148 
149 	mm = fph->mm;
150 	if (futex_ref_put(fph))
151 		wake_up_var(mm);
152 }
153 
154 static struct futex_hash_bucket *
155 __futex_hash_private(union futex_key *key, struct futex_private_hash *fph)
156 {
157 	u32 hash;
158 
159 	hash = jhash2((void *)&key->private.address, sizeof(key->private.address) / 4,
160 		      key->both.offset);
161 
162 	return &fph->queues[hash & fph->hash_mask];
163 }
164 
165 static void futex_rehash_private(struct futex_private_hash *old,
166 				 struct futex_private_hash *new)
167 {
168 	struct futex_hash_bucket *hb_old, *hb_new;
169 	unsigned int slots = old->hash_mask + 1;
170 	unsigned int i;
171 
172 	for (i = 0; i < slots; i++) {
173 		struct futex_q *this, *tmp;
174 
175 		hb_old = &old->queues[i];
176 
177 		spin_lock(&hb_old->lock);
178 		plist_for_each_entry_safe(this, tmp, &hb_old->chain, list) {
179 			plist_del(&this->list, &hb_old->chain);
180 			futex_hb_waiters_dec(hb_old);
181 
182 			WARN_ON_ONCE(this->lock_ptr != &hb_old->lock);
183 
184 			hb_new = __futex_hash(&this->key, new, NULL);
185 			futex_hb_waiters_inc(hb_new);
186 			/*
187 			 * The new pointer isn't published yet but an already
188 			 * moved user can be unqueued due to timeout or signal.
189 			 */
190 			spin_lock_nested(&hb_new->lock, SINGLE_DEPTH_NESTING);
191 			plist_add(&this->list, &hb_new->chain);
192 			this->lock_ptr = &hb_new->lock;
193 			spin_unlock(&hb_new->lock);
194 		}
195 		spin_unlock(&hb_old->lock);
196 	}
197 }
198 
199 static bool __futex_pivot_hash(struct mm_struct *mm, struct futex_private_hash *new)
200 {
201 	struct futex_mm_phash *mmph = &mm->futex.phash;
202 	struct futex_private_hash *fph;
203 
204 	WARN_ON_ONCE(mmph->hash_new);
205 
206 	fph = rcu_dereference_protected(mmph->hash, lockdep_is_held(&mmph->lock));
207 	if (fph) {
208 		if (!futex_ref_is_dead(fph)) {
209 			mmph->hash_new = new;
210 			return false;
211 		}
212 
213 		futex_rehash_private(fph, new);
214 	}
215 	new->state = FR_PERCPU;
216 	rcu_assign_pointer(mmph->hash, new);
217 	/*
218 	 * mmph->batches must reference a grace period which started after
219 	 * mmph->hash was assigned. See futex_ref_drop().
220 	 */
221 	mmph->batches = get_state_synchronize_rcu();
222 	kvfree_rcu(fph, rcu);
223 	return true;
224 }
225 
226 static void futex_pivot_hash(struct mm_struct *mm)
227 {
228 	scoped_guard(mutex, &mm->futex.phash.lock) {
229 		struct futex_private_hash *fph;
230 
231 		fph = mm->futex.phash.hash_new;
232 		if (fph) {
233 			mm->futex.phash.hash_new = NULL;
234 			__futex_pivot_hash(mm, fph);
235 		}
236 	}
237 }
238 
239 struct futex_private_hash *futex_private_hash(struct mm_struct *mm)
240 {
241 	/*
242 	 * Ideally we don't loop. If there is a replacement in progress
243 	 * then a new private hash is already prepared and a reference can't be
244 	 * obtained once the last user dropped it's.
245 	 * In that case we block on mm_struct::futex_hash_lock and either have
246 	 * to perform the replacement or wait while someone else is doing the
247 	 * job. Eitherway, on the second iteration we acquire a reference on the
248 	 * new private hash or loop again because a new replacement has been
249 	 * requested.
250 	 */
251 again:
252 	scoped_guard(rcu) {
253 		struct futex_private_hash *fph;
254 
255 		fph = rcu_dereference(mm->futex.phash.hash);
256 		if (!fph)
257 			return NULL;
258 
259 		if (futex_private_hash_get(fph))
260 			return fph;
261 	}
262 	futex_pivot_hash(mm);
263 	goto again;
264 }
265 
266 struct futex_bucket_ref futex_hash(union futex_key *key)
267 {
268 again:
269 	scoped_guard(rcu) {
270 		struct futex_private_hash *fph = NULL;
271 		struct futex_hash_bucket *hb;
272 
273 		hb = __futex_hash(key, NULL, &fph);
274 
275 		if (!fph || futex_private_hash_get(fph))
276 			return (struct futex_bucket_ref){ .hb = hb, .fph = fph };
277 	}
278 	futex_pivot_hash(key->private.mm);
279 	goto again;
280 }
281 
282 #else /* !CONFIG_FUTEX_PRIVATE_HASH */
283 
284 struct futex_bucket_ref futex_hash(union futex_key *key)
285 {
286 	return (struct futex_bucket_ref){ .hb = __futex_hash(key, NULL, NULL), .fph = NULL };
287 }
288 
289 #endif /* CONFIG_FUTEX_PRIVATE_HASH */
290 
291 #ifdef CONFIG_FUTEX_MPOL
292 
293 static int __futex_key_to_node(struct mm_struct *mm, unsigned long addr)
294 {
295 	struct vm_area_struct *vma = vma_lookup(mm, addr);
296 	struct mempolicy *mpol;
297 	int node = FUTEX_NO_NODE;
298 
299 	if (!vma)
300 		return FUTEX_NO_NODE;
301 
302 	mpol = READ_ONCE(vma->vm_policy);
303 	if (!mpol)
304 		return FUTEX_NO_NODE;
305 
306 	switch (mpol->mode) {
307 	case MPOL_PREFERRED:
308 		node = first_node(mpol->nodes);
309 		break;
310 	case MPOL_PREFERRED_MANY:
311 	case MPOL_BIND:
312 		if (mpol->home_node != NUMA_NO_NODE)
313 			node = mpol->home_node;
314 		break;
315 	default:
316 		break;
317 	}
318 
319 	return node;
320 }
321 
322 static int futex_key_to_node_opt(struct mm_struct *mm, unsigned long addr)
323 {
324 	int seq, node;
325 
326 	guard(rcu)();
327 
328 	if (!mmap_lock_speculate_try_begin(mm, &seq))
329 		return -EBUSY;
330 
331 	node = __futex_key_to_node(mm, addr);
332 
333 	if (mmap_lock_speculate_retry(mm, seq))
334 		return -EAGAIN;
335 
336 	return node;
337 }
338 
339 static int futex_mpol(struct mm_struct *mm, unsigned long addr)
340 {
341 	int node;
342 
343 	node = futex_key_to_node_opt(mm, addr);
344 	if (node >= FUTEX_NO_NODE)
345 		return node;
346 
347 	guard(mmap_read_lock)(mm);
348 	return __futex_key_to_node(mm, addr);
349 }
350 
351 #else /* !CONFIG_FUTEX_MPOL */
352 
353 static int futex_mpol(struct mm_struct *mm, unsigned long addr)
354 {
355 	return FUTEX_NO_NODE;
356 }
357 
358 #endif /* CONFIG_FUTEX_MPOL */
359 
360 /**
361  * __futex_hash - Return the hash bucket
362  * @key:	Pointer to the futex key for which the hash is calculated
363  * @fph:	Pointer to private hash if known
364  * @fph_p:	Pointer to a private hash pointer; output for the private hash
365  *              used when set.
366  *
367  * We hash on the keys returned from get_futex_key (see below) and return the
368  * corresponding hash bucket.
369  * If the FUTEX is PROCESS_PRIVATE then a per-process hash bucket (from the
370  * private hash) is returned if existing. Otherwise a hash bucket from the
371  * global hash is returned.
372  */
373 static struct futex_hash_bucket *
374 __futex_hash(union futex_key *key, struct futex_private_hash *fph, struct futex_private_hash **fph_p)
375 {
376 	int node = key->both.node;
377 	u32 hash;
378 
379 #ifdef CONFIG_FUTEX_PRIVATE_HASH
380 	if (node == FUTEX_NO_NODE && futex_key_is_private(key)) {
381 		if (!fph)
382 			fph = rcu_dereference(key->private.mm->futex.phash.hash);
383 		if (fph && fph->hash_mask) {
384 			if (fph_p)
385 				*fph_p = fph;
386 			return __futex_hash_private(key, fph);
387 		}
388 	}
389 #endif
390 
391 	hash = jhash2((u32 *)key, offsetof(typeof(*key), both.offset) / sizeof(u32),
392 		      key->both.offset);
393 
394 	if (node == FUTEX_NO_NODE) {
395 		/*
396 		 * In case of !FLAGS_NUMA, use some unused hash bits to pick a
397 		 * node -- this ensures regular futexes are interleaved across
398 		 * the nodes and avoids having to allocate multiple
399 		 * hash-tables.
400 		 *
401 		 * NOTE: this isn't perfectly uniform, but it is fast and
402 		 * handles sparse node masks.
403 		 */
404 		node = runtime_const_shift_right_32(hash, __futex_shift) % nr_node_ids;
405 		if (!node_possible(node)) {
406 			node = find_next_bit_wrap(node_possible_map.bits, nr_node_ids, node);
407 		}
408 	}
409 
410 	return &futex_queues()[node][runtime_const_mask_32(hash, __futex_mask)];
411 }
412 
413 /**
414  * futex_setup_timer - set up the sleeping hrtimer.
415  * @time:	ptr to the given timeout value
416  * @timeout:	the hrtimer_sleeper structure to be set up
417  * @flags:	futex flags
418  * @range_ns:	optional range in ns
419  *
420  * Return: Initialized hrtimer_sleeper structure or NULL if no timeout
421  *	   value given
422  */
423 struct hrtimer_sleeper *futex_setup_timer(ktime_t *time, struct hrtimer_sleeper *timeout,
424 					  int flags, u64 range_ns)
425 {
426 	if (!time)
427 		return NULL;
428 
429 	hrtimer_setup_sleeper_on_stack(timeout,
430 				       (flags & FLAGS_CLOCKRT) ? CLOCK_REALTIME : CLOCK_MONOTONIC,
431 				       HRTIMER_MODE_ABS);
432 	/*
433 	 * If range_ns is 0, calling hrtimer_set_expires_range_ns() is
434 	 * effectively the same as calling hrtimer_set_expires().
435 	 */
436 	hrtimer_set_expires_range_ns(&timeout->timer, *time, range_ns);
437 
438 	return timeout;
439 }
440 
441 /*
442  * Generate a machine wide unique identifier for this inode.
443  *
444  * This relies on u64 not wrapping in the life-time of the machine; which with
445  * 1ns resolution means almost 585 years.
446  *
447  * This further relies on the fact that a well formed program will not unmap
448  * the file while it has a (shared) futex waiting on it. This mapping will have
449  * a file reference which pins the mount and inode.
450  *
451  * If for some reason an inode gets evicted and read back in again, it will get
452  * a new sequence number and will _NOT_ match, even though it is the exact same
453  * file.
454  *
455  * It is important that futex_match() will never have a false-positive, esp.
456  * for PI futexes that can mess up the state. The above argues that false-negatives
457  * are only possible for malformed programs.
458  */
459 static u64 get_inode_sequence_number(struct inode *inode)
460 {
461 	static atomic64_t i_seq;
462 	u64 old;
463 
464 	/* Does the inode already have a sequence number? */
465 	old = atomic64_read(&inode->i_sequence);
466 	if (likely(old))
467 		return old;
468 
469 	for (;;) {
470 		u64 new = atomic64_inc_return(&i_seq);
471 		if (WARN_ON_ONCE(!new))
472 			continue;
473 
474 		old = 0;
475 		if (!atomic64_try_cmpxchg_relaxed(&inode->i_sequence, &old, new))
476 			return old;
477 		return new;
478 	}
479 }
480 
481 /**
482  * get_futex_key() - Get parameters which are the keys for a futex
483  * @uaddr:	virtual address of the futex
484  * @flags:	FLAGS_*
485  * @key:	address where result is stored.
486  * @rw:		mapping needs to be read/write (values: FUTEX_READ,
487  *              FUTEX_WRITE)
488  *
489  * Return: a negative error code or 0
490  *
491  * The key words are stored in @key on success.
492  *
493  * For shared mappings (when @fshared), the key is:
494  *
495  *   ( inode->i_sequence, page offset within mapping, offset_within_page )
496  *
497  * [ also see get_inode_sequence_number() ]
498  *
499  * For private mappings (or when !@fshared), the key is:
500  *
501  *   ( current->mm, address, 0 )
502  *
503  * This allows (cross process, where applicable) identification of the futex
504  * without keeping the page pinned for the duration of the FUTEX_WAIT.
505  *
506  * lock_page() might sleep, the caller should not hold a spinlock.
507  */
508 int get_futex_key(u32 __user *uaddr, unsigned int flags, union futex_key *key,
509 		  enum futex_access rw)
510 {
511 	unsigned long address = (unsigned long)uaddr;
512 	struct mm_struct *mm = current->mm;
513 	struct page *page;
514 	struct folio *folio;
515 	struct address_space *mapping;
516 	int node, err, size, ro = 0;
517 	bool node_updated = false;
518 	bool fshared;
519 
520 	fshared = flags & FLAGS_SHARED;
521 	size = futex_size(flags);
522 	if (flags & FLAGS_NUMA)
523 		size *= 2;
524 
525 	/*
526 	 * The futex address must be "naturally" aligned.
527 	 */
528 	key->both.offset = address % PAGE_SIZE;
529 	if (unlikely((address & (size-1)) != 0))
530 		return -EINVAL;
531 	address -= key->both.offset;
532 
533 	if (unlikely(!access_ok(uaddr, size)))
534 		return -EFAULT;
535 
536 	if (unlikely(should_fail_futex(fshared)))
537 		return -EFAULT;
538 
539 	node = FUTEX_NO_NODE;
540 
541 	if (flags & FLAGS_NUMA) {
542 		u32 __user *naddr = (void *)uaddr + size / 2;
543 
544 		if (get_user_inline(node, naddr))
545 			return -EFAULT;
546 
547 		if ((node != FUTEX_NO_NODE) &&
548 		    ((unsigned int)node >= MAX_NUMNODES || !node_possible(node)))
549 			return -EINVAL;
550 	}
551 
552 	if (node == FUTEX_NO_NODE && (flags & FLAGS_MPOL)) {
553 		node = futex_mpol(mm, address);
554 		node_updated = true;
555 	}
556 
557 	if (flags & FLAGS_NUMA) {
558 		u32 __user *naddr = (void *)uaddr + size / 2;
559 
560 		if (node == FUTEX_NO_NODE) {
561 			node = numa_node_id();
562 			node_updated = true;
563 		}
564 		if (node_updated && put_user_inline(node, naddr))
565 			return -EFAULT;
566 	}
567 
568 	key->both.node = node;
569 
570 	/*
571 	 * PROCESS_PRIVATE futexes are fast.
572 	 * As the mm cannot disappear under us and the 'key' only needs
573 	 * virtual address, we dont even have to find the underlying vma.
574 	 * Note : We do have to check 'uaddr' is a valid user address,
575 	 *        but access_ok() should be faster than find_vma()
576 	 */
577 	if (!fshared) {
578 		/*
579 		 * On no-MMU, shared futexes are treated as private, therefore
580 		 * we must not include the current process in the key. Since
581 		 * there is only one address space, the address is a unique key
582 		 * on its own.
583 		 */
584 		if (IS_ENABLED(CONFIG_MMU))
585 			key->private.mm = mm;
586 		else
587 			key->private.mm = NULL;
588 
589 		key->private.address = address;
590 		return 0;
591 	}
592 
593 again:
594 	/* Ignore any VERIFY_READ mapping (futex common case) */
595 	if (unlikely(should_fail_futex(true)))
596 		return -EFAULT;
597 
598 	err = get_user_pages_fast(address, 1, FOLL_WRITE, &page);
599 	/*
600 	 * If write access is not required (eg. FUTEX_WAIT), try
601 	 * and get read-only access.
602 	 */
603 	if (err == -EFAULT && rw == FUTEX_READ) {
604 		err = get_user_pages_fast(address, 1, 0, &page);
605 		ro = 1;
606 	}
607 	if (err < 0)
608 		return err;
609 	else
610 		err = 0;
611 
612 	/*
613 	 * The treatment of mapping from this point on is critical. The folio
614 	 * lock protects many things but in this context the folio lock
615 	 * stabilizes mapping, prevents inode freeing in the shared
616 	 * file-backed region case and guards against movement to swap cache.
617 	 *
618 	 * Strictly speaking the folio lock is not needed in all cases being
619 	 * considered here and folio lock forces unnecessarily serialization.
620 	 * From this point on, mapping will be re-verified if necessary and
621 	 * folio lock will be acquired only if it is unavoidable
622 	 *
623 	 * Mapping checks require the folio so it is looked up now. For
624 	 * anonymous pages, it does not matter if the folio is split
625 	 * in the future as the key is based on the address. For
626 	 * filesystem-backed pages, the precise page is required as the
627 	 * index of the page determines the key.
628 	 */
629 	folio = page_folio(page);
630 	mapping = READ_ONCE(folio->mapping);
631 
632 	/*
633 	 * If folio->mapping is NULL, then it cannot be an anonymous
634 	 * page; but it might be the ZERO_PAGE or in the gate area or
635 	 * in a special mapping (all cases which we are happy to fail);
636 	 * or it may have been a good file page when get_user_pages_fast
637 	 * found it, but truncated or holepunched or subjected to
638 	 * invalidate_complete_page2 before we got the folio lock (also
639 	 * cases which we are happy to fail).  And we hold a reference,
640 	 * so refcount care in invalidate_inode_page's remove_mapping
641 	 * prevents drop_caches from setting mapping to NULL beneath us.
642 	 *
643 	 * The case we do have to guard against is when memory pressure made
644 	 * shmem_writepage move it from filecache to swapcache beneath us:
645 	 * an unlikely race, but we do need to retry for folio->mapping.
646 	 */
647 	if (unlikely(!mapping)) {
648 		int shmem_swizzled;
649 
650 		/*
651 		 * Folio lock is required to identify which special case above
652 		 * applies. If this is really a shmem page then the folio lock
653 		 * will prevent unexpected transitions.
654 		 */
655 		folio_lock(folio);
656 		shmem_swizzled = folio_test_swapcache(folio) || folio->mapping;
657 		folio_unlock(folio);
658 		folio_put(folio);
659 
660 		if (shmem_swizzled)
661 			goto again;
662 
663 		return -EFAULT;
664 	}
665 
666 	/*
667 	 * Private mappings are handled in a simple way.
668 	 *
669 	 * If the futex key is stored in anonymous memory, then the associated
670 	 * object is the mm which is implicitly pinned by the calling process.
671 	 *
672 	 * NOTE: When userspace waits on a MAP_SHARED mapping, even if
673 	 * it's a read-only handle, it's expected that futexes attach to
674 	 * the object not the particular process.
675 	 */
676 	if (folio_test_anon(folio)) {
677 		/*
678 		 * A RO anonymous page will never change and thus doesn't make
679 		 * sense for futex operations.
680 		 */
681 		if (unlikely(should_fail_futex(true)) || ro) {
682 			err = -EFAULT;
683 			goto out;
684 		}
685 
686 		key->both.offset |= FUT_OFF_MMSHARED; /* ref taken on mm */
687 		key->private.mm = mm;
688 		key->private.address = address;
689 
690 	} else {
691 		struct inode *inode;
692 
693 		/*
694 		 * The associated futex object in this case is the inode and
695 		 * the folio->mapping must be traversed. Ordinarily this should
696 		 * be stabilised under folio lock but it's not strictly
697 		 * necessary in this case as we just want to pin the inode, not
698 		 * update i_pages or anything like that.
699 		 *
700 		 * The RCU read lock is taken as the inode is finally freed
701 		 * under RCU. If the mapping still matches expectations then the
702 		 * mapping->host can be safely accessed as being a valid inode.
703 		 */
704 		rcu_read_lock();
705 
706 		if (READ_ONCE(folio->mapping) != mapping) {
707 			rcu_read_unlock();
708 			folio_put(folio);
709 
710 			goto again;
711 		}
712 
713 		inode = READ_ONCE(mapping->host);
714 		if (!inode) {
715 			rcu_read_unlock();
716 			folio_put(folio);
717 
718 			goto again;
719 		}
720 
721 		key->both.offset |= FUT_OFF_INODE; /* inode-based key */
722 		key->shared.i_seq = get_inode_sequence_number(inode);
723 		key->shared.pgoff = page_pgoff(folio, page);
724 		rcu_read_unlock();
725 	}
726 
727 out:
728 	folio_put(folio);
729 	return err;
730 }
731 
732 /**
733  * fault_in_user_writeable() - Fault in user address and verify RW access
734  * @uaddr:	pointer to faulting user space address
735  *
736  * Slow path to fixup the fault we just took in the atomic write
737  * access to @uaddr.
738  *
739  * We have no generic implementation of a non-destructive write to the
740  * user address. We know that we faulted in the atomic pagefault
741  * disabled section so we can as well avoid the #PF overhead by
742  * calling get_user_pages() right away.
743  */
744 int fault_in_user_writeable(u32 __user *uaddr)
745 {
746 	struct mm_struct *mm = current->mm;
747 	int ret;
748 
749 	mmap_read_lock(mm);
750 	ret = fixup_user_fault(mm, (unsigned long)uaddr,
751 			       FAULT_FLAG_WRITE, NULL);
752 	mmap_read_unlock(mm);
753 
754 	return ret < 0 ? ret : 0;
755 }
756 
757 /**
758  * futex_top_waiter() - Return the highest priority waiter on a futex
759  * @hb:		the hash bucket the futex_q's reside in
760  * @key:	the futex key (to distinguish it from other futex futex_q's)
761  *
762  * Must be called with the hb lock held.
763  */
764 struct futex_q *futex_top_waiter(struct futex_hash_bucket *hb, union futex_key *key)
765 {
766 	struct futex_q *this;
767 
768 	plist_for_each_entry(this, &hb->chain, list) {
769 		if (futex_match(&this->key, key))
770 			return this;
771 	}
772 	return NULL;
773 }
774 
775 /**
776  * wait_for_owner_exiting - Block until the owner has exited
777  * @ret: owner's current futex lock status
778  * @exiting:	Pointer to the exiting task
779  *
780  * Caller must hold a refcount on @exiting.
781  */
782 void wait_for_owner_exiting(int ret, struct task_struct *exiting)
783 {
784 	if (ret != -EBUSY) {
785 		WARN_ON_ONCE(exiting);
786 		return;
787 	}
788 
789 	if (WARN_ON_ONCE(ret == -EBUSY && !exiting))
790 		return;
791 
792 	mutex_lock(&exiting->futex.exit_mutex);
793 	/*
794 	 * No point in doing state checking here. If the waiter got here
795 	 * while the task was in exec()->exec_futex_release() then it can
796 	 * have any FUTEX_STATE_* value when the waiter has acquired the
797 	 * mutex. OK, if running, EXITING or DEAD if it reached exit()
798 	 * already. Highly unlikely and not a problem. Just one more round
799 	 * through the futex maze.
800 	 */
801 	mutex_unlock(&exiting->futex.exit_mutex);
802 
803 	put_task_struct(exiting);
804 }
805 
806 /**
807  * __futex_unqueue() - Remove the futex_q from its futex_hash_bucket
808  * @q:	The futex_q to unqueue
809  *
810  * The q->lock_ptr must not be NULL and must be held by the caller.
811  */
812 void __futex_unqueue(struct futex_q *q)
813 {
814 	struct futex_hash_bucket *hb;
815 
816 	if (WARN_ON_SMP(!q->lock_ptr) || WARN_ON(plist_node_empty(&q->list)))
817 		return;
818 	lockdep_assert_held(q->lock_ptr);
819 
820 	hb = container_of(q->lock_ptr, struct futex_hash_bucket, lock);
821 	plist_del(&q->list, &hb->chain);
822 	futex_hb_waiters_dec(hb);
823 }
824 
825 /* The key must be already stored in q->key. */
826 void futex_q_lock(struct futex_q *q, struct futex_hash_bucket *hb)
827 {
828 	/*
829 	 * Increment the counter before taking the lock so that
830 	 * a potential waker won't miss a to-be-slept task that is
831 	 * waiting for the spinlock. This is safe as all futex_q_lock()
832 	 * users end up calling futex_queue(). Similarly, for housekeeping,
833 	 * decrement the counter at futex_q_unlock() when some error has
834 	 * occurred and we don't end up adding the task to the list.
835 	 */
836 	futex_hb_waiters_inc(hb); /* implies smp_mb(); (A) */
837 
838 	q->lock_ptr = &hb->lock;
839 
840 	spin_lock(&hb->lock);
841 	__acquire(q->lock_ptr);
842 }
843 
844 void futex_q_unlock(struct futex_hash_bucket *hb)
845 {
846 	futex_hb_waiters_dec(hb);
847 	spin_unlock(&hb->lock);
848 }
849 
850 void __futex_queue(struct futex_q *q, struct futex_hash_bucket *hb,
851 		   struct task_struct *task)
852 {
853 	int prio;
854 
855 	/*
856 	 * The priority used to register this element is
857 	 * - either the real thread-priority for the real-time threads
858 	 * (i.e. threads with a priority lower than MAX_RT_PRIO)
859 	 * - or MAX_RT_PRIO for non-RT threads.
860 	 * Thus, all RT-threads are woken first in priority order, and
861 	 * the others are woken last, in FIFO order.
862 	 */
863 	prio = min(current->normal_prio, MAX_RT_PRIO);
864 
865 	plist_node_init(&q->list, prio);
866 	plist_add(&q->list, &hb->chain);
867 	q->task = task;
868 }
869 
870 /**
871  * futex_unqueue() - Remove the futex_q from its futex_hash_bucket
872  * @q:	The futex_q to unqueue
873  *
874  * The q->lock_ptr must not be held by the caller. A call to futex_unqueue() must
875  * be paired with exactly one earlier call to futex_queue().
876  *
877  * Return:
878  *  - 1 - if the futex_q was still queued (and we removed unqueued it);
879  *  - 0 - if the futex_q was already removed by the waking thread
880  */
881 int futex_unqueue(struct futex_q *q)
882 {
883 	spinlock_t *lock_ptr;
884 	int ret = 0;
885 
886 	/* RCU so lock_ptr is not going away during locking. */
887 	guard(rcu)();
888 	/* In the common case we don't take the spinlock, which is nice. */
889 retry:
890 	/*
891 	 * q->lock_ptr can change between this read and the following spin_lock.
892 	 * Use READ_ONCE to forbid the compiler from reloading q->lock_ptr and
893 	 * optimizing lock_ptr out of the logic below.
894 	 */
895 	lock_ptr = READ_ONCE(q->lock_ptr);
896 	if (lock_ptr != NULL) {
897 		spin_lock(lock_ptr);
898 		/*
899 		 * q->lock_ptr can change between reading it and
900 		 * spin_lock(), causing us to take the wrong lock.  This
901 		 * corrects the race condition.
902 		 *
903 		 * Reasoning goes like this: if we have the wrong lock,
904 		 * q->lock_ptr must have changed (maybe several times)
905 		 * between reading it and the spin_lock().  It can
906 		 * change again after the spin_lock() but only if it was
907 		 * already changed before the spin_lock().  It cannot,
908 		 * however, change back to the original value.  Therefore
909 		 * we can detect whether we acquired the correct lock.
910 		 */
911 		if (unlikely(lock_ptr != q->lock_ptr)) {
912 			spin_unlock(lock_ptr);
913 			goto retry;
914 		}
915 		__futex_unqueue(q);
916 
917 		BUG_ON(q->pi_state);
918 
919 		spin_unlock(lock_ptr);
920 		ret = 1;
921 	}
922 
923 	return ret;
924 }
925 
926 void futex_q_lockptr_lock(struct futex_q *q)
927 {
928 	spinlock_t *lock_ptr;
929 
930 	/*
931 	 * See futex_unqueue() why lock_ptr can change.
932 	 */
933 	guard(rcu)();
934 retry:
935 	lock_ptr = READ_ONCE(q->lock_ptr);
936 	spin_lock(lock_ptr);
937 
938 	if (unlikely(lock_ptr != q->lock_ptr)) {
939 		spin_unlock(lock_ptr);
940 		goto retry;
941 	}
942 }
943 
944 /*
945  * PI futexes can not be requeued and must remove themselves from the hash
946  * bucket. The hash bucket lock (i.e. lock_ptr) is held.
947  */
948 void futex_unqueue_pi(struct futex_q *q)
949 {
950 	/*
951 	 * If the lock was not acquired (due to timeout or signal) then the
952 	 * rt_waiter is removed before futex_q is. If this is observed by
953 	 * an unlocker after dropping the rtmutex wait lock and before
954 	 * acquiring the hash bucket lock, then the unlocker dequeues the
955 	 * futex_q from the hash bucket list to guarantee consistent state
956 	 * vs. userspace. Therefore the dequeue here must be conditional.
957 	 */
958 	if (!plist_node_empty(&q->list))
959 		__futex_unqueue(q);
960 
961 	BUG_ON(!q->pi_state);
962 	put_pi_state(q->pi_state);
963 	q->pi_state = NULL;
964 }
965 
966 /* Constants for the pending_op argument of handle_futex_death */
967 #define HANDLE_DEATH_PENDING	true
968 #define HANDLE_DEATH_LIST	false
969 
970 /*
971  * Process a futex-list entry, check whether it's owned by the
972  * dying task, and do notification if so:
973  */
974 static int handle_futex_death(u32 __user *uaddr, struct task_struct *curr,
975 			      unsigned int mod, bool pending_op)
976 {
977 	bool pi = !!(mod & FUTEX_ROBUST_MOD_PI);
978 	u32 uval, nval, mval;
979 	pid_t owner;
980 	int err;
981 
982 	/* Futex address must be 32bit aligned */
983 	if ((((unsigned long)uaddr) % sizeof(*uaddr)) != 0)
984 		return -1;
985 
986 retry:
987 	if (get_user(uval, uaddr))
988 		return -1;
989 
990 	/*
991 	 * Special case for regular (non PI) futexes. Ordinarily, we do
992 	 * not perform any processing here unless the current thread was
993 	 * the owner of the futex (by the TID check below).
994 	 *
995 	 * However, the unlock path has three race scenarios:
996 	 *
997 	 * 1. The unlock path releases the user space futex value and
998 	 *    before it can execute the futex() syscall to wake up
999 	 *    waiters it is killed.
1000 	 *
1001 	 * 2. A woken up waiter is killed before it can acquire the
1002 	 *    futex in user space.
1003 	 *
1004 	 * 3. A woken up waiter is killed in user space after another
1005 	 *    thread has acquired the futex, but before it can set
1006 	 *    FUTEX_WAITERS.
1007 	 *
1008 	 * Note that, if userspace uses the FUTEX_ROBUST_UNLOCK flag, we
1009 	 * will not see case 1 here.
1010 	 *
1011 	 * In the second and third case, the wake up notification could
1012 	 * be generated from any of:
1013 	 *
1014 	 *    i.   An ordinary futex wakeup after unlock (with or
1015 	 *         without FUTEX_ROBUST_UNLOCK)
1016 	 *    ii.  A robust wakeup from another thread's death
1017 	 *    iii. A previous round through this special case
1018 	 *
1019 	 * As a result, the futex world will be in one of four states:
1020 	 *
1021 	 *    A. The futex word is 0 (unlocked)
1022 	 *    B. The futex word is owned by another thread
1023 	 *       (FUTEX_WAITERS is not set)
1024 	 *    C. The futex word is owned by another thread
1025 	 *       (FUTEX_WAITERS set)
1026 	 *    D. The futex's owner died and OWNER_DIED is set
1027 	 *       (the owner part of the word is 0)
1028 	 *
1029 	 * The key issue is that the kernel usually (at least from
1030 	 * sources ii. and iii. or when so requested by userspace from
1031 	 * source i.) only ever wakes *one* waiter at a time. If this
1032 	 * waiter dies before acquiring the futex (or setting the
1033 	 * FUTEX_WAITERS bit), the kernel *must* still wake the next
1034 	 * waiter down the line to uphold the futex invariants and
1035 	 * avoid lost wakeups. Note we do not need to handle state C,
1036 	 * as it does not matter to us whether *we* successfully set
1037 	 * the bit or a third thread did so in the meantime.
1038 	 *
1039 	 * Therefore, in these cases we must issue an additional
1040 	 * futex_wake(). Note however that we *must not* set OWNER_DIED
1041 	 * here. Our thread is *not* the owner of the futex.
1042 	 *
1043 	 * Thus to summarize, the conditions for needing the additional
1044 	 * futex_wake() are:
1045 	 *
1046 	 *	1) @pending_op == true (the thread has not finished the
1047 	 *	   mutex operation)
1048 	 *	2) The futex word is in one of the states A, B or D
1049 	 *	3) Regular futex: @pi == false
1050 	 *
1051 	 * Note in particular that in all of the states A-D the owner
1052 	 * portion of the futex word differs from our thread's TID
1053 	 * (unless the actual owner has the same TID in another PID
1054 	 * namespace, but we cannot currently distinguish that
1055 	 * scenario), so this can be a special-case wakeup in the bail
1056 	 * path of the ordinary TID check.
1057 	 */
1058 	owner = uval & FUTEX_TID_MASK;
1059 
1060 	if (owner != task_pid_vnr(curr)) {
1061 		if (pending_op && !pi && (!owner || !(uval & FUTEX_WAITERS))) {
1062 			futex_wake(uaddr, FLAGS_SIZE_32 | FLAGS_SHARED, NULL, 1,
1063 				   FUTEX_BITSET_MATCH_ANY);
1064 		}
1065 		return 0;
1066 	}
1067 
1068 	/*
1069 	 * Ok, this dying thread is truly holding a futex
1070 	 * of interest. Set the OWNER_DIED bit atomically
1071 	 * via cmpxchg, and if the value had FUTEX_WAITERS
1072 	 * set, wake up a waiter (if any). (We have to do a
1073 	 * futex_wake() even if OWNER_DIED is already set -
1074 	 * to handle the rare but possible case of recursive
1075 	 * thread-death.) The rest of the cleanup is done in
1076 	 * userspace.
1077 	 */
1078 	mval = (uval & FUTEX_WAITERS) | FUTEX_OWNER_DIED;
1079 
1080 	/*
1081 	 * We are not holding a lock here, but we want to have
1082 	 * the pagefault_disable/enable() protection because
1083 	 * we want to handle the fault gracefully. If the
1084 	 * access fails we try to fault in the futex with R/W
1085 	 * verification via get_user_pages. get_user() above
1086 	 * does not guarantee R/W access. If that fails we
1087 	 * give up and leave the futex locked.
1088 	 */
1089 	if ((err = futex_cmpxchg_value_locked(&nval, uaddr, uval, mval))) {
1090 		switch (err) {
1091 		case -EFAULT:
1092 			if (fault_in_user_writeable(uaddr))
1093 				return -1;
1094 			goto retry;
1095 
1096 		case -EAGAIN:
1097 			cond_resched();
1098 			goto retry;
1099 
1100 		default:
1101 			WARN_ON_ONCE(1);
1102 			return err;
1103 		}
1104 	}
1105 
1106 	if (nval != uval)
1107 		goto retry;
1108 
1109 	/*
1110 	 * Wake robust non-PI futexes here. The wakeup of
1111 	 * PI futexes happens in exit_pi_state():
1112 	 */
1113 	if (!pi && (uval & FUTEX_WAITERS)) {
1114 		futex_wake(uaddr, FLAGS_SIZE_32 | FLAGS_SHARED, NULL, 1,
1115 			   FUTEX_BITSET_MATCH_ANY);
1116 	}
1117 
1118 	return 0;
1119 }
1120 
1121 /*
1122  * Fetch a robust-list pointer. Bit 0 signals PI futexes:
1123  */
1124 static inline int fetch_robust_entry(struct robust_list __user **entry,
1125 				     struct robust_list __user * __user *head,
1126 				     unsigned int *mod)
1127 {
1128 	unsigned long uentry;
1129 
1130 	if (get_user(uentry, (unsigned long __user *)head))
1131 		return -EFAULT;
1132 
1133 	*entry = (void __user *)(uentry & ~FUTEX_ROBUST_MOD_MASK);
1134 	*mod = uentry & FUTEX_ROBUST_MOD_MASK;
1135 
1136 	return 0;
1137 }
1138 
1139 /*
1140  * Walk curr->futex.robust_list (very carefully, it's a userspace list!)
1141  * and mark any locks found there dead, and notify any waiters.
1142  *
1143  * We silently return on any sign of list-walking problem.
1144  */
1145 static void exit_robust_list(struct task_struct *curr)
1146 {
1147 	struct robust_list_head __user *head = curr->futex.robust_list;
1148 	unsigned int limit = ROBUST_LIST_LIMIT, cur_mod, next_mod, pend_mod;
1149 	struct robust_list __user *entry, *next_entry, *pending;
1150 	unsigned long futex_offset;
1151 	int rc;
1152 
1153 	/*
1154 	 * Fetch the list head (which was registered earlier, via
1155 	 * sys_set_robust_list()):
1156 	 */
1157 	if (fetch_robust_entry(&entry, &head->list.next, &cur_mod))
1158 		return;
1159 	/*
1160 	 * Fetch the relative futex offset:
1161 	 */
1162 	if (get_user(futex_offset, &head->futex_offset))
1163 		return;
1164 	/*
1165 	 * Fetch any possibly pending lock-add first, and handle it
1166 	 * if it exists:
1167 	 */
1168 	if (fetch_robust_entry(&pending, &head->list_op_pending, &pend_mod))
1169 		return;
1170 
1171 	next_entry = NULL;	/* avoid warning with gcc */
1172 	while (entry != &head->list) {
1173 		/*
1174 		 * Fetch the next entry in the list before calling
1175 		 * handle_futex_death:
1176 		 */
1177 		rc = fetch_robust_entry(&next_entry, &entry->next, &next_mod);
1178 		/*
1179 		 * A pending lock might already be on the list, so
1180 		 * don't process it twice:
1181 		 */
1182 		if (entry != pending) {
1183 			if (handle_futex_death((void __user *)entry + futex_offset,
1184 						curr, cur_mod, HANDLE_DEATH_LIST))
1185 				return;
1186 		}
1187 		if (rc)
1188 			return;
1189 		entry = next_entry;
1190 		cur_mod = next_mod;
1191 		/*
1192 		 * Avoid excessively long or circular lists:
1193 		 */
1194 		if (!--limit)
1195 			break;
1196 
1197 		cond_resched();
1198 	}
1199 
1200 	if (pending) {
1201 		handle_futex_death((void __user *)pending + futex_offset,
1202 				   curr, pend_mod, HANDLE_DEATH_PENDING);
1203 	}
1204 }
1205 
1206 static bool robust_list_clear_pending(unsigned long __user *pop)
1207 {
1208 	struct robust_list_head __user *head = current->futex.robust_list;
1209 
1210 	if (!put_user(0UL, pop))
1211 		return true;
1212 
1213 	/*
1214 	 * Just give up. The robust list head is usually part of TLS, so the
1215 	 * chance that this gets resolved is close to zero.
1216 	 *
1217 	 * If @pop_addr is the robust_list_head::list_op_pending pointer then
1218 	 * clear the robust list head pointer to prevent further damage when the
1219 	 * task exits.  Better a few stale futexes than corrupted memory. But
1220 	 * that's mostly an academic exercise.
1221 	 */
1222 	if (pop == (unsigned long __user *)&head->list_op_pending)
1223 		current->futex.robust_list = NULL;
1224 	return false;
1225 }
1226 
1227 #ifdef CONFIG_COMPAT
1228 static void __user *futex_uaddr(struct robust_list __user *entry,
1229 				compat_long_t futex_offset)
1230 {
1231 	compat_uptr_t base = ptr_to_compat(entry);
1232 	void __user *uaddr = compat_ptr(base + futex_offset);
1233 
1234 	return uaddr;
1235 }
1236 
1237 /*
1238  * Fetch a robust-list pointer. Bit 0 signals PI futexes:
1239  */
1240 static inline int
1241 compat_fetch_robust_entry(compat_uptr_t *uentry, struct robust_list __user **entry,
1242 		   compat_uptr_t __user *head, unsigned int *pflags)
1243 {
1244 	if (get_user(*uentry, head))
1245 		return -EFAULT;
1246 
1247 	*entry = compat_ptr((*uentry) & ~FUTEX_ROBUST_MOD_MASK);
1248 	*pflags = (unsigned int)(*uentry) & FUTEX_ROBUST_MOD_MASK;
1249 
1250 	return 0;
1251 }
1252 
1253 /*
1254  * Walk curr->futex.robust_list (very carefully, it's a userspace list!)
1255  * and mark any locks found there dead, and notify any waiters.
1256  *
1257  * We silently return on any sign of list-walking problem.
1258  */
1259 static void compat_exit_robust_list(struct task_struct *curr)
1260 {
1261 	struct compat_robust_list_head __user *head = current->futex.compat_robust_list;
1262 	unsigned int limit = ROBUST_LIST_LIMIT, cur_mod, next_mod, pend_mod;
1263 	struct robust_list __user *entry, *next_entry, *pending;
1264 	compat_uptr_t uentry, next_uentry, upending;
1265 	compat_long_t futex_offset;
1266 	int rc;
1267 
1268 	/*
1269 	 * Fetch the list head (which was registered earlier, via
1270 	 * sys_set_robust_list()):
1271 	 */
1272 	if (compat_fetch_robust_entry(&uentry, &entry, &head->list.next, &cur_mod))
1273 		return;
1274 	/*
1275 	 * Fetch the relative futex offset:
1276 	 */
1277 	if (get_user(futex_offset, &head->futex_offset))
1278 		return;
1279 	/*
1280 	 * Fetch any possibly pending lock-add first, and handle it
1281 	 * if it exists:
1282 	 */
1283 	if (compat_fetch_robust_entry(&upending, &pending, &head->list_op_pending, &pend_mod))
1284 		return;
1285 
1286 	next_entry = NULL;	/* avoid warning with gcc */
1287 	while (entry != (struct robust_list __user *) &head->list) {
1288 		/*
1289 		 * Fetch the next entry in the list before calling
1290 		 * handle_futex_death:
1291 		 */
1292 		rc = compat_fetch_robust_entry(&next_uentry, &next_entry,
1293 			(compat_uptr_t __user *)&entry->next, &next_mod);
1294 		/*
1295 		 * A pending lock might already be on the list, so
1296 		 * dont process it twice:
1297 		 */
1298 		if (entry != pending) {
1299 			void __user *uaddr = futex_uaddr(entry, futex_offset);
1300 
1301 			if (handle_futex_death(uaddr, curr, cur_mod, HANDLE_DEATH_LIST))
1302 				return;
1303 		}
1304 		if (rc)
1305 			return;
1306 		uentry = next_uentry;
1307 		entry = next_entry;
1308 		cur_mod = next_mod;
1309 		/*
1310 		 * Avoid excessively long or circular lists:
1311 		 */
1312 		if (!--limit)
1313 			break;
1314 
1315 		cond_resched();
1316 	}
1317 	if (pending) {
1318 		void __user *uaddr = futex_uaddr(pending, futex_offset);
1319 
1320 		handle_futex_death(uaddr, curr, pend_mod, HANDLE_DEATH_PENDING);
1321 	}
1322 }
1323 
1324 static bool compat_robust_list_clear_pending(u32 __user *pop)
1325 {
1326 	struct compat_robust_list_head __user *head = current->futex.compat_robust_list;
1327 
1328 	if (!put_user(0U, pop))
1329 		return true;
1330 
1331 	/* See comment in robust_list_clear_pending(). */
1332 	if (pop == &head->list_op_pending)
1333 		current->futex.compat_robust_list = NULL;
1334 	return false;
1335 }
1336 #else
1337 static bool compat_robust_list_clear_pending(u32 __user *pop_addr) { return false; }
1338 #endif
1339 
1340 #ifdef CONFIG_FUTEX_PI
1341 
1342 /*
1343  * This task is holding PI mutexes at exit time => bad.
1344  * Kernel cleans up PI-state, but userspace is likely hosed.
1345  * (Robust-futex cleanup is separate and might save the day for userspace.)
1346  */
1347 static void exit_pi_state_list(struct task_struct *curr)
1348 {
1349 	struct list_head *next, *head = &curr->futex.pi_state_list;
1350 	struct futex_pi_state *pi_state;
1351 	union futex_key key = FUTEX_KEY_INIT;
1352 
1353 	/*
1354 	 * The mutex mm_struct::futex_hash_lock might be acquired.
1355 	 */
1356 	might_sleep();
1357 	/*
1358 	 * Ensure the hash remains stable (no resize) during the while loop
1359 	 * below. The hb pointer is acquired under the pi_lock so we can't block
1360 	 * on the mutex.
1361 	 */
1362 	WARN_ON(curr != current);
1363 	guard(private_hash)(current->mm);
1364 	/*
1365 	 * We are a ZOMBIE and nobody can enqueue itself on
1366 	 * pi_state_list anymore, but we have to be careful
1367 	 * versus waiters unqueueing themselves:
1368 	 */
1369 	raw_spin_lock_irq(&curr->pi_lock);
1370 	while (!list_empty(head)) {
1371 		next = head->next;
1372 		pi_state = list_entry(next, struct futex_pi_state, list);
1373 		key = pi_state->key;
1374 		if (1) {
1375 			CLASS(hbr, hbr)(&key);
1376 			auto hb = hbr.hb;
1377 
1378 			/*
1379 			 * We can race against put_pi_state() removing itself from the
1380 			 * list (a waiter going away). put_pi_state() will first
1381 			 * decrement the reference count and then modify the list, so
1382 			 * its possible to see the list entry but fail this reference
1383 			 * acquire.
1384 			 *
1385 			 * In that case; drop the locks to let put_pi_state() make
1386 			 * progress and retry the loop.
1387 			 */
1388 			if (!refcount_inc_not_zero(&pi_state->refcount)) {
1389 				raw_spin_unlock_irq(&curr->pi_lock);
1390 				cpu_relax();
1391 				raw_spin_lock_irq(&curr->pi_lock);
1392 				continue;
1393 			}
1394 			raw_spin_unlock_irq(&curr->pi_lock);
1395 
1396 			spin_lock(&hb->lock);
1397 			raw_spin_lock_irq(&pi_state->pi_mutex.wait_lock);
1398 			raw_spin_lock(&curr->pi_lock);
1399 			/*
1400 			 * We dropped the pi-lock, so re-check whether this
1401 			 * task still owns the PI-state:
1402 			 */
1403 			if (head->next != next) {
1404 				/* retain curr->pi_lock for the loop invariant */
1405 				raw_spin_unlock(&pi_state->pi_mutex.wait_lock);
1406 				spin_unlock(&hb->lock);
1407 				put_pi_state(pi_state);
1408 				continue;
1409 			}
1410 
1411 			WARN_ON(pi_state->owner != curr);
1412 			WARN_ON(list_empty(&pi_state->list));
1413 			list_del_init(&pi_state->list);
1414 			pi_state->owner = NULL;
1415 
1416 			raw_spin_unlock(&curr->pi_lock);
1417 			raw_spin_unlock_irq(&pi_state->pi_mutex.wait_lock);
1418 			spin_unlock(&hb->lock);
1419 		}
1420 
1421 		rt_mutex_futex_unlock(&pi_state->pi_mutex);
1422 		put_pi_state(pi_state);
1423 
1424 		raw_spin_lock_irq(&curr->pi_lock);
1425 	}
1426 	raw_spin_unlock_irq(&curr->pi_lock);
1427 }
1428 #else
1429 static inline void exit_pi_state_list(struct task_struct *curr) { }
1430 #endif
1431 
1432 bool futex_robust_list_clear_pending(void __user *pop, unsigned int flags)
1433 {
1434 	bool size32bit = !!(flags & FLAGS_ROBUST_LIST32);
1435 
1436 	if (!IS_ENABLED(CONFIG_64BIT) && !size32bit)
1437 		return false;
1438 
1439 	if (IS_ENABLED(CONFIG_64BIT) && size32bit)
1440 		return compat_robust_list_clear_pending(pop);
1441 
1442 	return robust_list_clear_pending(pop);
1443 }
1444 
1445 #ifdef CONFIG_FUTEX_ROBUST_UNLOCK
1446 void __futex_fixup_robust_unlock(struct pt_regs *regs, struct futex_unlock_cs_range *csr)
1447 {
1448 	/*
1449 	 * arch_futex_robust_unlock_get_pop() returns the list pending op pointer from
1450 	 * @regs if the try_cmpxchg() succeeded.
1451 	 */
1452 	void __user *pop = arch_futex_robust_unlock_get_pop(regs);
1453 
1454 	if (!pop)
1455 		return;
1456 
1457 	futex_robust_list_clear_pending(pop, csr->pop_size32 ? FLAGS_ROBUST_LIST32 : 0);
1458 }
1459 #endif /* CONFIG_FUTEX_ROBUST_UNLOCK */
1460 
1461 static void futex_cleanup(struct task_struct *tsk)
1462 {
1463 	if (unlikely(tsk->futex.robust_list)) {
1464 		exit_robust_list(tsk);
1465 		tsk->futex.robust_list = NULL;
1466 	}
1467 
1468 #ifdef CONFIG_COMPAT
1469 	if (unlikely(tsk->futex.compat_robust_list)) {
1470 		compat_exit_robust_list(tsk);
1471 		tsk->futex.compat_robust_list = NULL;
1472 	}
1473 #endif
1474 
1475 	if (unlikely(!list_empty(&tsk->futex.pi_state_list)))
1476 		exit_pi_state_list(tsk);
1477 }
1478 
1479 /**
1480  * futex_exit_recursive - Set the tasks futex state to FUTEX_STATE_DEAD
1481  * @tsk:	task to set the state on
1482  *
1483  * Set the futex exit state of the task lockless. The futex waiter code
1484  * observes that state when a task is exiting and loops until the task has
1485  * actually finished the futex cleanup. The worst case for this is that the
1486  * waiter runs through the wait loop until the state becomes visible.
1487  *
1488  * This is called from the recursive fault handling path in make_task_dead().
1489  *
1490  * This is best effort. Either the futex exit code has run already or
1491  * not. If the OWNER_DIED bit has been set on the futex then the waiter can
1492  * take it over. If not, the problem is pushed back to user space. If the
1493  * futex exit code did not run yet, then an already queued waiter might
1494  * block forever, but there is nothing which can be done about that.
1495  */
1496 void futex_exit_recursive(struct task_struct *tsk)
1497 {
1498 	/* If the state is FUTEX_STATE_EXITING then futex_exit_mutex is held */
1499 	if (tsk->futex.state == FUTEX_STATE_EXITING) {
1500 		__assume_ctx_lock(&tsk->futex.exit_mutex);
1501 		mutex_unlock(&tsk->futex.exit_mutex);
1502 	}
1503 	tsk->futex.state = FUTEX_STATE_DEAD;
1504 }
1505 
1506 static void futex_cleanup_begin(struct task_struct *tsk)
1507 	__acquires(&tsk->futex.exit_mutex)
1508 {
1509 	/*
1510 	 * Prevent various race issues against a concurrent incoming waiter
1511 	 * including live locks by forcing the waiter to block on
1512 	 * tsk->futex.exit_mutex when it observes FUTEX_STATE_EXITING in
1513 	 * attach_to_pi_owner().
1514 	 */
1515 	mutex_lock(&tsk->futex.exit_mutex);
1516 
1517 	/*
1518 	 * Switch the state to FUTEX_STATE_EXITING under tsk->pi_lock.
1519 	 *
1520 	 * This ensures that all subsequent checks of tsk->futex_state in
1521 	 * attach_to_pi_owner() must observe FUTEX_STATE_EXITING with
1522 	 * tsk->pi_lock held.
1523 	 *
1524 	 * It guarantees also that a pi_state which was queued right before
1525 	 * the state change under tsk->pi_lock by a concurrent waiter must
1526 	 * be observed in exit_pi_state_list().
1527 	 */
1528 	raw_spin_lock_irq(&tsk->pi_lock);
1529 	tsk->futex.state = FUTEX_STATE_EXITING;
1530 	raw_spin_unlock_irq(&tsk->pi_lock);
1531 }
1532 
1533 static void futex_cleanup_end(struct task_struct *tsk)
1534 	__releases(&tsk->futex.exit_mutex)
1535 {
1536 	scoped_guard(raw_spinlock_irq, &tsk->pi_lock)
1537 		tsk->futex.state = FUTEX_STATE_DEAD;
1538 
1539 	/*
1540 	 * Drop the exit protection. This unblocks waiters which observed
1541 	 * FUTEX_STATE_EXITING to reevaluate the state.
1542 	 */
1543 	mutex_unlock(&tsk->futex.exit_mutex);
1544 }
1545 
1546 /*
1547  * Invoked from mm_exit_exec_release() to cleanup the robust lists and pi state
1548  * of the outgoing task.
1549  *
1550  * exec() makes it interesting for futexes because the TID of the task stays the
1551  * same, but from a futex perspective the task has to be treated like an exiting
1552  * task. This is especially important for the sanity check for private futexes
1553  * in attach_to_pi_owner() which compares the owner's mm with the waiter's mm.
1554  *
1555  * That check would give the wrong answer if futex_cleanup_end() would
1556  * set the state to FUTEX_STATE_OK as long as the task still has the old
1557  * mm.
1558  *
1559  * After the task has switched to the new mm it sets it to
1560  * FUTEX_STATE_OK again in futex_exec_done().
1561  */
1562 void futex_exit_exec_release(struct task_struct *tsk)
1563 {
1564 	futex_cleanup_begin(tsk);
1565 	futex_cleanup(tsk);
1566 	futex_cleanup_end(tsk);
1567 }
1568 
1569 /*
1570  * exec() has switched to the new mm. Futex operations are safe again.
1571  */
1572 void futex_exec_done(struct task_struct *tsk)
1573 {
1574 	/*
1575 	 * This store does not have to take tsk::futex::exit_mutex because the
1576 	 * phase where waiters block on it during state FUTEX_STATE_EXITING has
1577 	 * been finished when futex_cleanup_end() set the state to
1578 	 * FUTEX_STATE_DEAD.
1579 	 *
1580 	 * This transitions back from FUTEX_STATE_DEAD to FUTEX_STATE_OK. The
1581 	 * ordering guarantee required here is that the previous store to
1582 	 * tsk::mm in the calling code cannot be reordered against this store.
1583 	 */
1584 	guard(raw_spinlock_irq)(&tsk->pi_lock);
1585 	tsk->futex.state = FUTEX_STATE_OK;
1586 }
1587 
1588 static void futex_hash_bucket_init(struct futex_hash_bucket *fhb)
1589 {
1590 	atomic_set(&fhb->waiters, 0);
1591 	plist_head_init(&fhb->chain);
1592 	spin_lock_init(&fhb->lock);
1593 }
1594 
1595 #define FH_CUSTOM	0x01
1596 
1597 #ifdef CONFIG_FUTEX_PRIVATE_HASH
1598 
1599 /*
1600  * futex-ref
1601  *
1602  * Heavily inspired by percpu-rwsem/percpu-refcount; not reusing any of that
1603  * code because it just doesn't fit right.
1604  *
1605  * Dual counter, per-cpu / atomic approach like percpu-refcount, except it
1606  * re-initializes the state automatically, such that the fph swizzle is also a
1607  * transition back to per-cpu.
1608  */
1609 
1610 static void futex_ref_rcu(struct rcu_head *head);
1611 
1612 static void __futex_ref_atomic_begin(struct futex_private_hash *fph)
1613 {
1614 	struct mm_struct *mm = fph->mm;
1615 
1616 	/*
1617 	 * The counter we're about to switch to must have fully switched;
1618 	 * otherwise it would be impossible for it to have reported success
1619 	 * from futex_ref_is_dead().
1620 	 */
1621 	WARN_ON_ONCE(atomic_long_read(&mm->futex.phash.atomic) != 0);
1622 
1623 	/*
1624 	 * Set the atomic to the bias value such that futex_ref_{get,put}()
1625 	 * will never observe 0. Will be fixed up in __futex_ref_atomic_end()
1626 	 * when folding in the percpu count.
1627 	 */
1628 	atomic_long_set(&mm->futex.phash.atomic, LONG_MAX);
1629 	smp_store_release(&fph->state, FR_ATOMIC);
1630 
1631 	call_rcu_hurry(&mm->futex.phash.rcu, futex_ref_rcu);
1632 }
1633 
1634 static void __futex_ref_atomic_end(struct futex_private_hash *fph)
1635 {
1636 	struct mm_struct *mm = fph->mm;
1637 	unsigned int count = 0;
1638 	long ret;
1639 	int cpu;
1640 
1641 	/*
1642 	 * Per __futex_ref_atomic_begin() the state of the fph must be ATOMIC
1643 	 * and per this RCU callback, everybody must now observe this state and
1644 	 * use the atomic variable.
1645 	 */
1646 	WARN_ON_ONCE(fph->state != FR_ATOMIC);
1647 
1648 	/*
1649 	 * Therefore the per-cpu counter is now stable, sum and reset.
1650 	 */
1651 	for_each_possible_cpu(cpu) {
1652 		unsigned int *ptr = per_cpu_ptr(mm->futex.phash.ref, cpu);
1653 		count += *ptr;
1654 		*ptr = 0;
1655 	}
1656 
1657 	/*
1658 	 * Re-init for the next cycle.
1659 	 */
1660 	this_cpu_inc(*mm->futex.phash.ref); /* 0 -> 1 */
1661 
1662 	/*
1663 	 * Add actual count, subtract bias and initial refcount.
1664 	 *
1665 	 * The moment this atomic operation happens, futex_ref_is_dead() can
1666 	 * become true.
1667 	 */
1668 	ret = atomic_long_add_return(count - LONG_MAX - 1, &mm->futex.phash.atomic);
1669 	if (!ret)
1670 		wake_up_var(mm);
1671 
1672 	WARN_ON_ONCE(ret < 0);
1673 	mmput_async(mm);
1674 }
1675 
1676 static void futex_ref_rcu(struct rcu_head *head)
1677 {
1678 	struct mm_struct *mm = container_of(head, struct mm_struct, futex.phash.rcu);
1679 	struct futex_private_hash *fph = rcu_dereference_raw(mm->futex.phash.hash);
1680 
1681 	if (fph->state == FR_PERCPU) {
1682 		/*
1683 		 * Per this extra grace-period, everybody must now observe
1684 		 * fph as the current fph and no previously observed fph's
1685 		 * are in-flight.
1686 		 *
1687 		 * Notably, nobody will now rely on the atomic
1688 		 * futex_ref_is_dead() state anymore so we can begin the
1689 		 * migration of the per-cpu counter into the atomic.
1690 		 */
1691 		__futex_ref_atomic_begin(fph);
1692 		return;
1693 	}
1694 
1695 	__futex_ref_atomic_end(fph);
1696 }
1697 
1698 /*
1699  * Drop the initial refcount and transition to atomics.
1700  */
1701 static void futex_ref_drop(struct futex_private_hash *fph)
1702 {
1703 	struct mm_struct *mm = fph->mm;
1704 
1705 	/*
1706 	 * Can only transition the current fph;
1707 	 */
1708 	WARN_ON_ONCE(rcu_dereference_raw(mm->futex.phash.hash) != fph);
1709 	/*
1710 	 * We enqueue at least one RCU callback. Ensure mm stays if the task
1711 	 * exits before the transition is completed.
1712 	 */
1713 	mmget(mm);
1714 
1715 	/*
1716 	 * In order to avoid the following scenario:
1717 	 *
1718 	 * futex_hash()			__futex_pivot_hash()
1719 	 *   guard(rcu);		  guard(mm->futex.phash.lock);
1720 	 *   fph = mm->futex.phash.hash;
1721 	 *				  rcu_assign_pointer(&mm->futex.phash.hash, new);
1722 	 *				futex_hash_allocate()
1723 	 *				  futex_ref_drop()
1724 	 *				    fph->state = FR_ATOMIC;
1725 	 *				    atomic_set(, BIAS);
1726 	 *
1727 	 *   futex_private_hash_get(fph); // OOPS
1728 	 *
1729 	 * Where an old fph (which is FR_ATOMIC) and should fail on
1730 	 * inc_not_zero, will succeed because a new transition is started and
1731 	 * the atomic is bias'ed away from 0.
1732 	 *
1733 	 * There must be at least one full grace-period between publishing a
1734 	 * new fph and trying to replace it.
1735 	 */
1736 	if (poll_state_synchronize_rcu(mm->futex.phash.batches)) {
1737 		/*
1738 		 * There was a grace-period, we can begin now.
1739 		 */
1740 		__futex_ref_atomic_begin(fph);
1741 		return;
1742 	}
1743 
1744 	call_rcu_hurry(&mm->futex.phash.rcu, futex_ref_rcu);
1745 }
1746 
1747 static bool futex_ref_get(struct futex_private_hash *fph)
1748 {
1749 	struct mm_struct *mm = fph->mm;
1750 
1751 	guard(preempt)();
1752 
1753 	if (READ_ONCE(fph->state) == FR_PERCPU) {
1754 		__this_cpu_inc(*mm->futex.phash.ref);
1755 		return true;
1756 	}
1757 
1758 	return atomic_long_inc_not_zero(&mm->futex.phash.atomic);
1759 }
1760 
1761 static bool futex_ref_put(struct futex_private_hash *fph)
1762 {
1763 	struct mm_struct *mm = fph->mm;
1764 
1765 	guard(preempt)();
1766 
1767 	if (READ_ONCE(fph->state) == FR_PERCPU) {
1768 		__this_cpu_dec(*mm->futex.phash.ref);
1769 		return false;
1770 	}
1771 
1772 	return atomic_long_dec_and_test(&mm->futex.phash.atomic);
1773 }
1774 
1775 static bool futex_ref_is_dead(struct futex_private_hash *fph)
1776 {
1777 	struct mm_struct *mm = fph->mm;
1778 
1779 	guard(rcu)();
1780 
1781 	if (smp_load_acquire(&fph->state) == FR_PERCPU)
1782 		return false;
1783 
1784 	return atomic_long_read(&mm->futex.phash.atomic) == 0;
1785 }
1786 
1787 static void futex_hash_init_mm(struct futex_mm_data *fd)
1788 {
1789 	memset(&fd->phash, 0, sizeof(fd->phash));
1790 	mutex_init(&fd->phash.lock);
1791 	fd->phash.batches = get_state_synchronize_rcu();
1792 }
1793 
1794 void futex_hash_free(struct mm_struct *mm)
1795 {
1796 	struct futex_private_hash *fph;
1797 
1798 	free_percpu(mm->futex.phash.ref);
1799 	kvfree(mm->futex.phash.hash_new);
1800 	fph = rcu_dereference_raw(mm->futex.phash.hash);
1801 	kvfree(fph);
1802 }
1803 
1804 static bool futex_pivot_pending(struct mm_struct *mm)
1805 {
1806 	struct futex_mm_phash *mmph = &mm->futex.phash;
1807 	struct futex_private_hash *fph;
1808 
1809 	guard(mutex)(&mmph->lock);
1810 
1811 	if (!mmph->hash_new)
1812 		return true;
1813 
1814 	fph = rcu_dereference_raw(mmph->hash);
1815 	return futex_ref_is_dead(fph);
1816 }
1817 
1818 static bool futex_hash_less(struct futex_private_hash *a,
1819 			    struct futex_private_hash *b)
1820 {
1821 	/* user provided always wins */
1822 	if (!a->custom && b->custom)
1823 		return true;
1824 	if (a->custom && !b->custom)
1825 		return false;
1826 
1827 	/* zero-sized hash wins */
1828 	if (!b->hash_mask)
1829 		return true;
1830 	if (!a->hash_mask)
1831 		return false;
1832 
1833 	/* keep the biggest */
1834 	if (a->hash_mask < b->hash_mask)
1835 		return true;
1836 	if (a->hash_mask > b->hash_mask)
1837 		return false;
1838 
1839 	return false; /* equal */
1840 }
1841 
1842 static int futex_hash_allocate(unsigned int hash_slots, unsigned int flags)
1843 {
1844 	struct mm_struct *mm = current->mm;
1845 	struct futex_private_hash *fph;
1846 	bool custom = flags & FH_CUSTOM;
1847 	int i;
1848 
1849 	if (hash_slots && (hash_slots == 1 || !is_power_of_2(hash_slots)))
1850 		return -EINVAL;
1851 
1852 	/*
1853 	 * Once we've disabled the global hash there is no way back.
1854 	 */
1855 	scoped_guard(rcu) {
1856 		fph = rcu_dereference(mm->futex.phash.hash);
1857 		if (fph && !fph->hash_mask) {
1858 			if (custom)
1859 				return -EBUSY;
1860 			return 0;
1861 		}
1862 	}
1863 
1864 	if (!mm->futex.phash.ref) {
1865 		unsigned int __percpu *ref = alloc_percpu(unsigned int);
1866 
1867 		if (!ref)
1868 			return -ENOMEM;
1869 
1870 		/*
1871 		 * Tasks sharing the mm can run this concurrently, so take the
1872 		 * initial reference before publishing the counter.
1873 		 */
1874 		this_cpu_inc(*ref); /* 0 -> 1 */
1875 		if (cmpxchg(&mm->futex.phash.ref, NULL, ref))
1876 			free_percpu(ref);
1877 	}
1878 
1879 	fph = kvzalloc_flex(*fph, queues, hash_slots,
1880 			    GFP_KERNEL_ACCOUNT | __GFP_NOWARN);
1881 	if (!fph)
1882 		return -ENOMEM;
1883 
1884 	fph->hash_mask = hash_slots ? hash_slots - 1 : 0;
1885 	fph->custom = custom;
1886 	fph->mm = mm;
1887 
1888 	for (i = 0; i < hash_slots; i++)
1889 		futex_hash_bucket_init(&fph->queues[i]);
1890 
1891 	if (custom) {
1892 		struct wait_bit_queue_entry __wbq_entry;
1893 		struct wait_queue_head *__wq_head;
1894 
1895 		/*
1896 		 * Only let prctl() wait / retry; don't unduly delay clone().
1897 		 */
1898 again:
1899 		__wq_head = __var_waitqueue(mm);
1900 		init_wait_var_entry(&__wbq_entry, mm, 0);
1901 		__wbq_entry.wq_entry.func = woken_wake_bit_function;
1902 		add_wait_queue(__wq_head, &__wbq_entry.wq_entry);
1903 
1904 		/*
1905 		 * add_wait_queue()		futex_ref_put()
1906 		 * MB (this)			MB (implied)
1907 		 * futex_pivot_pending()	wake_up_var()
1908 		 *                                waitqueue_active()
1909 		 *
1910 		 * Notably, it must not be possible to see
1911 		 * !futex_pivot_pending() && !waitqueue_active().
1912 		 */
1913 		smp_mb();
1914 
1915 		while (!futex_pivot_pending(mm) &&
1916 		       wait_woken(&__wbq_entry.wq_entry, TASK_UNINTERRUPTIBLE,
1917 				  MAX_SCHEDULE_TIMEOUT))
1918 			/* empty */;
1919 
1920 		remove_wait_queue(__wq_head, &__wbq_entry.wq_entry);
1921 	}
1922 
1923 	scoped_guard(mutex, &mm->futex.phash.lock) {
1924 		struct futex_private_hash *free __free(kvfree) = NULL;
1925 		struct futex_private_hash *cur, *new;
1926 
1927 		cur = rcu_dereference_protected(mm->futex.phash.hash,
1928 						lockdep_is_held(&mm->futex.phash.lock));
1929 		new = mm->futex.phash.hash_new;
1930 		mm->futex.phash.hash_new = NULL;
1931 
1932 		if (fph) {
1933 			if (cur && !cur->hash_mask) {
1934 				/*
1935 				 * If two threads simultaneously request the global
1936 				 * hash then the first one performs the switch,
1937 				 * the second one returns here.
1938 				 */
1939 				free = fph;
1940 				mm->futex.phash.hash_new = new;
1941 				return -EBUSY;
1942 			}
1943 			if (cur && !new) {
1944 				/*
1945 				 * If we have an existing hash, but do not yet have
1946 				 * allocated a replacement hash, drop the initial
1947 				 * reference on the existing hash.
1948 				 */
1949 				futex_ref_drop(cur);
1950 			}
1951 
1952 			if (new) {
1953 				/*
1954 				 * Two updates raced; throw out the lesser one.
1955 				 */
1956 				if (futex_hash_less(new, fph)) {
1957 					free = new;
1958 					new = fph;
1959 				} else {
1960 					free = fph;
1961 				}
1962 			} else {
1963 				new = fph;
1964 			}
1965 			fph = NULL;
1966 		}
1967 
1968 		if (new) {
1969 			/*
1970 			 * Will set mm->futex.phash.new_hash on failure;
1971 			 * futex_private_hash_get() will try again.
1972 			 */
1973 			if (!__futex_pivot_hash(mm, new) && custom)
1974 				goto again;
1975 		}
1976 	}
1977 	return 0;
1978 }
1979 
1980 int futex_hash_allocate_default(void)
1981 {
1982 	unsigned int threads, buckets, current_buckets = 0;
1983 	struct futex_private_hash *fph;
1984 
1985 	if (!current->mm)
1986 		return 0;
1987 
1988 	scoped_guard(rcu) {
1989 		threads = min_t(unsigned int, get_nr_threads(current), num_online_cpus());
1990 
1991 		fph = rcu_dereference(current->mm->futex.phash.hash);
1992 		if (fph) {
1993 			if (fph->custom)
1994 				return 0;
1995 
1996 			current_buckets = fph->hash_mask + 1;
1997 		}
1998 	}
1999 
2000 	/*
2001 	 * The default allocation will remain within
2002 	 *   16 <= threads * 4 <= global hash size
2003 	 */
2004 	buckets = roundup_pow_of_two(4 * threads);
2005 	buckets = clamp(buckets, 16, __futex_mask + 1);
2006 
2007 	if (current_buckets >= buckets)
2008 		return 0;
2009 
2010 	return futex_hash_allocate(buckets, 0);
2011 }
2012 
2013 static int futex_hash_get_slots(void)
2014 {
2015 	struct futex_private_hash *fph;
2016 
2017 	guard(rcu)();
2018 	fph = rcu_dereference(current->mm->futex.phash.hash);
2019 	if (fph && fph->hash_mask)
2020 		return fph->hash_mask + 1;
2021 	return 0;
2022 }
2023 #else  /* CONFIG_FUTEX_PRIVATE_HASH */
2024 static inline int futex_hash_allocate(unsigned int hslots, unsigned int flags) { return -EINVAL; }
2025 static inline int futex_hash_get_slots(void) { return 0; }
2026 static inline void futex_hash_init_mm(struct futex_mm_data *fd) { }
2027 #endif /* !CONFIG_FUTEX_PRIVATE_HASH */
2028 
2029 #ifdef CONFIG_FUTEX_ROBUST_UNLOCK
2030 static void futex_invalidate_cs_ranges(struct futex_mm_data *fd)
2031 {
2032 	/*
2033 	 * Invalidate start_ip so that the quick check fails for ip >= start_ip
2034 	 * if VDSO is not mapped or the second slot is not available for compat
2035 	 * tasks as they use VDSO32 which does not provide the 64-bit pointer
2036 	 * variant.
2037 	 */
2038 	for (int i = 0; i < FUTEX_ROBUST_MAX_CS_RANGES; i++)
2039 		fd->unlock.cs_ranges[i].start_ip = ~0UL;
2040 }
2041 
2042 void futex_reset_cs_ranges(struct futex_mm_data *fd)
2043 {
2044 	memset(fd->unlock.cs_ranges, 0, sizeof(fd->unlock.cs_ranges));
2045 	futex_invalidate_cs_ranges(fd);
2046 }
2047 
2048 static void futex_robust_unlock_init_mm(struct futex_mm_data *fd)
2049 {
2050 	/* mm_dup() preserves the range, mm_alloc() clears it */
2051 	if (!fd->unlock.cs_ranges[0].start_ip)
2052 		futex_invalidate_cs_ranges(fd);
2053 }
2054 #else  /* CONFIG_FUTEX_ROBUST_UNLOCK */
2055 static inline void futex_robust_unlock_init_mm(struct futex_mm_data *fd) { }
2056 #endif /* !CONFIG_FUTEX_ROBUST_UNLOCK */
2057 
2058 #if defined(CONFIG_FUTEX_PRIVATE_HASH) || defined(CONFIG_FUTEX_ROBUST_UNLOCK)
2059 void futex_mm_init(struct mm_struct *mm)
2060 {
2061 	futex_hash_init_mm(&mm->futex);
2062 	futex_robust_unlock_init_mm(&mm->futex);
2063 }
2064 #endif
2065 
2066 int futex_hash_prctl(unsigned long arg2, unsigned long arg3, unsigned long arg4)
2067 {
2068 	unsigned int flags = FH_CUSTOM;
2069 	int ret;
2070 
2071 	switch (arg2) {
2072 	case PR_FUTEX_HASH_SET_SLOTS:
2073 		if (arg4)
2074 			return -EINVAL;
2075 		ret = futex_hash_allocate(arg3, flags);
2076 		break;
2077 
2078 	case PR_FUTEX_HASH_GET_SLOTS:
2079 		ret = futex_hash_get_slots();
2080 		break;
2081 
2082 	default:
2083 		ret = -EINVAL;
2084 		break;
2085 	}
2086 	return ret;
2087 }
2088 
2089 static int __init futex_init(void)
2090 {
2091 	unsigned long hashsize, i;
2092 	unsigned int order, n;
2093 	unsigned long size;
2094 
2095 #ifdef CONFIG_BASE_SMALL
2096 	hashsize = 16;
2097 #else
2098 	hashsize = 256 * num_possible_cpus();
2099 	hashsize /= num_possible_nodes();
2100 	hashsize = max(4, hashsize);
2101 	hashsize = roundup_pow_of_two(hashsize);
2102 #endif
2103 	__futex_mask = hashsize - 1;
2104 	__futex_shift = ilog2(hashsize);
2105 	size = sizeof(struct futex_hash_bucket) * hashsize;
2106 	order = get_order(size);
2107 
2108 	__futex_queues = kzalloc_objs(*__futex_queues, nr_node_ids);
2109 	kmemleak_not_leak(__futex_queues);
2110 
2111 	runtime_const_init(shift, __futex_shift);
2112 	runtime_const_init(mask,  __futex_mask);
2113 	runtime_const_init(ptr,   __futex_queues);
2114 
2115 	barrier();
2116 
2117 	BUG_ON(!futex_queues());
2118 
2119 	for_each_node(n) {
2120 		struct futex_hash_bucket *table;
2121 
2122 		if (order > MAX_PAGE_ORDER)
2123 			table = vmalloc_huge_node(size, GFP_KERNEL, n);
2124 		else
2125 			table = alloc_pages_exact_nid(n, size, GFP_KERNEL);
2126 
2127 		BUG_ON(!table);
2128 
2129 		for (i = 0; i < hashsize; i++)
2130 			futex_hash_bucket_init(&table[i]);
2131 
2132 		futex_queues()[n] = table;
2133 	}
2134 
2135 	pr_info("futex hash table entries: %lu (%lu bytes on %d NUMA nodes, total %lu KiB, %s).\n",
2136 		hashsize, size, num_possible_nodes(), size * num_possible_nodes() / 1024,
2137 		order > MAX_PAGE_ORDER ? "vmalloc" : "linear");
2138 	return 0;
2139 }
2140 core_initcall(futex_init);
2141