waitwake.c 20 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734
  1. // SPDX-License-Identifier: GPL-2.0-or-later
  2. #include <linux/plist.h>
  3. #include <linux/sched/task.h>
  4. #include <linux/sched/signal.h>
  5. #include <linux/freezer.h>
  6. #include "futex.h"
  7. /*
  8. * READ this before attempting to hack on futexes!
  9. *
  10. * Basic futex operation and ordering guarantees
  11. * =============================================
  12. *
  13. * The waiter reads the futex value in user space and calls
  14. * futex_wait(). This function computes the hash bucket and acquires
  15. * the hash bucket lock. After that it reads the futex user space value
  16. * again and verifies that the data has not changed. If it has not changed
  17. * it enqueues itself into the hash bucket, releases the hash bucket lock
  18. * and schedules.
  19. *
  20. * The waker side modifies the user space value of the futex and calls
  21. * futex_wake(). This function computes the hash bucket and acquires the
  22. * hash bucket lock. Then it looks for waiters on that futex in the hash
  23. * bucket and wakes them.
  24. *
  25. * In futex wake up scenarios where no tasks are blocked on a futex, taking
  26. * the hb spinlock can be avoided and simply return. In order for this
  27. * optimization to work, ordering guarantees must exist so that the waiter
  28. * being added to the list is acknowledged when the list is concurrently being
  29. * checked by the waker, avoiding scenarios like the following:
  30. *
  31. * CPU 0 CPU 1
  32. * val = *futex;
  33. * sys_futex(WAIT, futex, val);
  34. * futex_wait(futex, val);
  35. * uval = *futex;
  36. * *futex = newval;
  37. * sys_futex(WAKE, futex);
  38. * futex_wake(futex);
  39. * if (queue_empty())
  40. * return;
  41. * if (uval == val)
  42. * lock(hash_bucket(futex));
  43. * queue();
  44. * unlock(hash_bucket(futex));
  45. * schedule();
  46. *
  47. * This would cause the waiter on CPU 0 to wait forever because it
  48. * missed the transition of the user space value from val to newval
  49. * and the waker did not find the waiter in the hash bucket queue.
  50. *
  51. * The correct serialization ensures that a waiter either observes
  52. * the changed user space value before blocking or is woken by a
  53. * concurrent waker:
  54. *
  55. * CPU 0 CPU 1
  56. * val = *futex;
  57. * sys_futex(WAIT, futex, val);
  58. * futex_wait(futex, val);
  59. *
  60. * waiters++; (a)
  61. * smp_mb(); (A) <-- paired with -.
  62. * |
  63. * lock(hash_bucket(futex)); |
  64. * |
  65. * uval = *futex; |
  66. * | *futex = newval;
  67. * | sys_futex(WAKE, futex);
  68. * | futex_wake(futex);
  69. * |
  70. * `--------> smp_mb(); (B)
  71. * if (uval == val)
  72. * queue();
  73. * unlock(hash_bucket(futex));
  74. * schedule(); if (waiters)
  75. * lock(hash_bucket(futex));
  76. * else wake_waiters(futex);
  77. * waiters--; (b) unlock(hash_bucket(futex));
  78. *
  79. * Where (A) orders the waiters increment and the futex value read through
  80. * atomic operations (see futex_hb_waiters_inc) and where (B) orders the write
  81. * to futex and the waiters read (see futex_hb_waiters_pending()).
  82. *
  83. * This yields the following case (where X:=waiters, Y:=futex):
  84. *
  85. * X = Y = 0
  86. *
  87. * w[X]=1 w[Y]=1
  88. * MB MB
  89. * r[Y]=y r[X]=x
  90. *
  91. * Which guarantees that x==0 && y==0 is impossible; which translates back into
  92. * the guarantee that we cannot both miss the futex variable change and the
  93. * enqueue.
  94. *
  95. * Note that a new waiter is accounted for in (a) even when it is possible that
  96. * the wait call can return error, in which case we backtrack from it in (b).
  97. * Refer to the comment in futex_q_lock().
  98. *
  99. * Similarly, in order to account for waiters being requeued on another
  100. * address we always increment the waiters for the destination bucket before
  101. * acquiring the lock. It then decrements them again after releasing it -
  102. * the code that actually moves the futex(es) between hash buckets (requeue_futex)
  103. * will do the additional required waiter count housekeeping. This is done for
  104. * double_lock_hb() and double_unlock_hb(), respectively.
  105. */
  106. bool __futex_wake_mark(struct futex_q *q)
  107. {
  108. if (WARN(q->pi_state || q->rt_waiter, "refusing to wake PI futex\n"))
  109. return false;
  110. __futex_unqueue(q);
  111. /*
  112. * The waiting task can free the futex_q as soon as q->lock_ptr = NULL
  113. * is written, without taking any locks. This is possible in the event
  114. * of a spurious wakeup, for example. A memory barrier is required here
  115. * to prevent the following store to lock_ptr from getting ahead of the
  116. * plist_del in __futex_unqueue().
  117. */
  118. smp_store_release(&q->lock_ptr, NULL);
  119. return true;
  120. }
  121. /*
  122. * The hash bucket lock must be held when this is called.
  123. * Afterwards, the futex_q must not be accessed. Callers
  124. * must ensure to later call wake_up_q() for the actual
  125. * wakeups to occur.
  126. */
  127. void futex_wake_mark(struct wake_q_head *wake_q, struct futex_q *q)
  128. {
  129. struct task_struct *p = q->task;
  130. get_task_struct(p);
  131. if (!__futex_wake_mark(q)) {
  132. put_task_struct(p);
  133. return;
  134. }
  135. /*
  136. * Queue the task for later wakeup for after we've released
  137. * the hb->lock.
  138. */
  139. wake_q_add_safe(wake_q, p);
  140. }
  141. /*
  142. * Wake up waiters matching bitset queued on this futex (uaddr).
  143. */
  144. int futex_wake(u32 __user *uaddr, unsigned int flags, int nr_wake, u32 bitset)
  145. {
  146. struct futex_hash_bucket *hb;
  147. struct futex_q *this, *next;
  148. union futex_key key = FUTEX_KEY_INIT;
  149. DEFINE_WAKE_Q(wake_q);
  150. int ret;
  151. if (!bitset)
  152. return -EINVAL;
  153. ret = get_futex_key(uaddr, flags, &key, FUTEX_READ);
  154. if (unlikely(ret != 0))
  155. return ret;
  156. if ((flags & FLAGS_STRICT) && !nr_wake)
  157. return 0;
  158. hb = futex_hash(&key);
  159. /* Make sure we really have tasks to wakeup */
  160. if (!futex_hb_waiters_pending(hb))
  161. return ret;
  162. spin_lock(&hb->lock);
  163. plist_for_each_entry_safe(this, next, &hb->chain, list) {
  164. if (futex_match (&this->key, &key)) {
  165. if (this->pi_state || this->rt_waiter) {
  166. ret = -EINVAL;
  167. break;
  168. }
  169. /* Check if one of the bits is set in both bitsets */
  170. if (!(this->bitset & bitset))
  171. continue;
  172. this->wake(&wake_q, this);
  173. if (++ret >= nr_wake)
  174. break;
  175. }
  176. }
  177. spin_unlock(&hb->lock);
  178. wake_up_q(&wake_q);
  179. return ret;
  180. }
  181. static int futex_atomic_op_inuser(unsigned int encoded_op, u32 __user *uaddr)
  182. {
  183. unsigned int op = (encoded_op & 0x70000000) >> 28;
  184. unsigned int cmp = (encoded_op & 0x0f000000) >> 24;
  185. int oparg = sign_extend32((encoded_op & 0x00fff000) >> 12, 11);
  186. int cmparg = sign_extend32(encoded_op & 0x00000fff, 11);
  187. int oldval, ret;
  188. if (encoded_op & (FUTEX_OP_OPARG_SHIFT << 28)) {
  189. if (oparg < 0 || oparg > 31) {
  190. char comm[sizeof(current->comm)];
  191. /*
  192. * kill this print and return -EINVAL when userspace
  193. * is sane again
  194. */
  195. pr_info_ratelimited("futex_wake_op: %s tries to shift op by %d; fix this program\n",
  196. get_task_comm(comm, current), oparg);
  197. oparg &= 31;
  198. }
  199. oparg = 1 << oparg;
  200. }
  201. pagefault_disable();
  202. ret = arch_futex_atomic_op_inuser(op, oparg, &oldval, uaddr);
  203. pagefault_enable();
  204. if (ret)
  205. return ret;
  206. switch (cmp) {
  207. case FUTEX_OP_CMP_EQ:
  208. return oldval == cmparg;
  209. case FUTEX_OP_CMP_NE:
  210. return oldval != cmparg;
  211. case FUTEX_OP_CMP_LT:
  212. return oldval < cmparg;
  213. case FUTEX_OP_CMP_GE:
  214. return oldval >= cmparg;
  215. case FUTEX_OP_CMP_LE:
  216. return oldval <= cmparg;
  217. case FUTEX_OP_CMP_GT:
  218. return oldval > cmparg;
  219. default:
  220. return -ENOSYS;
  221. }
  222. }
  223. /*
  224. * Wake up all waiters hashed on the physical page that is mapped
  225. * to this virtual address:
  226. */
  227. int futex_wake_op(u32 __user *uaddr1, unsigned int flags, u32 __user *uaddr2,
  228. int nr_wake, int nr_wake2, int op)
  229. {
  230. union futex_key key1 = FUTEX_KEY_INIT, key2 = FUTEX_KEY_INIT;
  231. struct futex_hash_bucket *hb1, *hb2;
  232. struct futex_q *this, *next;
  233. int ret, op_ret;
  234. DEFINE_WAKE_Q(wake_q);
  235. retry:
  236. ret = get_futex_key(uaddr1, flags, &key1, FUTEX_READ);
  237. if (unlikely(ret != 0))
  238. return ret;
  239. ret = get_futex_key(uaddr2, flags, &key2, FUTEX_WRITE);
  240. if (unlikely(ret != 0))
  241. return ret;
  242. hb1 = futex_hash(&key1);
  243. hb2 = futex_hash(&key2);
  244. retry_private:
  245. double_lock_hb(hb1, hb2);
  246. op_ret = futex_atomic_op_inuser(op, uaddr2);
  247. if (unlikely(op_ret < 0)) {
  248. double_unlock_hb(hb1, hb2);
  249. if (!IS_ENABLED(CONFIG_MMU) ||
  250. unlikely(op_ret != -EFAULT && op_ret != -EAGAIN)) {
  251. /*
  252. * we don't get EFAULT from MMU faults if we don't have
  253. * an MMU, but we might get them from range checking
  254. */
  255. ret = op_ret;
  256. return ret;
  257. }
  258. if (op_ret == -EFAULT) {
  259. ret = fault_in_user_writeable(uaddr2);
  260. if (ret)
  261. return ret;
  262. }
  263. cond_resched();
  264. if (!(flags & FLAGS_SHARED))
  265. goto retry_private;
  266. goto retry;
  267. }
  268. plist_for_each_entry_safe(this, next, &hb1->chain, list) {
  269. if (futex_match (&this->key, &key1)) {
  270. if (this->pi_state || this->rt_waiter) {
  271. ret = -EINVAL;
  272. goto out_unlock;
  273. }
  274. this->wake(&wake_q, this);
  275. if (++ret >= nr_wake)
  276. break;
  277. }
  278. }
  279. if (op_ret > 0) {
  280. op_ret = 0;
  281. plist_for_each_entry_safe(this, next, &hb2->chain, list) {
  282. if (futex_match (&this->key, &key2)) {
  283. if (this->pi_state || this->rt_waiter) {
  284. ret = -EINVAL;
  285. goto out_unlock;
  286. }
  287. this->wake(&wake_q, this);
  288. if (++op_ret >= nr_wake2)
  289. break;
  290. }
  291. }
  292. ret += op_ret;
  293. }
  294. out_unlock:
  295. double_unlock_hb(hb1, hb2);
  296. wake_up_q(&wake_q);
  297. return ret;
  298. }
  299. static long futex_wait_restart(struct restart_block *restart);
  300. /**
  301. * futex_wait_queue() - futex_queue() and wait for wakeup, timeout, or signal
  302. * @hb: the futex hash bucket, must be locked by the caller
  303. * @q: the futex_q to queue up on
  304. * @timeout: the prepared hrtimer_sleeper, or null for no timeout
  305. */
  306. void futex_wait_queue(struct futex_hash_bucket *hb, struct futex_q *q,
  307. struct hrtimer_sleeper *timeout)
  308. {
  309. /*
  310. * The task state is guaranteed to be set before another task can
  311. * wake it. set_current_state() is implemented using smp_store_mb() and
  312. * futex_queue() calls spin_unlock() upon completion, both serializing
  313. * access to the hash list and forcing another memory barrier.
  314. */
  315. set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE);
  316. futex_queue(q, hb, current);
  317. /* Arm the timer */
  318. if (timeout)
  319. hrtimer_sleeper_start_expires(timeout, HRTIMER_MODE_ABS);
  320. /*
  321. * If we have been removed from the hash list, then another task
  322. * has tried to wake us, and we can skip the call to schedule().
  323. */
  324. if (likely(!plist_node_empty(&q->list))) {
  325. /*
  326. * If the timer has already expired, current will already be
  327. * flagged for rescheduling. Only call schedule if there
  328. * is no timeout, or if it has yet to expire.
  329. */
  330. if (!timeout || timeout->task)
  331. schedule();
  332. }
  333. __set_current_state(TASK_RUNNING);
  334. }
  335. /**
  336. * futex_unqueue_multiple - Remove various futexes from their hash bucket
  337. * @v: The list of futexes to unqueue
  338. * @count: Number of futexes in the list
  339. *
  340. * Helper to unqueue a list of futexes. This can't fail.
  341. *
  342. * Return:
  343. * - >=0 - Index of the last futex that was awoken;
  344. * - -1 - No futex was awoken
  345. */
  346. int futex_unqueue_multiple(struct futex_vector *v, int count)
  347. {
  348. int ret = -1, i;
  349. for (i = 0; i < count; i++) {
  350. if (!futex_unqueue(&v[i].q))
  351. ret = i;
  352. }
  353. return ret;
  354. }
  355. /**
  356. * futex_wait_multiple_setup - Prepare to wait and enqueue multiple futexes
  357. * @vs: The futex list to wait on
  358. * @count: The size of the list
  359. * @woken: Index of the last woken futex, if any. Used to notify the
  360. * caller that it can return this index to userspace (return parameter)
  361. *
  362. * Prepare multiple futexes in a single step and enqueue them. This may fail if
  363. * the futex list is invalid or if any futex was already awoken. On success the
  364. * task is ready to interruptible sleep.
  365. *
  366. * Return:
  367. * - 1 - One of the futexes was woken by another thread
  368. * - 0 - Success
  369. * - <0 - -EFAULT, -EWOULDBLOCK or -EINVAL
  370. */
  371. int futex_wait_multiple_setup(struct futex_vector *vs, int count, int *woken)
  372. {
  373. struct futex_hash_bucket *hb;
  374. bool retry = false;
  375. int ret, i;
  376. u32 uval;
  377. /*
  378. * Enqueuing multiple futexes is tricky, because we need to enqueue
  379. * each futex on the list before dealing with the next one to avoid
  380. * deadlocking on the hash bucket. But, before enqueuing, we need to
  381. * make sure that current->state is TASK_INTERRUPTIBLE, so we don't
  382. * lose any wake events, which cannot be done before the get_futex_key
  383. * of the next key, because it calls get_user_pages, which can sleep.
  384. * Thus, we fetch the list of futexes keys in two steps, by first
  385. * pinning all the memory keys in the futex key, and only then we read
  386. * each key and queue the corresponding futex.
  387. *
  388. * Private futexes doesn't need to recalculate hash in retry, so skip
  389. * get_futex_key() when retrying.
  390. */
  391. retry:
  392. for (i = 0; i < count; i++) {
  393. if (!(vs[i].w.flags & FLAGS_SHARED) && retry)
  394. continue;
  395. ret = get_futex_key(u64_to_user_ptr(vs[i].w.uaddr),
  396. vs[i].w.flags,
  397. &vs[i].q.key, FUTEX_READ);
  398. if (unlikely(ret))
  399. return ret;
  400. }
  401. set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE);
  402. for (i = 0; i < count; i++) {
  403. u32 __user *uaddr = (u32 __user *)(unsigned long)vs[i].w.uaddr;
  404. struct futex_q *q = &vs[i].q;
  405. u32 val = vs[i].w.val;
  406. hb = futex_q_lock(q);
  407. ret = futex_get_value_locked(&uval, uaddr);
  408. if (!ret && uval == val) {
  409. /*
  410. * The bucket lock can't be held while dealing with the
  411. * next futex. Queue each futex at this moment so hb can
  412. * be unlocked.
  413. */
  414. futex_queue(q, hb, current);
  415. continue;
  416. }
  417. futex_q_unlock(hb);
  418. __set_current_state(TASK_RUNNING);
  419. /*
  420. * Even if something went wrong, if we find out that a futex
  421. * was woken, we don't return error and return this index to
  422. * userspace
  423. */
  424. *woken = futex_unqueue_multiple(vs, i);
  425. if (*woken >= 0)
  426. return 1;
  427. if (ret) {
  428. /*
  429. * If we need to handle a page fault, we need to do so
  430. * without any lock and any enqueued futex (otherwise
  431. * we could lose some wakeup). So we do it here, after
  432. * undoing all the work done so far. In success, we
  433. * retry all the work.
  434. */
  435. if (get_user(uval, uaddr))
  436. return -EFAULT;
  437. retry = true;
  438. goto retry;
  439. }
  440. if (uval != val)
  441. return -EWOULDBLOCK;
  442. }
  443. return 0;
  444. }
  445. /**
  446. * futex_sleep_multiple - Check sleeping conditions and sleep
  447. * @vs: List of futexes to wait for
  448. * @count: Length of vs
  449. * @to: Timeout
  450. *
  451. * Sleep if and only if the timeout hasn't expired and no futex on the list has
  452. * been woken up.
  453. */
  454. static void futex_sleep_multiple(struct futex_vector *vs, unsigned int count,
  455. struct hrtimer_sleeper *to)
  456. {
  457. if (to && !to->task)
  458. return;
  459. for (; count; count--, vs++) {
  460. if (!READ_ONCE(vs->q.lock_ptr))
  461. return;
  462. }
  463. schedule();
  464. }
  465. /**
  466. * futex_wait_multiple - Prepare to wait on and enqueue several futexes
  467. * @vs: The list of futexes to wait on
  468. * @count: The number of objects
  469. * @to: Timeout before giving up and returning to userspace
  470. *
  471. * Entry point for the FUTEX_WAIT_MULTIPLE futex operation, this function
  472. * sleeps on a group of futexes and returns on the first futex that is
  473. * wake, or after the timeout has elapsed.
  474. *
  475. * Return:
  476. * - >=0 - Hint to the futex that was awoken
  477. * - <0 - On error
  478. */
  479. int futex_wait_multiple(struct futex_vector *vs, unsigned int count,
  480. struct hrtimer_sleeper *to)
  481. {
  482. int ret, hint = 0;
  483. if (to)
  484. hrtimer_sleeper_start_expires(to, HRTIMER_MODE_ABS);
  485. while (1) {
  486. ret = futex_wait_multiple_setup(vs, count, &hint);
  487. if (ret) {
  488. if (ret > 0) {
  489. /* A futex was woken during setup */
  490. ret = hint;
  491. }
  492. return ret;
  493. }
  494. futex_sleep_multiple(vs, count, to);
  495. __set_current_state(TASK_RUNNING);
  496. ret = futex_unqueue_multiple(vs, count);
  497. if (ret >= 0)
  498. return ret;
  499. if (to && !to->task)
  500. return -ETIMEDOUT;
  501. else if (signal_pending(current))
  502. return -ERESTARTSYS;
  503. /*
  504. * The final case is a spurious wakeup, for
  505. * which just retry.
  506. */
  507. }
  508. }
  509. /**
  510. * futex_wait_setup() - Prepare to wait on a futex
  511. * @uaddr: the futex userspace address
  512. * @val: the expected value
  513. * @flags: futex flags (FLAGS_SHARED, etc.)
  514. * @q: the associated futex_q
  515. * @hb: storage for hash_bucket pointer to be returned to caller
  516. *
  517. * Setup the futex_q and locate the hash_bucket. Get the futex value and
  518. * compare it with the expected value. Handle atomic faults internally.
  519. * Return with the hb lock held on success, and unlocked on failure.
  520. *
  521. * Return:
  522. * - 0 - uaddr contains val and hb has been locked;
  523. * - <1 - -EFAULT or -EWOULDBLOCK (uaddr does not contain val) and hb is unlocked
  524. */
  525. int futex_wait_setup(u32 __user *uaddr, u32 val, unsigned int flags,
  526. struct futex_q *q, struct futex_hash_bucket **hb)
  527. {
  528. u32 uval;
  529. int ret;
  530. /*
  531. * Access the page AFTER the hash-bucket is locked.
  532. * Order is important:
  533. *
  534. * Userspace waiter: val = var; if (cond(val)) futex_wait(&var, val);
  535. * Userspace waker: if (cond(var)) { var = new; futex_wake(&var); }
  536. *
  537. * The basic logical guarantee of a futex is that it blocks ONLY
  538. * if cond(var) is known to be true at the time of blocking, for
  539. * any cond. If we locked the hash-bucket after testing *uaddr, that
  540. * would open a race condition where we could block indefinitely with
  541. * cond(var) false, which would violate the guarantee.
  542. *
  543. * On the other hand, we insert q and release the hash-bucket only
  544. * after testing *uaddr. This guarantees that futex_wait() will NOT
  545. * absorb a wakeup if *uaddr does not match the desired values
  546. * while the syscall executes.
  547. */
  548. retry:
  549. ret = get_futex_key(uaddr, flags, &q->key, FUTEX_READ);
  550. if (unlikely(ret != 0))
  551. return ret;
  552. retry_private:
  553. *hb = futex_q_lock(q);
  554. ret = futex_get_value_locked(&uval, uaddr);
  555. if (ret) {
  556. futex_q_unlock(*hb);
  557. ret = get_user(uval, uaddr);
  558. if (ret)
  559. return ret;
  560. if (!(flags & FLAGS_SHARED))
  561. goto retry_private;
  562. goto retry;
  563. }
  564. if (uval != val) {
  565. futex_q_unlock(*hb);
  566. ret = -EWOULDBLOCK;
  567. }
  568. return ret;
  569. }
  570. int __futex_wait(u32 __user *uaddr, unsigned int flags, u32 val,
  571. struct hrtimer_sleeper *to, u32 bitset)
  572. {
  573. struct futex_q q = futex_q_init;
  574. struct futex_hash_bucket *hb;
  575. int ret;
  576. if (!bitset)
  577. return -EINVAL;
  578. q.bitset = bitset;
  579. retry:
  580. /*
  581. * Prepare to wait on uaddr. On success, it holds hb->lock and q
  582. * is initialized.
  583. */
  584. ret = futex_wait_setup(uaddr, val, flags, &q, &hb);
  585. if (ret)
  586. return ret;
  587. /* futex_queue and wait for wakeup, timeout, or a signal. */
  588. futex_wait_queue(hb, &q, to);
  589. /* If we were woken (and unqueued), we succeeded, whatever. */
  590. if (!futex_unqueue(&q))
  591. return 0;
  592. if (to && !to->task)
  593. return -ETIMEDOUT;
  594. /*
  595. * We expect signal_pending(current), but we might be the
  596. * victim of a spurious wakeup as well.
  597. */
  598. if (!signal_pending(current))
  599. goto retry;
  600. return -ERESTARTSYS;
  601. }
  602. int futex_wait(u32 __user *uaddr, unsigned int flags, u32 val, ktime_t *abs_time, u32 bitset)
  603. {
  604. struct hrtimer_sleeper timeout, *to;
  605. struct restart_block *restart;
  606. int ret;
  607. to = futex_setup_timer(abs_time, &timeout, flags,
  608. current->timer_slack_ns);
  609. ret = __futex_wait(uaddr, flags, val, to, bitset);
  610. /* No timeout, nothing to clean up. */
  611. if (!to)
  612. return ret;
  613. hrtimer_cancel(&to->timer);
  614. destroy_hrtimer_on_stack(&to->timer);
  615. if (ret == -ERESTARTSYS) {
  616. restart = &current->restart_block;
  617. restart->futex.uaddr = uaddr;
  618. restart->futex.val = val;
  619. restart->futex.time = *abs_time;
  620. restart->futex.bitset = bitset;
  621. restart->futex.flags = flags | FLAGS_HAS_TIMEOUT;
  622. return set_restart_fn(restart, futex_wait_restart);
  623. }
  624. return ret;
  625. }
  626. static long futex_wait_restart(struct restart_block *restart)
  627. {
  628. u32 __user *uaddr = restart->futex.uaddr;
  629. ktime_t t, *tp = NULL;
  630. if (restart->futex.flags & FLAGS_HAS_TIMEOUT) {
  631. t = restart->futex.time;
  632. tp = &t;
  633. }
  634. restart->fn = do_no_restart_syscall;
  635. return (long)futex_wait(uaddr, restart->futex.flags,
  636. restart->futex.val, tp, restart->futex.bitset);
  637. }