4646#include <linux/slab.h>
4747#include <linux/vmalloc.h>
4848#include <linux/kmemleak.h>
49+ #include <linux/wait_bit.h>
4950
5051#include <vdso/futex.h>
5152
@@ -1527,44 +1528,59 @@ static void futex_cleanup_begin(struct task_struct *tsk)
15271528 raw_spin_unlock_irq (& tsk -> pi_lock );
15281529}
15291530
1530- static void futex_cleanup_end (struct task_struct * tsk , int state )
1531+ static void futex_cleanup_end (struct task_struct * tsk )
15311532 __releases (& tsk - > futex .exit_mutex )
15321533{
1533- /*
1534- * Lockless store. The only side effect is that an observer might
1535- * take another loop until it becomes visible.
1536- */
1537- tsk -> futex .state = state ;
1534+ scoped_guard (raw_spinlock_irq , & tsk -> pi_lock )
1535+ tsk -> futex .state = FUTEX_STATE_DEAD ;
1536+
15381537 /*
15391538 * Drop the exit protection. This unblocks waiters which observed
15401539 * FUTEX_STATE_EXITING to reevaluate the state.
15411540 */
15421541 mutex_unlock (& tsk -> futex .exit_mutex );
15431542}
15441543
1545- void futex_exec_release (struct task_struct * tsk )
1544+ /*
1545+ * Invoked from mm_exit_exec_release() to cleanup the robust lists and pi state
1546+ * of the outgoing task.
1547+ *
1548+ * exec() makes it interesting for futexes because the TID of the task stays the
1549+ * same, but from a futex perspective the task has to be treated like an exiting
1550+ * task. This is especially important for the sanity check for private futexes
1551+ * in attach_to_pi_owner() which compares the owner's mm with the waiter's mm.
1552+ *
1553+ * That check would give the wrong answer if futex_cleanup_end() would
1554+ * set the state to FUTEX_STATE_OK as long as the task still has the old
1555+ * mm.
1556+ *
1557+ * After the task has switched to the new mm it sets it to
1558+ * FUTEX_STATE_OK again in futex_exec_done().
1559+ */
1560+ void futex_exit_exec_release (struct task_struct * tsk )
15461561{
1547- /*
1548- * The state handling is done for consistency, but in the case of
1549- * exec() there is no way to prevent further damage as the PID stays
1550- * the same. But for the unlikely and arguably buggy case that a
1551- * futex is held on exec(), this provides at least as much state
1552- * consistency protection which is possible.
1553- */
15541562 futex_cleanup_begin (tsk );
15551563 futex_cleanup (tsk );
1556- /*
1557- * Reset the state to FUTEX_STATE_OK. The task is alive and about
1558- * exec a new binary.
1559- */
1560- futex_cleanup_end (tsk , FUTEX_STATE_OK );
1564+ futex_cleanup_end (tsk );
15611565}
15621566
1563- void futex_exit_release (struct task_struct * tsk )
1567+ /*
1568+ * exec() has switched to the new mm. Futex operations are safe again.
1569+ */
1570+ void futex_exec_done (struct task_struct * tsk )
15641571{
1565- futex_cleanup_begin (tsk );
1566- futex_cleanup (tsk );
1567- futex_cleanup_end (tsk , FUTEX_STATE_DEAD );
1572+ /*
1573+ * This store does not have to take tsk::futex::exit_mutex because the
1574+ * phase where waiters block on it during state FUTEX_STATE_EXITING has
1575+ * been finished when futex_cleanup_end() set the state to
1576+ * FUTEX_STATE_DEAD.
1577+ *
1578+ * This transitions back from FUTEX_STATE_DEAD to FUTEX_STATE_OK. The
1579+ * ordering guarantee required here is that the previous store to
1580+ * tsk::mm in the calling code cannot be reordered against this store.
1581+ */
1582+ guard (raw_spinlock_irq )(& tsk -> pi_lock );
1583+ tsk -> futex .state = FUTEX_STATE_OK ;
15681584}
15691585
15701586static void futex_hash_bucket_init (struct futex_hash_bucket * fhb )
@@ -1844,14 +1860,18 @@ static int futex_hash_allocate(unsigned int hash_slots, unsigned int flags)
18441860 }
18451861
18461862 if (!mm -> futex .phash .ref ) {
1863+ unsigned int __percpu * ref = alloc_percpu (unsigned int );
1864+
1865+ if (!ref )
1866+ return - ENOMEM ;
1867+
18471868 /*
1848- * This will always be allocated by the first thread and
1849- * therefore requires no locking .
1869+ * Tasks sharing the mm can run this concurrently, so take the
1870+ * initial reference before publishing the counter .
18501871 */
1851- mm -> futex .phash .ref = alloc_percpu (unsigned int );
1852- if (!mm -> futex .phash .ref )
1853- return - ENOMEM ;
1854- this_cpu_inc (* mm -> futex .phash .ref ); /* 0 -> 1 */
1872+ this_cpu_inc (* ref ); /* 0 -> 1 */
1873+ if (cmpxchg (& mm -> futex .phash .ref , NULL , ref ))
1874+ free_percpu (ref );
18551875 }
18561876
18571877 fph = kvzalloc (struct_size (fph , queues , hash_slots ),
@@ -1867,11 +1887,35 @@ static int futex_hash_allocate(unsigned int hash_slots, unsigned int flags)
18671887 futex_hash_bucket_init (& fph -> queues [i ]);
18681888
18691889 if (custom ) {
1890+ struct wait_bit_queue_entry __wbq_entry ;
1891+ struct wait_queue_head * __wq_head ;
1892+
18701893 /*
18711894 * Only let prctl() wait / retry; don't unduly delay clone().
18721895 */
18731896again :
1874- wait_var_event (mm , futex_pivot_pending (mm ));
1897+ __wq_head = __var_waitqueue (mm );
1898+ init_wait_var_entry (& __wbq_entry , mm , 0 );
1899+ __wbq_entry .wq_entry .func = woken_wake_bit_function ;
1900+ add_wait_queue (__wq_head , & __wbq_entry .wq_entry );
1901+
1902+ /*
1903+ * add_wait_queue() futex_ref_put()
1904+ * MB (this) MB (implied)
1905+ * futex_pivot_pending() wake_up_var()
1906+ * waitqueue_active()
1907+ *
1908+ * Notably, it must not be possible to see
1909+ * !futex_pivot_pending() && !waitqueue_active().
1910+ */
1911+ smp_mb ();
1912+
1913+ while (!futex_pivot_pending (mm ) &&
1914+ wait_woken (& __wbq_entry .wq_entry , TASK_UNINTERRUPTIBLE ,
1915+ MAX_SCHEDULE_TIMEOUT ))
1916+ /* empty */ ;
1917+
1918+ remove_wait_queue (__wq_head , & __wbq_entry .wq_entry );
18751919 }
18761920
18771921 scoped_guard (mutex , & mm -> futex .phash .lock ) {
0 commit comments