From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pl1-f197.google.com (mail-pl1-f197.google.com [209.85.214.197]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 40B273845B0 for ; Thu, 17 Sep 2026 04:33:55 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.214.197 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619636; cv=none; b=u3KJAh1AqkXr1DayUG4hZy1IY87eFy+eXDPaCmal0DJtQH7QuoSX2tgHVwLJWGVW7Tr+hVwtx/c9s6m8KvmmDZ6yJ+nXUxyIcqZWdT8Sw7+LztTwCJhGLs4RwWJ7QTi4WAfBi2IBsMXU/7m7rV8Svetm0a/Tppo0PGQ2eHgDmz4= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619636; c=relaxed/simple; bh=KaT9NI0t5qrzBJ05W03jZ2i9577k9kTBGqpTJPCzuf8=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=nC9Jjc4MFQIni8mOTiIGbJl5K4nCHe4rWT+kDmsQIGhvxsQZRaGL80KxrJQ/VyMToJs+6ifOMiCCc/Lafq2ERNZCEaIFWOy+TIdlZfbqfJGf2IFCRw9bXIFy6rOk+JdEUQxFvPkLpJZ9r9kfb3UgVDJ2ihYKWVUXZOytvfODnu0= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=CZfRZlMg; arc=none smtp.client-ip=209.85.214.197 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="CZfRZlMg" Received: by mail-pl1-f197.google.com with SMTP id d9443c01a7336-2dd7d0751efso4700895ad.0 for ; Wed, 16 Sep 2026 21:33:55 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619635; x=1790224435; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=sdNyvCbcOZhKLcqXh8qFyV08oHNPuc9KtGoMTg2NP9I=; b=CZfRZlMgoqBPr4q4N3dStkfk/aeqjW1RkpS1/q+985/IbIwYj1Tb/r/Vo6UfEv/VzM YPe5f6AZtwG2x1tE4FdQgZXQECfYF7Ma5tZMNCEFGTdgyTCiKuyFAEeDOcAy6qa+zEJg lUAQY5HyDFvghswufc/rVhXsz2b+8JJ+KtUll1aTQPsfRPP9KGHsDUTQLjmP8cTjAUvs TNmuWWE1C8GvMHT9W+peDBo+MybeEgSX7PkJ03FVEQ5yuypYemxQXZqbTMgiHBOYKO8f zhm7+1fBXjRBaS/w2OWEVJTqIQB+brWdVTbPMxqwGQ3/b1AHFbfFHRJabMPvKu2C6xEQ 8Sng== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619635; x=1790224435; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=sdNyvCbcOZhKLcqXh8qFyV08oHNPuc9KtGoMTg2NP9I=; b=d6Ee7HkX+FI7IaPcQpcDR16yNSMAmc4nAGXZhRTdo6WzuX725KynDSnqwHmYvEkBth fm3LvFVTWiBr0Vam2P/j+cWtVUcdEWoAGlGF7ZdwLkjRtf39ZBCg1JJLUaTL8uHP+/7S cEI5Gp/0czPq07W3heCAYQKwKAxswZOQe2sOimKXY18Nn+YG4rXasrYv+PJ4pUMpI9EV V5XPL/5VFt2FswmO80GySkbK+27JgRRa/6nYFYEp34d1lh5g31OZg8XB9ay8OF2g64cE 5EyIzYrL+lZm5XFuP0lR4BOjHvnKN8PM4kfP9GUyw0iPhYQA5pW190qXLfjH2/He/qa2 v4tQ== X-Gm-Message-State: AFuF++mmr8BF7RqB4y3xcVdx3A/dm8SpgVeTpHiTH+CkXOtHP/XfvM2U eyN8ZT0VGOiSS663fpcj+IdDWdHM3yXkWF7r9Had4P0Gh35HpcigyATonRH1qUlKV/Kot64dJeb HlIcUUPBeLPPQ8yz8zjpd2FUOFw9S5xDTL26ctB7Tl7UtREf/rOIBGNNukBpDjztU/UlddKfABk kc0W0g9ATZT1T19ZUulDvMnTsacHG64fRPV1mkzKGNqRvVBDEMC3uzDpk= X-Received: from plgo14.prod.google.com ([2002:a17:902:d4ce:b0:2dd:47fd:3a9c]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a17:902:f60c:b0:2d3:6cb4:e79e with SMTP id d9443c01a7336-2dd9c8cc333mr28001715ad.5.1789619634309; Wed, 16 Sep 2026 21:33:54 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:25 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-2-suleiman@google.com> Subject: [RFC PATCH 01/12] sched: Abstract task_struct->blocked_on by locking primitive. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Abstract task_struct->blocked_on by type of locking primitive, so that it can be used by things other than mutexes. Signed-off-by: John Stultz Signed-off-by: Suleiman Souhlal --- include/linux/sched.h | 40 +++++++++++++++++++-------- kernel/fork.c | 2 +- kernel/locking/mutex.c | 8 +++--- kernel/sched/core.c | 62 ++++++++++++++++++++++++++++++++++++------ kernel/sched/sched.h | 2 +- 5 files changed, 89 insertions(+), 25 deletions(-) diff --git a/include/linux/sched.h b/include/linux/sched.h index 705970d07614..6edd0c7891c5 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -832,6 +832,16 @@ struct task_ipi_mask { struct task_ipi_mask { }; #endif =20 +enum blocked_on_type { + BO_T_NONE, + BO_T_MUTEX, +}; + +struct blocked_on_lock { + void *lock; + enum blocked_on_type type; +}; + struct task_struct { #ifdef CONFIG_THREAD_INFO_IN_TASK /* @@ -1259,7 +1269,7 @@ struct task_struct { struct rt_mutex_waiter *pi_blocked_on; #endif =20 - struct mutex *blocked_on; /* lock we're blocked on */ + struct blocked_on_lock blocked_on; /* lock we're blocked on */ raw_spinlock_t blocked_lock; =20 /* @@ -2221,10 +2231,15 @@ extern int __cond_resched_rwlock_write(rwlock_t *lo= ck) __must_hold(lock); static inline struct mutex *__get_task_blocked_on(struct task_struct *p) { lockdep_assert_held_once(&p->blocked_lock); - return p->blocked_on; + return p->blocked_on.lock; } =20 -static inline void __set_task_blocked_on(struct task_struct *p, struct mut= ex *m) +/* + * These helpers set and clear the task blocked_on pointer, as well + * as setting the initial blocked_on_state, or clearing it + */ +static inline void __set_task_blocked_on(struct task_struct *p, void *m, + enum blocked_on_type type) { WARN_ON_ONCE(!m); /* The task should only be setting itself as blocked */ @@ -2236,11 +2251,12 @@ static inline void __set_task_blocked_on(struct tas= k_struct *p, struct mutex *m) * with a different mutex. Note, setting it to the same * lock repeatedly is ok. */ - WARN_ON_ONCE(p->blocked_on && p->blocked_on !=3D m); - p->blocked_on =3D m; + WARN_ON_ONCE(p->blocked_on.lock && p->blocked_on.lock !=3D m); + p->blocked_on.lock =3D m; + p->blocked_on.type =3D type; } =20 -static inline void __clear_task_blocked_on(struct task_struct *p, struct m= utex *m) +static inline void __clear_task_blocked_on(struct task_struct *p, void *m) { /* Currently we serialize blocked_on under the task::blocked_lock */ lockdep_assert_held_once(&p->blocked_lock); @@ -2249,21 +2265,23 @@ static inline void __clear_task_blocked_on(struct t= ask_struct *p, struct mutex * * blocked_on relationships, but make sure we are not * clearing the relationship with a different lock. */ - WARN_ON_ONCE(m && p->blocked_on && p->blocked_on !=3D m); - p->blocked_on =3D NULL; + WARN_ON_ONCE(m && p->blocked_on.lock && p->blocked_on.lock !=3D m); + p->blocked_on.lock =3D NULL; + p->blocked_on.type =3D BO_T_NONE; } =20 -static inline void clear_task_blocked_on(struct task_struct *p, struct mut= ex *m) +static inline void clear_task_blocked_on(struct task_struct *p, void *m) { guard(raw_spinlock_irqsave)(&p->blocked_lock); __clear_task_blocked_on(p, m); } + #else -static inline void __clear_task_blocked_on(struct task_struct *p, struct r= t_mutex *m) +static inline void __clear_task_blocked_on(struct task_struct *p, void *m) { } =20 -static inline void clear_task_blocked_on(struct task_struct *p, struct rt_= mutex *m) +static inline void clear_task_blocked_on(struct task_struct *p, void *m) { } #endif /* !CONFIG_PREEMPT_RT */ diff --git a/kernel/fork.c b/kernel/fork.c index a5934a317634..6b3f369aad2b 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -2266,7 +2266,7 @@ __latent_entropy struct task_struct *copy_process( =20 lockdep_init_task(p); =20 - p->blocked_on =3D NULL; /* not blocked yet */ + p->blocked_on.lock =3D NULL; /* not blocked yet */ p->blocked_donor =3D NULL; /* nobody is boosting p yet */ =20 #ifdef CONFIG_BCACHE diff --git a/kernel/locking/mutex.c b/kernel/locking/mutex.c index 942a939cee95..b7565ad15494 100644 --- a/kernel/locking/mutex.c +++ b/kernel/locking/mutex.c @@ -689,7 +689,7 @@ __mutex_lock_common(struct mutex *lock, unsigned int st= ate, unsigned int subclas } =20 raw_spin_lock(¤t->blocked_lock); - __set_task_blocked_on(current, lock); + __set_task_blocked_on(current, lock, BO_T_MUTEX); set_current_state(state); trace_contention_begin(lock, LCB_F_MUTEX); for (;;) { @@ -734,7 +734,7 @@ __mutex_lock_common(struct mutex *lock, unsigned int st= ate, unsigned int subclas * that has cleared our blocked_on state, re-set * it to the lock we are trying to acquire. */ - __set_task_blocked_on(current, lock); + __set_task_blocked_on(current, lock, BO_T_MUTEX); set_current_state(state); /* * Here we order against unlock; we must either see it change @@ -762,7 +762,7 @@ __mutex_lock_common(struct mutex *lock, unsigned int st= ate, unsigned int subclas =20 raw_spin_lock_irqsave(&lock->wait_lock, flags); raw_spin_lock(¤t->blocked_lock); - __set_task_blocked_on(current, lock); + __set_task_blocked_on(current, lock, BO_T_MUTEX); set_current_state(state); =20 if (opt_acquired) @@ -1038,7 +1038,7 @@ static noinline void __sched __mutex_unlock_slowpath(= struct mutex *lock, unsigne */ donor =3D current->blocked_donor; if (donor) { - struct mutex *next_lock; + void *next_lock; =20 raw_spin_lock_nested(&donor->blocked_lock, SINGLE_DEPTH_NESTING); next_lock =3D __get_task_blocked_on(donor); diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 7885ff76e69f..2e8fe4b9bb88 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -150,6 +150,24 @@ static int __init setup_proxy_exec(char *str) } return 1; } + +static inline struct task_struct *__blocked_on_owner(struct blocked_on_loc= k *bo) +{ + switch (bo->type) { + case BO_T_NONE: + return NULL; + case BO_T_MUTEX: + return __mutex_owner(bo->lock); + default: + WARN_ON_ONCE(1); + return NULL; + } +} + +static inline struct task_struct *task_blocked_on_owner(struct task_struct= *p) +{ + return __blocked_on_owner(&p->blocked_on); +} #else static int __init setup_proxy_exec(char *str) { @@ -6799,7 +6817,7 @@ static void proxy_deactivate(struct rq *rq, struct ta= sk_struct *donor) unsigned long state =3D READ_ONCE(donor->__state); =20 WARN_ON_ONCE(state =3D=3D TASK_RUNNING); - WARN_ON_ONCE(donor->blocked_on); + WARN_ON_ONCE(donor->blocked_on.lock); /* * Because we got donor from pick_next_task(), it is *crucial* * that we call proxy_resched_idle() before we deactivate it. @@ -6884,6 +6902,28 @@ static void proxy_migrate_task(struct rq *rq, struct= rq_flags *rf, proxy_reacquire_rq_lock(rq, rf); } =20 +static void +lock_blocked_on_lock(struct blocked_on_lock *bo) +{ + if (bo->type =3D=3D BO_T_MUTEX) + raw_spin_lock(&((struct mutex *)bo->lock)->wait_lock); + else + WARN_ON_ONCE(1); +} + +static void +unlock_blocked_on_lock(struct blocked_on_lock *bo) +{ + if (bo->type =3D=3D BO_T_MUTEX) + raw_spin_unlock(&((struct mutex *)bo->lock)->wait_lock); + else + WARN_ON_ONCE(1); +} + +DEFINE_LOCK_GUARD_1(blocked_on_lock, struct blocked_on_lock, + lock_blocked_on_lock(_T->lock), + unlock_blocked_on_lock(_T->lock)) + /* * Find runnable lock owner to proxy for mutex blocked donor * @@ -6914,6 +6954,7 @@ static struct task_struct * find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags = *rf) __must_hold(__rq_lockp(rq)) { + struct blocked_on_lock bo, *blocked_on; struct task_struct *owner =3D NULL; bool curr_in_chain =3D false; int this_cpu =3D cpu_of(rq); @@ -6922,10 +6963,15 @@ find_proxy_task(struct rq *rq, struct task_struct *= donor, struct rq_flags *rf) =20 /* Follow blocked_on chain. */ for (p =3D donor; p->is_blocked; p =3D owner) { - /* if its PROXY_WAKING, do return migration or run if current */ - struct mutex *mutex =3D p->blocked_on; - if (!mutex) { - clear_task_blocked_on(p, mutex); + /* copy the entire blocked_on structure */ + raw_spin_lock(&p->blocked_lock); + bo =3D p->blocked_on; + raw_spin_unlock(&p->blocked_lock); + blocked_on =3D &bo; + + /* Something changed in the chain, so pick again */ + if (!blocked_on->lock) { + clear_task_blocked_on(p, NULL); if (task_current(rq, p)) { p->is_blocked =3D 0; return p; @@ -6937,11 +6983,11 @@ find_proxy_task(struct rq *rq, struct task_struct *= donor, struct rq_flags *rf) * By taking mutex->wait_lock we hold off concurrent mutex_unlock() * and ensure @owner sticks around. */ - guard(raw_spinlock)(&mutex->wait_lock); + guard(blocked_on_lock)(blocked_on); guard(raw_spinlock)(&p->blocked_lock); =20 /* Check again that p is blocked with blocked_lock held */ - if (mutex !=3D __get_task_blocked_on(p)) { + if (blocked_on->lock !=3D __get_task_blocked_on(p)) { /* * Something changed in the blocked_on chain and * we don't know if only at this level. So, let's @@ -6954,7 +7000,7 @@ find_proxy_task(struct rq *rq, struct task_struct *do= nor, struct rq_flags *rf) if (task_current(rq, p)) curr_in_chain =3D true; =20 - owner =3D __mutex_owner(mutex); + owner =3D __blocked_on_owner(blocked_on); if (!owner) { /* * If there is no owner, either clear blocked_on diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index e656c7059bf8..a386ac33e295 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -2505,7 +2505,7 @@ static inline bool task_is_blocked(struct task_struct= *p) if (!sched_proxy_exec()) return false; =20 - return !!p->blocked_on; + return !!p->blocked_on.lock; } =20 static inline int task_on_cpu(struct rq *rq, struct task_struct *p) --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pf1-f198.google.com (mail-pf1-f198.google.com [209.85.210.198]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 1DAF4308F03 for ; Thu, 17 Sep 2026 04:34:10 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.210.198 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619653; cv=none; b=HyPYOynmDECxF+uGcw+oodexwjN0dlHG+lZW7qjKwHyni87aKPLZQbhSsKsvNwKynm4gJKdRItVcm9AiRhc+w5wapyUENucbs4H0uHfcl9Mkji72wyAuSYyRZo7dHbApxXIsZFzitxMahRdmOyxTHcYDOB4WeeOzIu18emLnSyo= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619653; c=relaxed/simple; bh=A+sazoCMTCHkkpJA961/vMdwqnJh5sK8J0YEg37gHAU=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=TSXYZWfSLzUO/yONWkn/gZc+WmbmkQ2jc8UvZzzylFqyNHtfbgUItBjN55IbzPvoJOQJGjdzdbngURDBnmvB+bRh50U4cOVdTXT8bwgbkJdfqpm3Fd7vcSaU4wC+Tp5Uh/Pr6ALRNBkWtgmgy5HYQ0rAURd8G5iGGn6Xdkr/IEU= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=jZhKwRwM; arc=none smtp.client-ip=209.85.210.198 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="jZhKwRwM" Received: by mail-pf1-f198.google.com with SMTP id d2e1a72fcca58-8663802b58fso554390b3a.2 for ; Wed, 16 Sep 2026 21:34:10 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619650; x=1790224450; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=wK7P7ZXM1Y+/SB1GLNPOHQ3UpocI9LUwl87jU+Ccbrc=; b=jZhKwRwMbkC4QAMX6qkrmDOETSpUhMIhceDTdH/B2C190P9YRXrhj9kI6hC2BrUkW0 8eALoU5ht1e24NwUP78dVaTEuWGG8LSGhRJnsFcFMcyNQklXIhlD6mzSskGj76LZ668p zat4LeAf6Bd2Rpbj1f5DrgzqfPOQs4wVAj1bGqRhnnpwcXfofiJ3H8TKDrxZUccJ26P2 fxzcU0V/Btp32fDYU6wU11ouFbKD4U2TeHRJuDL18V/6fhq9j/w1DTt+gHp4rGPQwa/5 lPBAgD+Sq+iD1gHfJxsE+c/W8i207AbvJpTFSyC213V3W4bbt3p/c5YVcdE/e3tkes8l Ovvg== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619650; x=1790224450; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=wK7P7ZXM1Y+/SB1GLNPOHQ3UpocI9LUwl87jU+Ccbrc=; b=JykpoDjae/1mWp/jvul+3jFrVBwYUXK1wXgb5mx5kxA1x4hCwMsgD89ILkkGTbsbAV Kix3NmtOGMuRk56MDRIp51ri//2iHkI0B7t5TptnVGlFKsAeV4YFb6rnPRjXm6OObRzD 2iUEa5Ym3qTPKDta1vRoOIYhgfkLxOkrQRhrmpRJjh3wmKExpVGEFOhYwB2ii0xT76Ir Qqtc8Xz09awMo+FOG4MkxhJ1pQELb+n1E4GWJIp/++8uKM7MlIqweRt0fvWzDQ2hUmnl sFBQCXXctMLKhMCJwXqS9EV9sWgQGQflYRaig8RK7++bwsoypcAtNnQbwt6tGJNdFPxG wjsQ== X-Gm-Message-State: AFuF++kY9psb/5jhVroZQ3Mnnb2yGO9rFCmlPGxODukLiptl/e+1ICz2 n3CKK27tNXqCgYAJfPKH132PFc+dEnf7JUiezh6Aea2RBNwSrGbx6uiSLb8quoVZw9mg38U3mpN UAR11+EB8cA1z4lrtBTdq5J6MF6cSXyP0IgP3Z1DEHnQPknZUzO5CuFQAUQNdLxZVhJExBzEDIR 72BndwxOxSs+gB41a9QHDvJALFk11bRaYoDJ25h38HkavGVUhDkyZ8EZ8= X-Received: from pfks17.prod.google.com ([2002:a05:6a00:1951:b0:86a:c4c7:7f02]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a05:6a00:2e83:b0:86e:ff2c:49dc with SMTP id d2e1a72fcca58-8723988b639mr10689588b3a.24.1789619635736; Wed, 16 Sep 2026 21:33:55 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:26 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-3-suleiman@google.com> Subject: [RFC PATCH 02/12] futex: Switch PI futex to use p->pi_futex_lock instead of p->pi_lock. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Switch PI futexes to use p->pi_futex_lock instead of p->pi_lock. When augmenting PING futexes with proxy execution, we get lock order inversions, due to the lock order being p->pi_lock -> mutex->wait_lock in the scheduler, but wait_lock -> p->pi_lock in futex code. So move the futex code to use a new lock, p->pi_futex_lock, to protect p->pi_state_list and pi_state->owner. Signed-off-by: Suleiman Souhlal --- include/linux/sched.h | 1 + init/init_task.c | 1 + kernel/fork.c | 1 + kernel/futex/core.c | 39 ++++++++++++++++++++------------------ kernel/futex/pi.c | 44 +++++++++++++++++++++---------------------- 5 files changed, 46 insertions(+), 40 deletions(-) diff --git a/include/linux/sched.h b/include/linux/sched.h index 6edd0c7891c5..a7de5c496e3c 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1257,6 +1257,7 @@ struct task_struct { =20 /* Protection of the PI data structures: */ raw_spinlock_t pi_lock; + raw_spinlock_t pi_futex_lock; =20 struct wake_q_node wake_q; =20 diff --git a/init/init_task.c b/init/init_task.c index adb207cd987c..3e9d62f2668a 100644 --- a/init/init_task.c +++ b/init/init_task.c @@ -181,6 +181,7 @@ struct task_struct init_task __aligned(L1_CACHE_BYTES) = =3D { .journal_info =3D NULL, INIT_CPU_TIMERS(init_task) .pi_lock =3D __RAW_SPIN_LOCK_UNLOCKED(init_task.pi_lock), + .pi_futex_lock =3D __RAW_SPIN_LOCK_UNLOCKED(init_task.pi_futex_lock), .blocked_lock =3D __RAW_SPIN_LOCK_UNLOCKED(init_task.blocked_lock), .timer_slack_ns =3D 50000, /* 50 usec default slack */ .thread_pid =3D &init_struct_pid, diff --git a/kernel/fork.c b/kernel/fork.c index 6b3f369aad2b..80fa3c2d6ea4 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1835,6 +1835,7 @@ SYSCALL_DEFINE1(set_tid_address, int __user *, tidptr) static void rt_mutex_init_task(struct task_struct *p) { raw_spin_lock_init(&p->pi_lock); + raw_spin_lock_init(&p->pi_futex_lock); #ifdef CONFIG_RT_MUTEXES p->pi_waiters =3D RB_ROOT_CACHED; p->pi_top_task =3D NULL; diff --git a/kernel/futex/core.c b/kernel/futex/core.c index a061f54b606d..13c7ea3a26b3 100644 --- a/kernel/futex/core.c +++ b/kernel/futex/core.c @@ -1354,8 +1354,8 @@ static void exit_pi_state_list(struct task_struct *cu= rr) might_sleep(); /* * Ensure the hash remains stable (no resize) during the while loop - * below. The hb pointer is acquired under the pi_lock so we can't block - * on the mutex. + * below. The hb pointer is acquired under the pi_futex_lock so we + * can't block on the mutex. */ WARN_ON(curr !=3D current); guard(private_hash)(current->mm); @@ -1364,7 +1364,7 @@ static void exit_pi_state_list(struct task_struct *cu= rr) * pi_state_list anymore, but we have to be careful * versus waiters unqueueing themselves: */ - raw_spin_lock_irq(&curr->pi_lock); + raw_spin_lock_irq(&curr->pi_futex_lock); while (!list_empty(head)) { next =3D head->next; pi_state =3D list_entry(next, struct futex_pi_state, list); @@ -1384,22 +1384,25 @@ static void exit_pi_state_list(struct task_struct *= curr) * progress and retry the loop. */ if (!refcount_inc_not_zero(&pi_state->refcount)) { - raw_spin_unlock_irq(&curr->pi_lock); + raw_spin_unlock_irq(&curr->pi_futex_lock); cpu_relax(); - raw_spin_lock_irq(&curr->pi_lock); + raw_spin_lock_irq(&curr->pi_futex_lock); continue; } - raw_spin_unlock_irq(&curr->pi_lock); + raw_spin_unlock_irq(&curr->pi_futex_lock); =20 spin_lock(&hb->lock); raw_spin_lock_irq(&pi_state->pi_mutex.wait_lock); - raw_spin_lock(&curr->pi_lock); + raw_spin_lock(&curr->pi_futex_lock); /* * We dropped the pi-lock, so re-check whether this * task still owns the PI-state: */ if (head->next !=3D next) { - /* retain curr->pi_lock for the loop invariant */ + /* + * retain curr->pi_futex_lock for the loop + * invariant + */ raw_spin_unlock(&pi_state->pi_mutex.wait_lock); spin_unlock(&hb->lock); put_pi_state(pi_state); @@ -1411,7 +1414,7 @@ static void exit_pi_state_list(struct task_struct *cu= rr) list_del_init(&pi_state->list); pi_state->owner =3D NULL; =20 - raw_spin_unlock(&curr->pi_lock); + raw_spin_unlock(&curr->pi_futex_lock); raw_spin_unlock_irq(&pi_state->pi_mutex.wait_lock); spin_unlock(&hb->lock); } @@ -1419,9 +1422,9 @@ static void exit_pi_state_list(struct task_struct *cu= rr) rt_mutex_futex_unlock(&pi_state->pi_mutex); put_pi_state(pi_state); =20 - raw_spin_lock_irq(&curr->pi_lock); + raw_spin_lock_irq(&curr->pi_futex_lock); } - raw_spin_unlock_irq(&curr->pi_lock); + raw_spin_unlock_irq(&curr->pi_futex_lock); } #else static inline void exit_pi_state_list(struct task_struct *curr) { } @@ -1513,25 +1516,25 @@ static void futex_cleanup_begin(struct task_struct = *tsk) mutex_lock(&tsk->futex.exit_mutex); =20 /* - * Switch the state to FUTEX_STATE_EXITING under tsk->pi_lock. + * Switch the state to FUTEX_STATE_EXITING under tsk->pi_futex_lock. * * This ensures that all subsequent checks of tsk->futex_state in * attach_to_pi_owner() must observe FUTEX_STATE_EXITING with - * tsk->pi_lock held. + * tsk->pi_futex_lock held. * * It guarantees also that a pi_state which was queued right before - * the state change under tsk->pi_lock by a concurrent waiter must + * the state change under tsk->pi_futex_lock by a concurrent waiter must * be observed in exit_pi_state_list(). */ - raw_spin_lock_irq(&tsk->pi_lock); + raw_spin_lock_irq(&tsk->pi_futex_lock); tsk->futex.state =3D FUTEX_STATE_EXITING; - raw_spin_unlock_irq(&tsk->pi_lock); + raw_spin_unlock_irq(&tsk->pi_futex_lock); } =20 static void futex_cleanup_end(struct task_struct *tsk) __releases(&tsk->futex.exit_mutex) { - scoped_guard(raw_spinlock_irq, &tsk->pi_lock) + scoped_guard(raw_spinlock_irq, &tsk->pi_futex_lock) tsk->futex.state =3D FUTEX_STATE_DEAD; =20 /* @@ -1579,7 +1582,7 @@ void futex_exec_done(struct task_struct *tsk) * ordering guarantee required here is that the previous store to * tsk::mm in the calling code cannot be reordered against this store. */ - guard(raw_spinlock_irq)(&tsk->pi_lock); + guard(raw_spinlock_irq)(&tsk->pi_futex_lock); tsk->futex.state =3D FUTEX_STATE_OK; } =20 diff --git a/kernel/futex/pi.c b/kernel/futex/pi.c index 98f1b962e59a..ceeeca1910ca 100644 --- a/kernel/futex/pi.c +++ b/kernel/futex/pi.c @@ -51,18 +51,18 @@ static void pi_state_update_owner(struct futex_pi_state= *pi_state, lockdep_assert_held(&pi_state->pi_mutex.wait_lock); =20 if (old_owner) { - raw_spin_lock(&old_owner->pi_lock); + raw_spin_lock(&old_owner->pi_futex_lock); WARN_ON(list_empty(&pi_state->list)); list_del_init(&pi_state->list); - raw_spin_unlock(&old_owner->pi_lock); + raw_spin_unlock(&old_owner->pi_futex_lock); } =20 if (new_owner) { - raw_spin_lock(&new_owner->pi_lock); + raw_spin_lock(&new_owner->pi_futex_lock); WARN_ON(!list_empty(&pi_state->list)); list_add(&pi_state->list, &new_owner->futex.pi_state_list); pi_state->owner =3D new_owner; - raw_spin_unlock(&new_owner->pi_lock); + raw_spin_unlock(&new_owner->pi_futex_lock); } } =20 @@ -177,7 +177,7 @@ void put_pi_state(struct futex_pi_state *pi_state) * * (and pi_mutex 'obviously') * - * p->pi_lock: + * p->pi_futex_lock: * * p->futex.pi_state_list -> pi_state->list, relation * pi_mutex->owner -> pi_state->owner, relation @@ -191,7 +191,7 @@ void put_pi_state(struct futex_pi_state *pi_state) * * hb->lock * pi_mutex->wait_lock - * p->pi_lock + * p->pi_futex_lock * * Futex kernel state: * @@ -222,12 +222,12 @@ void put_pi_state(struct futex_pi_state *pi_state) * * The state has two related locks: * - * 1) p::pi_lock + * 1) p::pi_futex_lock * - * p::pi_lock has to be taken by the waiter when evaluating the state to - * protect against a concurrent exit/exec cleanup by the owner. If the = state - * is OK then the waiter can be attached to the owner while still holdi= ng - * pi_lock. + * p::pi_futex_lock has to be taken by the waiter when evaluating the s= tate + * to protect against a concurrent exit/exec cleanup by the owner. If t= he + * state is OK then the waiter can be attached to the owner while still + * holding pi_futex_lock. * * The cleanup code has to hold it for all state transitions to ensure = that * the stores to the state cannot be reordered against previous stores = on @@ -482,13 +482,13 @@ static int attach_to_pi_owner(u32 __user *uaddr, u32 = uval, union futex_key *key, * We need to look at the task state to figure out whether the task is * exiting. To protect against the change of the task state from * FUTEX_STATE_OK to FUTEX_STATE_EXISTING in futex_cleanup_begin() it is - * required to do this protected by p->pi_lock, which prevents the owner - * from concurrently starting the exit cleanup. + * required to do this protected by p->pi_futex_lock, which prevents + * the owner from concurrently starting the exit cleanup. * - * If the state is FUTEX_STATE_OK pi_lock must be held until the waiter - * is attached to protect against a concurrent exit()/exec(). + * If the state is FUTEX_STATE_OK pi_futex_lock must be held until the + * waiter is attached to protect against a concurrent exit()/exec(). */ - raw_spin_lock_irq(&p->pi_lock); + raw_spin_lock_irq(&p->pi_futex_lock); =20 /* Validate that the task is ready for futex operations. */ if (unlikely(p->futex.state !=3D FUTEX_STATE_OK)) { @@ -503,14 +503,14 @@ static int attach_to_pi_owner(u32 __user *uaddr, u32 = uval, union futex_key *key, * re-evaluates the situation. */ if (p->futex.state =3D=3D FUTEX_STATE_EXITING) { - raw_spin_unlock_irq(&p->pi_lock); + raw_spin_unlock_irq(&p->pi_futex_lock); *exiting =3D p; return -EBUSY; } =20 int ret =3D handle_exit_race(uaddr, uval); =20 - raw_spin_unlock_irq(&p->pi_lock); + raw_spin_unlock_irq(&p->pi_futex_lock); put_task_struct(p); return ret; } @@ -524,14 +524,14 @@ static int attach_to_pi_owner(u32 __user *uaddr, u32 = uval, union futex_key *key, * key's mm is freed. */ if (unlikely(p->mm !=3D key->private.mm)) { - raw_spin_unlock_irq(&p->pi_lock); + raw_spin_unlock_irq(&p->pi_futex_lock); put_task_struct(p); return -EPERM; } } =20 __attach_to_pi_owner(p, key, ps); - raw_spin_unlock_irq(&p->pi_lock); + raw_spin_unlock_irq(&p->pi_futex_lock); =20 put_task_struct(p); =20 @@ -650,9 +650,9 @@ int futex_lock_pi_atomic(u32 __user *uaddr, struct fute= x_hash_bucket *hb, * because @task is known and valid. */ if (set_waiters) { - raw_spin_lock_irq(&task->pi_lock); + raw_spin_lock_irq(&task->pi_futex_lock); __attach_to_pi_owner(task, key, ps); - raw_spin_unlock_irq(&task->pi_lock); + raw_spin_unlock_irq(&task->pi_futex_lock); } return 1; } --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pf1-f198.google.com (mail-pf1-f198.google.com [209.85.210.198]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 06AFE3BD65D for ; Thu, 17 Sep 2026 04:33:58 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.210.198 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619640; cv=none; b=eEqLPXAVzIsthVzyvsweS3zcH4tWbs7p/KrEnUlY9PB1GP7Z7nviKoOKVufGJkBC0IFxqQPb5W1E26Nt0Vdbhop8ept6yn6qIK/0zOisBIUXsZ79pvxqgz8EtVqZz4WxYWRJQe32HZ6HOZBI6M4lKbIe55/+ELPfcX1UP+aIjrg= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619640; c=relaxed/simple; bh=JhcozAA+A9t065vQxyaqGS89nG6vza/RKha5TtWKih8=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=i5vPwdKcNRUiWqzxDjecOyNbW008Sfkb+J57UhqrgdVLhA85dYbdXS2RS+P6ybJXOOToaSI2BZDFFwa4pcF4aZDkLXRvsDN88pquWxAMPd5dYrXGo3TzV17gcdGyfha5FnGHiCUheYBCQ5dZSVqPqPOA2ct8+84pvn8rErOeaJA= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=o+62nOvL; arc=none smtp.client-ip=209.85.210.198 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="o+62nOvL" Received: by mail-pf1-f198.google.com with SMTP id d2e1a72fcca58-8710450f731so589453b3a.0 for ; Wed, 16 Sep 2026 21:33:58 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619638; x=1790224438; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=k62DQxnbCWh2Ug5s4Hdgmh9l39pPAQ2b4OqnI14O8ic=; b=o+62nOvLNXkkNEHYY+1Z+buzVC5pTum2kqeaaD9m3bKpUywHpzVWyk/LHNP21unR8A S19pZaxWnDN9YmX4YGT8wOI9JNGJF7OHojHTevuosuVpTZWQeKGAAp5Rg+seM7zEunSJ I201Tc0uTxfNWcOEUfgsrGAOWKvMp2xLjRqr7NgZiDcYGiCywBi8cfwLdptgw5+XXI8r 3sn6Bgu3T8Y0GohusHjgTYBL30CPqEQfBuQchOcepkGAgQJVV9jRXaeqIBNsJdtxXk8r iNV2GXNCJkHOychTx04p74+q79W5DPj1Pbmvw30mZdYp9rtQNjcuLQj53EJyiEcLGtFa U/gQ== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619638; x=1790224438; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=k62DQxnbCWh2Ug5s4Hdgmh9l39pPAQ2b4OqnI14O8ic=; b=RJBPYUwcPRE+4UWpSHiHNDZ2abfZYoz9zQSClTPVb8R9NHQSXN/lSeidpbkSKECtvx 3MMPJHhPKirYy97csy3WOzu6cxsiNgPNCHGcoVvc1fPz5qCdapG3ZzKO8tuXhHl7N+Dn fNpvBXv3alqdJjSmCtwA4BUvbWk3U0RKi4C0KeX02GxaXW09x9XmHPjsVEzp+p6/6S1G R2YhAJn056WFCWLEj9UchzoTECHnFDiO66zn2lXHhOCos/l7PRFd1m4asfAPV/8BhYgS 8m1HLF1ReDQX+qssF5Ep2lhAeqECzypWmTbuDuJw8dPYKyvufuks+3iKxkeIo4+N/s1D /F9w== X-Gm-Message-State: AFuF++k93Uq5uUhwKm/sJZOpVssKvW8rf+dUZYX+KQYpJ0kXFKiuMOhI 6w5rskpaIuW0LovslYU7tXTS4CpLndjAeu6SWlsxmE93yIhKw3LdSvjlhuTq1qHi7BqX1EH6ZJr LpWZ6NQ52ucDWrJWuaTYyHS0yTvya/QTTL8t6Nw/cFR1yHc44w2gmuK2/3YSHt8TKEtdiR60hNy SE0y7zsX+a/ElTp4WZGyduoxTL2vOfhQ2H2cYwiALx/1y+vnnzSIK6oRc= X-Received: from pfbgd6.prod.google.com ([2002:a05:6a00:8306:b0:86a:854f:170b]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a05:6a00:2d8c:b0:86c:f33d:e3d6 with SMTP id d2e1a72fcca58-872374c06e9mr10511822b3a.9.1789619637384; Wed, 16 Sep 2026 21:33:57 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:27 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-4-suleiman@google.com> Subject: [RFC PATCH 03/12] futex: Add "ping" parameter to pi_state management functions and export them. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" In preparation to adding FUTEX_PING, allow the pi_state management functions to use a different wait_lock, and make them non-static. Also rename handle_exit_race() to pi_handle_exit_race(). Signed-off-by: Suleiman Souhlal --- kernel/futex/futex.h | 10 ++++++++ kernel/futex/pi.c | 61 +++++++++++++++++++++++++------------------- 2 files changed, 45 insertions(+), 26 deletions(-) diff --git a/kernel/futex/futex.h b/kernel/futex/futex.h index f00f0863ed44..e450f60b180b 100644 --- a/kernel/futex/futex.h +++ b/kernel/futex/futex.h @@ -395,6 +395,16 @@ extern int refill_pi_state_cache(void); extern void get_pi_state(struct futex_pi_state *pi_state); extern void put_pi_state(struct futex_pi_state *pi_state); extern int fixup_pi_owner(u32 __user *uaddr, struct futex_q *q, int locked= ); +extern int pi_handle_exit_race(u32 __user *uaddr, u32 uval); +extern struct futex_pi_state *alloc_pi_state(void); +extern int attach_to_pi_state(u32 __user *uaddr, u32 uval, + struct futex_pi_state *pi_state, + struct futex_pi_state **ps, + bool ping); +extern int attach_to_pi_owner(u32 __user *uaddr, u32 uval, union futex_key= *key, + struct futex_pi_state **ps, + struct task_struct **exiting, + bool ping); =20 /* * Express the locking dependencies for lockdep: diff --git a/kernel/futex/pi.c b/kernel/futex/pi.c index ceeeca1910ca..aefcc0491d60 100644 --- a/kernel/futex/pi.c +++ b/kernel/futex/pi.c @@ -33,7 +33,7 @@ int refill_pi_state_cache(void) return 0; } =20 -static struct futex_pi_state *alloc_pi_state(void) +struct futex_pi_state *alloc_pi_state(void) { struct futex_pi_state *pi_state =3D current->futex.pi_state_cache; =20 @@ -252,9 +252,10 @@ void put_pi_state(struct futex_pi_state *pi_state) * the pi_state against the user space value. If correct, attach to * it. */ -static int attach_to_pi_state(u32 __user *uaddr, u32 uval, - struct futex_pi_state *pi_state, - struct futex_pi_state **ps) +int attach_to_pi_state(u32 __user *uaddr, u32 uval, + struct futex_pi_state *pi_state, + struct futex_pi_state **ps, + bool ping) { pid_t pid =3D uval & FUTEX_TID_MASK; u32 uval2; @@ -284,7 +285,8 @@ static int attach_to_pi_state(u32 __user *uaddr, u32 uv= al, * Now that we have a pi_state, we can acquire wait_lock * and do the state validation. */ - raw_spin_lock_irq(&pi_state->pi_mutex.wait_lock); + if (!ping) + raw_spin_lock_irq(&pi_state->pi_mutex.wait_lock); =20 /* * Since {uval, pi_state} is serialized by wait_lock, and our current @@ -348,8 +350,10 @@ static int attach_to_pi_state(u32 __user *uaddr, u32 u= val, goto out_einval; =20 out_attach: - get_pi_state(pi_state); - raw_spin_unlock_irq(&pi_state->pi_mutex.wait_lock); + if (!ping) { + get_pi_state(pi_state); + raw_spin_unlock_irq(&pi_state->pi_mutex.wait_lock); + } *ps =3D pi_state; return 0; =20 @@ -370,7 +374,7 @@ static int attach_to_pi_state(u32 __user *uaddr, u32 uv= al, return ret; } =20 -static int handle_exit_race(u32 __user *uaddr, u32 uval) +int pi_handle_exit_race(u32 __user *uaddr, u32 uval) { u32 uval2; =20 @@ -419,7 +423,8 @@ static int handle_exit_race(u32 __user *uaddr, u32 uval) } =20 static void __attach_to_pi_owner(struct task_struct *p, union futex_key *k= ey, - struct futex_pi_state **ps) + struct futex_pi_state **ps, + bool ping) { /* * No existing pi state. First waiter. [2] @@ -429,18 +434,20 @@ static void __attach_to_pi_owner(struct task_struct *= p, union futex_key *key, */ struct futex_pi_state *pi_state =3D alloc_pi_state(); =20 - /* - * Initialize the pi_mutex in locked state and make @p - * the owner of it: - */ - __assume_ctx_lock(&pi_state->pi_mutex.wait_lock); - rt_mutex_init_proxy_locked(&pi_state->pi_mutex, p); + if (!ping) { + /* + * Initialize the pi_mutex in locked state and make @p + * the owner of it: + */ + __assume_ctx_lock(&pi_state->pi_mutex.wait_lock); + rt_mutex_init_proxy_locked(&pi_state->pi_mutex, p); + WARN_ON(!list_empty(&pi_state->list)); + list_add(&pi_state->list, &p->futex.pi_state_list); + } =20 /* Store the key for possible exit cleanups: */ pi_state->key =3D *key; =20 - WARN_ON(!list_empty(&pi_state->list)); - list_add(&pi_state->list, &p->futex.pi_state_list); /* * Assignment without holding pi_state->pi_mutex.wait_lock is safe * because there is no concurrency as the object is not published yet. @@ -453,9 +460,10 @@ static void __attach_to_pi_owner(struct task_struct *p= , union futex_key *key, * Lookup the task for the TID provided from user space and attach to * it after doing proper sanity checks. */ -static int attach_to_pi_owner(u32 __user *uaddr, u32 uval, union futex_key= *key, - struct futex_pi_state **ps, - struct task_struct **exiting) +int attach_to_pi_owner(u32 __user *uaddr, u32 uval, union futex_key *key, + struct futex_pi_state **ps, + struct task_struct **exiting, + bool ping) { pid_t pid =3D uval & FUTEX_TID_MASK; struct task_struct *p; @@ -471,7 +479,7 @@ static int attach_to_pi_owner(u32 __user *uaddr, u32 uv= al, union futex_key *key, return -EAGAIN; p =3D find_get_task_by_vpid(pid); if (!p) - return handle_exit_race(uaddr, uval); + return pi_handle_exit_race(uaddr, uval); =20 if (unlikely(p->flags & PF_KTHREAD)) { put_task_struct(p); @@ -508,7 +516,7 @@ static int attach_to_pi_owner(u32 __user *uaddr, u32 uv= al, union futex_key *key, return -EBUSY; } =20 - int ret =3D handle_exit_race(uaddr, uval); + int ret =3D pi_handle_exit_race(uaddr, uval); =20 raw_spin_unlock_irq(&p->pi_futex_lock); put_task_struct(p); @@ -530,7 +538,7 @@ static int attach_to_pi_owner(u32 __user *uaddr, u32 uv= al, union futex_key *key, } } =20 - __attach_to_pi_owner(p, key, ps); + __attach_to_pi_owner(p, key, ps, ping); raw_spin_unlock_irq(&p->pi_futex_lock); =20 put_task_struct(p); @@ -614,7 +622,8 @@ int futex_lock_pi_atomic(u32 __user *uaddr, struct fute= x_hash_bucket *hb, */ top_waiter =3D futex_top_waiter(hb, key); if (top_waiter) - return attach_to_pi_state(uaddr, uval, top_waiter->pi_state, ps); + return attach_to_pi_state(uaddr, uval, top_waiter->pi_state, + ps, false); =20 /* * No waiter and user TID is 0. We are here because the @@ -651,7 +660,7 @@ int futex_lock_pi_atomic(u32 __user *uaddr, struct fute= x_hash_bucket *hb, */ if (set_waiters) { raw_spin_lock_irq(&task->pi_futex_lock); - __attach_to_pi_owner(task, key, ps); + __attach_to_pi_owner(task, key, ps, false); raw_spin_unlock_irq(&task->pi_futex_lock); } return 1; @@ -671,7 +680,7 @@ int futex_lock_pi_atomic(u32 __user *uaddr, struct fute= x_hash_bucket *hb, * attach to the owner. If that fails, no harm done, we only * set the FUTEX_WAITERS bit in the user space variable. */ - return attach_to_pi_owner(uaddr, newval, key, ps, exiting); + return attach_to_pi_owner(uaddr, newval, key, ps, exiting, false); } =20 /* --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pl1-f200.google.com (mail-pl1-f200.google.com [209.85.214.200]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 65CD74E50A6 for ; Thu, 17 Sep 2026 04:34:00 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.214.200 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619642; cv=none; b=pLGBl4Jdcv4OEsVrBsl0W3PB0O5vXvQueo9VTdG8m07mu/8D8NNNkafy2YnkrTcX3ZRPhHinpd51xtwmd9Ukt/CcZ+kRhxGfhqB2be0gIEw9R+TUde4r+oDcqIzbM4f2E/jJWQproRLE63XuvC5RwuOt1Ae8AkHGYtun+S2s4FI= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619642; c=relaxed/simple; bh=Al6Q1DUpXZERjzS25RSPq/CvMSV4ceqw+EJvZwL0TTE=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=FlcMQE/tMhId9Zm84AJ28F8GMfLxnsExj/ObB7euBGAFwQeBH0wXJvqz8KRbLi4wXAqtMeUWC7TDS2srt6spF6OCNo84s3qo5pdVjAEvxYeasD/0LztlqOyAVjUaI4EST5AxlLBFclXWFZWEPmGDuezKSk0+JNyA3a2DtctZ028= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=dm6pKZ/Q; arc=none smtp.client-ip=209.85.214.200 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="dm6pKZ/Q" Received: by mail-pl1-f200.google.com with SMTP id d9443c01a7336-2d9057fab9eso7381035ad.1 for ; Wed, 16 Sep 2026 21:34:00 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619639; x=1790224439; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=BNXNZ2ZvOvqqCDqnBH76pSiKJ4ociBPqoDsGS+F/MjI=; b=dm6pKZ/QRUC9/YBlBZU0toSaxcNdQac78EoamqdyZbQji5AoIuafsTyMshKgBYRGNc KSIG6GmTFUmwdtz28V5LXKaETaUAR2LVMeGOluyWkTY2TX6zSv0DPANku7zOBnGpA1JZ nbGSBVwSUfd3NaZydA6676t7f7TZkLDpmcUb01YKzBzficwghEoc0QXIssHyMs6Fj4IF dShVEelyd+iAP0zhdBEHjth9IT74lrfW2ktoEi28uIw1hTI2cTVsi5UyBIH9VtfspU05 h5MwX1Dv2njvYR4WG5Cz0e4MUVCgnZ4x2uojz7TtWpTWOQ3qX6VgSWRCALXEYuAxaIgc ObaQ== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619639; x=1790224439; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=BNXNZ2ZvOvqqCDqnBH76pSiKJ4ociBPqoDsGS+F/MjI=; b=lgWUmSGk+Tj4fiOkiP+ABMY70yCzu/TmOZ+so3BUQu4VUQXnocDUkrDQoVsio+hxPC OD2cHVQNlgLwuYkixa2Xy/GokyrdL+bsfQhGw7zChdYCyJ0tCkLPS5VNf84yyZ24DR9z HLoa/E9GR04KKF7vRxa7sjfmEAcEpSqs7RqJavBx1d1tZpQ5amPc7gvrOWQKYQae0Bg1 ygbNypDzDjdET0xbWbiqqVASo0ZYvkG2g6XawaprNRF70Y25kXb+wb/X2yc26Nh+nZ8Q rWsj0lRE9q/utLQ0oHzCTF4qb+zIrwEqFNxyjm3Op7lcPu2795/QrWySB2mTfUNbOgp4 iTag== X-Gm-Message-State: AFuF++nKOHEIUD3npHrogZx8KPMBwhhmGuSL6Z1Ui5Ga7IhnyuVUg44e GttjUjNkwa7vkxyIYQX4vXslV1J7j4imBMBa9XgT0VNFwGZr4E151A2zaCxiOqyUyGcweB8RYDx XGeuaP9dTZGJkIAzf0vULkYz44monJtPv2wbM74sgXthKB5yACa598F0XZfeb4AzMiY07oXZmbx J0xLvMfsHjBgO/322YF0iuEBXzgDOqsK9UeSxMFif3nTmNq0oqQKRdqoY= X-Received: from plrp19.prod.google.com ([2002:a17:902:b093:b0:2dd:5048:500c]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a17:903:1904:b0:2d6:ffa1:429b with SMTP id d9443c01a7336-2dd8df0c3ebmr104225105ad.7.1789619638913; Wed, 16 Sep 2026 21:33:58 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:28 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-5-suleiman@google.com> Subject: [RFC PATCH 04/12] futex: Introduce stealable PI futex, FUTEX_*_PING. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Introduce a New Generation of PI futexes, FUTEX_*_PING. They are used similarly to FUTEX_*_PI, where the owner is expected to write its TID in the futex. The main user-visible difference with regular PI futexes is that the kernel is allowed to set the FUTEX_WAITERS bit with an empty TID. This is to allow for a new locker to steal the lock from the top waiter, which is supposed to improve performance in workloads that don't need strict RT handling. An example of how they're meant to be used: Lock: static __thread pid_t tid =3D gettid(); uint32_t oldval =3D 0; if (atomic_compare_exchange_strong(ftx, &oldval, tid) return; if (futex(ftx, FUTEX_LOCK_PING, 0, NULL) !=3D 0) err(1, "FUTEX_LOCK_PING"); Unlock: static __thread pid_t tid =3D gettid(); uint32_t oldval =3D tid; if (atomic_compare_exchange_strong(ftx, &oldval, 0)) return; if (futex(ftx, FUTEX_UNLOCK_PING, 0, NULL) !=3D 0) err(1, "FUTEX_UNLOCK_PING"); Signed-off-by: Suleiman Souhlal --- include/linux/futex.h | 11 + include/linux/futex_types.h | 1 + include/uapi/linux/futex.h | 3 + kernel/futex/Makefile | 2 +- kernel/futex/futex.h | 13 +- kernel/futex/pi.c | 15 ++ kernel/futex/ping.c | 503 ++++++++++++++++++++++++++++++++++++ kernel/futex/syscalls.c | 6 + 8 files changed, 552 insertions(+), 2 deletions(-) create mode 100644 kernel/futex/ping.c diff --git a/include/linux/futex.h b/include/linux/futex.h index 18ed18d5cbc1..b1b422b480fb 100644 --- a/include/linux/futex.h +++ b/include/linux/futex.h @@ -66,6 +66,7 @@ static inline void futex_init_task(struct task_struct *ts= k) { memset(&tsk->futex, 0, sizeof(tsk->futex)); INIT_LIST_HEAD(&tsk->futex.pi_state_list); + INIT_LIST_HEAD(&tsk->futex.ping_state_list); tsk->futex.state =3D FUTEX_STATE_OK; mutex_init(&tsk->futex.exit_mutex); } @@ -159,4 +160,14 @@ void futex_mm_init(struct mm_struct *mm); static inline void futex_mm_init(struct mm_struct *mm) { } #endif =20 +struct ping_mutex { + raw_spinlock_t wait_lock; + struct task_struct *owner; +}; + +static inline struct task_struct *ping_mutex_owner(struct ping_mutex *ping= _mutex) +{ + return READ_ONCE(ping_mutex->owner); +} + #endif /* _LINUX_FUTEX_H */ diff --git a/include/linux/futex_types.h b/include/linux/futex_types.h index d320c0571f0c..50cb30d6e54b 100644 --- a/include/linux/futex_types.h +++ b/include/linux/futex_types.h @@ -26,6 +26,7 @@ struct futex_sched_data { struct compat_robust_list_head __user *compat_robust_list; #endif struct list_head pi_state_list; + struct list_head ping_state_list; struct futex_pi_state *pi_state_cache; struct mutex exit_mutex; unsigned int state; diff --git a/include/uapi/linux/futex.h b/include/uapi/linux/futex.h index 10a36c551675..e2c15db7274e 100644 --- a/include/uapi/linux/futex.h +++ b/include/uapi/linux/futex.h @@ -22,6 +22,9 @@ #define FUTEX_WAIT_REQUEUE_PI 11 #define FUTEX_CMP_REQUEUE_PI 12 #define FUTEX_LOCK_PI2 13 +#define FUTEX_LOCK_PING 14 +#define FUTEX_UNLOCK_PING 15 +#define FUTEX_TRYLOCK_PING 16 =20 #define FUTEX_PRIVATE_FLAG 128 #define FUTEX_CLOCK_REALTIME 256 diff --git a/kernel/futex/Makefile b/kernel/futex/Makefile index dce70f8a322b..c4cfaf122894 100644 --- a/kernel/futex/Makefile +++ b/kernel/futex/Makefile @@ -2,4 +2,4 @@ =20 CONTEXT_ANALYSIS :=3D y =20 -obj-y +=3D core.o syscalls.o pi.o requeue.o waitwake.o +obj-y +=3D core.o syscalls.o pi.o ping.o requeue.o waitwake.o diff --git a/kernel/futex/futex.h b/kernel/futex/futex.h index e450f60b180b..c0560d30aaaa 100644 --- a/kernel/futex/futex.h +++ b/kernel/futex/futex.h @@ -167,7 +167,10 @@ struct futex_pi_state { /* * The PI object: */ - struct rt_mutex_base pi_mutex; + union { + struct rt_mutex_base pi_mutex; + struct ping_mutex ping_mutex; + }; =20 struct task_struct *owner; refcount_t refcount; @@ -214,6 +217,7 @@ struct futex_q { void *wake_data; union futex_key key; struct futex_pi_state *pi_state; + struct futex_pi_state *ping_state; struct rt_mutex_waiter *rt_waiter; union futex_key *requeue_pi_key; u32 bitset; @@ -405,6 +409,8 @@ extern int attach_to_pi_owner(u32 __user *uaddr, u32 uv= al, union futex_key *key, struct futex_pi_state **ps, struct task_struct **exiting, bool ping); +extern void get_ping_state(struct futex_pi_state *ping_state); +extern void put_ping_state(struct futex_pi_state *ping_state); =20 /* * Express the locking dependencies for lockdep: @@ -488,4 +494,9 @@ extern int futex_lock_pi(u32 __user *uaddr, unsigned in= t flags, ktime_t *time, i =20 bool futex_robust_list_clear_pending(void __user *pop, unsigned int flags); =20 +extern int futex_unlock_ping(u32 __user *uaddr, unsigned int flags); + +extern int futex_lock_ping(u32 __user *uaddr, unsigned int flags, ktime_t = *time, + int trylock); + #endif /* _FUTEX_H */ diff --git a/kernel/futex/pi.c b/kernel/futex/pi.c index aefcc0491d60..e7e6e347f97d 100644 --- a/kernel/futex/pi.c +++ b/kernel/futex/pi.c @@ -287,6 +287,8 @@ int attach_to_pi_state(u32 __user *uaddr, u32 uval, */ if (!ping) raw_spin_lock_irq(&pi_state->pi_mutex.wait_lock); + else + raw_spin_lock_irq(&pi_state->ping_mutex.wait_lock); =20 /* * Since {uval, pi_state} is serialized by wait_lock, and our current @@ -353,6 +355,9 @@ int attach_to_pi_state(u32 __user *uaddr, u32 uval, if (!ping) { get_pi_state(pi_state); raw_spin_unlock_irq(&pi_state->pi_mutex.wait_lock); + } else { + get_ping_state(pi_state); + raw_spin_unlock_irq(&pi_state->ping_mutex.wait_lock); } *ps =3D pi_state; return 0; @@ -443,6 +448,16 @@ static void __attach_to_pi_owner(struct task_struct *p= , union futex_key *key, rt_mutex_init_proxy_locked(&pi_state->pi_mutex, p); WARN_ON(!list_empty(&pi_state->list)); list_add(&pi_state->list, &p->futex.pi_state_list); + } else { + /* + * Initialize the pi_mutex in locked state and make @p + * the owner of it: + */ + pi_state->ping_mutex.owner =3D p; + raw_spin_lock_init(&pi_state->ping_mutex.wait_lock); + + WARN_ON(!list_empty(&pi_state->list)); + list_add(&pi_state->list, &p->futex.ping_state_list); } =20 /* Store the key for possible exit cleanups: */ diff --git a/kernel/futex/ping.c b/kernel/futex/ping.c new file mode 100644 index 000000000000..3d732489513b --- /dev/null +++ b/kernel/futex/ping.c @@ -0,0 +1,503 @@ +// SPDX-License-Identifier: GPL-2.0-only + +#include +#include +#include + +#include "futex.h" + +static int futex_trylock_ping_state(u32 __user *uaddr, + struct futex_pi_state *ping_state); + +static void ping_state_update_owner(struct futex_pi_state *ping_state, + struct task_struct *new_owner) +{ + struct task_struct *old_owner =3D ping_state->owner; + + lockdep_assert_held(&ping_state->ping_mutex.wait_lock); + + if (old_owner) { + raw_spin_lock(&old_owner->pi_futex_lock); + WARN_ON(list_empty(&ping_state->list)); + list_del_init(&ping_state->list); + raw_spin_unlock(&old_owner->pi_futex_lock); + } + + if (new_owner) { + raw_spin_lock(&new_owner->pi_futex_lock); + WARN_ON(!list_empty(&ping_state->list)); + list_add(&ping_state->list, &new_owner->futex.ping_state_list); + ping_state->owner =3D new_owner; + raw_spin_unlock(&new_owner->pi_futex_lock); + } +} + +static int lock_pi_update_atomic(u32 __user *uaddr, u32 uval, u32 newval) +{ + int err; + u32 curval; + + if (unlikely(should_fail_futex(true))) + return -EFAULT; + + err =3D futex_cmpxchg_value_locked(&curval, uaddr, uval, newval); + if (unlikely(err)) + return err; + + /* If user space value changed, let the caller retry */ + return curval !=3D uval ? -EAGAIN : 0; +} + +void +get_ping_state(struct futex_pi_state *ping_state) +{ + WARN_ON_ONCE(!refcount_inc_not_zero(&ping_state->refcount)); +} + +/* + * Drops a reference to the pi_state object and frees or caches it + * when the last reference is gone. + */ +void +put_ping_state(struct futex_pi_state *ping_state) +{ + if (!ping_state) + return; + + if (!refcount_dec_and_test(&ping_state->refcount)) + return; + + /* + * If ping_state->owner is NULL, the owner is most probably dying + * and has cleaned up the pi_state already + */ + if (ping_state->owner) { + unsigned long flags; + + raw_spin_lock_irqsave(&ping_state->ping_mutex.wait_lock, flags); + ping_state_update_owner(ping_state, NULL); + WRITE_ONCE(ping_state->ping_mutex.owner, NULL); + raw_spin_unlock_irqrestore(&ping_state->ping_mutex.wait_lock, + flags); + } + + if (current->futex.pi_state_cache) { + kfree(ping_state); + } else { + /* + * pi_state->list is already empty. + * clear pi_state->owner. + * refcount is at 0 - put it back to 1. + */ + ping_state->owner =3D NULL; + refcount_set(&ping_state->refcount, 1); + current->futex.pi_state_cache =3D ping_state; + } +} + +static void futex_unqueue_ping(struct futex_q *q) +{ + if (!plist_node_empty(&q->list)) + __futex_unqueue(q); + + WARN_ON(!q->ping_state); + put_ping_state(q->ping_state); + q->ping_state =3D NULL; +} + +/* Returns >0 if lock acquired, <0 on error */ +static int futex_trylock_ping_state(u32 __user *uaddr, + struct futex_pi_state *ping_state) +{ + struct task_struct *owner; + u32 uval, new, newtid; + int ret; + + ret =3D 0; + raw_spin_lock_irq(&ping_state->ping_mutex.wait_lock); + owner =3D ping_mutex_owner(&ping_state->ping_mutex); + if (owner =3D=3D NULL) { + newtid =3D task_pid_vnr(current); + + ret =3D futex_get_value_locked(&uval, uaddr); + if (ret) + goto err; + if (uval & FUTEX_TID_MASK) { + ret =3D -EAGAIN; + goto err; + } + new =3D newtid | FUTEX_WAITERS; + ret =3D lock_pi_update_atomic(uaddr, uval, new); + if (ret) + goto err; + if (ping_state->owner !=3D current) + ping_state_update_owner(ping_state, current); + WRITE_ONCE(ping_state->ping_mutex.owner, current); + ret =3D 1; + } + raw_spin_unlock_irq(&ping_state->ping_mutex.wait_lock); + return ret; + +err: + raw_spin_unlock_irq(&ping_state->ping_mutex.wait_lock); + switch (ret) { + case -EFAULT: + ret =3D fault_in_user_writeable(uaddr); + break; + case -EAGAIN: + ret =3D 0; + break; + case -EINVAL: + break; + default: + WARN_ON(1); + } + return ret; +} + +static int futex_lock_ping_atomic(u32 __user *uaddr, + struct futex_hash_bucket *hb, + union futex_key *key, + struct futex_pi_state **ps, + struct task_struct *task, + struct task_struct **exiting) +{ + u32 uval, newval, vpid =3D task_pid_vnr(task); + struct futex_q *top_waiter; + int ret; + + /* + * Read the user space value first so we can validate a few + * things before proceeding further. + */ + if (futex_get_value_locked(&uval, uaddr)) + return -EFAULT; + + if (unlikely(should_fail_futex(true))) + return -EFAULT; + + /* + * Detect deadlocks. + */ + if ((unlikely((uval & FUTEX_TID_MASK) =3D=3D vpid))) + return -EDEADLK; + + if ((unlikely(should_fail_futex(true)))) + return -EDEADLK; + + /* + * Lookup existing state first. If it exists, try to attach to + * its ping_state. + */ + top_waiter =3D futex_top_waiter(hb, key); + if (top_waiter) { + struct futex_pi_state *_ps; + + _ps =3D top_waiter->ping_state; + if (_ps =3D=3D NULL) + return -EINVAL; + ret =3D futex_trylock_ping_state(uaddr, _ps); + if (ret > 0) { + /* We stole the lock from the top waiter. */ + raw_spin_lock_irq(&_ps->ping_mutex.wait_lock); + WARN_ON_ONCE(!refcount_read(&_ps->refcount)); + get_ping_state(_ps); + raw_spin_unlock_irq(&_ps->ping_mutex.wait_lock); + *ps =3D _ps; + return 1; + } else if (ret < 0) + return ret; + return attach_to_pi_state(uaddr, uval, top_waiter->ping_state, + ps, true); + } + + /* + * No waiter and user TID is 0. We are here because the + * waiters or the owner died bit is set or called from + * requeue_cmp_pi or for whatever reason something took the + * syscall. + */ + if (!(uval & FUTEX_TID_MASK)) { + /* + * We take over the futex. No other waiters and the user space + * TID is 0. We preserve the owner died bit. + */ + newval =3D uval & FUTEX_OWNER_DIED; + newval |=3D vpid; + + ret =3D lock_pi_update_atomic(uaddr, uval, newval); + if (ret) + return ret; + return 1; + } + + /* + * First waiter. Set the waiters bit before attaching ourself to + * the owner. If owner tries to unlock, it will be forced into + * the kernel and blocked on hb->lock. + */ + newval =3D uval | FUTEX_WAITERS; + ret =3D lock_pi_update_atomic(uaddr, uval, newval); + if (ret) + return ret; + /* + * If the update of the user space value succeeded, we try to + * attach to the owner. If that fails, no harm done, we only + * set the FUTEX_WAITERS bit in the user space variable. + */ + return attach_to_pi_owner(uaddr, newval, key, ps, exiting, true); +} + +/* + * Return values: + * < 0: error. + * 0: did not get the lock. + * 1: got the lock. + */ +int futex_lock_ping(u32 __user *uaddr, unsigned int flags, ktime_t *time, + int trylock) +{ + struct hrtimer_sleeper timeout, *to; + struct task_struct *exiting; + struct futex_q q =3D futex_q_init; + bool queued; + int ret; + + if (refill_pi_state_cache()) + return -ENOMEM; + + to =3D futex_setup_timer(time, &timeout, flags, 0); + +retry: + ret =3D get_futex_key(uaddr, flags, &q.key, FUTEX_WRITE); + if (unlikely(ret !=3D 0)) + goto out; + +retry_private: + if (1) { + CLASS(hbr, hbr)(&q.key); + auto hb =3D hbr.hb; + + futex_q_lock(&q, hb); + + ret =3D futex_lock_ping_atomic(uaddr, hb, &q.key, &q.ping_state, + current, &exiting); + if (unlikely(ret)) { + /* + * Atomic work succeeded and we got the lock, + * or failed. Either way, we do _not_ block. + */ + switch (ret) { + case 1: + ret =3D 0; + /* Got the lock */ + put_ping_state(q.ping_state); + q.ping_state =3D NULL; + goto out_unlock; + case -EFAULT: + goto uaddr_faulted; + case -EBUSY: + case -EAGAIN: + /* + * Two reasons for this: + * - EBUSY: Task is exiting and we just wait + * for the exit to complete. + * - EAGAIN: The user space value changed. + */ + futex_q_unlock(hb); + /* + * Handle the case where the owner is in the + * middle of exiting. Wait for the exit to + * complete otherwise this task might loop + * forever, aka. live lock. + */ + wait_for_owner_exiting(ret, exiting); + cond_resched(); + goto retry; + default: + goto out_unlock; + } + } + + WARN_ON(!q.ping_state); + if (trylock) { + /* + * futex_lock_ping_atomic() trylocks and we did not + * get the lock. + */ + put_ping_state(q.ping_state); + q.ping_state =3D NULL; + ret =3D -EWOULDBLOCK; + goto out_unlock; + } + + queued =3D false; + while (1) { + set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE); + if (!queued) { + futex_queue(&q, hb, current); + queued =3D true; + } else { + WARN_ON_ONCE(plist_node_empty(&q.list)); + spin_unlock(&hb->lock); + __release(q->lock_ptr); + } + + futex_do_wait(&q, to); + + futex_q_lockptr_lock(&q); + if (to && !to->task) { + ret =3D -ETIMEDOUT; + goto out_unqueue; + } + if (signal_pending(current)) { + ret =3D -EINTR; + goto out_unqueue; + } + + ret =3D futex_trylock_ping_state(uaddr, q.ping_state); + if (ret > 0) { + /* Got the futex */ + ret =3D 0; + goto out_unqueue; + } else if (ret < 0) + goto out_unqueue; + } + +out_unqueue: + if (ret !=3D 0 && q.ping_state->owner =3D=3D current) { + /* + * We are pi_state owner but don't own the futex. + * This can happen if we get picked by the previous + * owner but get out without acquiring the lock for + * some reason. + * A later commit addresses this. + */ + WARN_ON_ONCE(1); + } + /* This also puts the ping_state */ + futex_unqueue_ping(&q); +out_unlock: + futex_q_unlock(hb); + __release(q.lock_ptr); + goto out; + +uaddr_faulted: + futex_q_unlock(hb); + __release(q.lock_ptr); + + ret =3D fault_in_user_writeable(uaddr); + if (ret) + goto out; + if (!(flags & FLAGS_SHARED)) + goto retry_private; + goto retry; + } + +out: + if (to) { + hrtimer_cancel(&to->timer); + destroy_hrtimer_on_stack(&to->timer); + } + + return ret; +} + +int futex_unlock_ping(u32 __user *uaddr, unsigned int flags) +{ + struct futex_pi_state *ping_state; + u32 new, uval, vpid =3D task_pid_vnr(current); + union futex_key key =3D FUTEX_KEY_INIT; + struct futex_q *top_waiter; + DEFINE_WAKE_Q(wake_q); + int ret; + +retry: + if (get_user(uval, uaddr)) + return -EFAULT; + /* + * We release only a lock we actually own: + */ + if ((uval & FUTEX_TID_MASK) !=3D vpid) + return -EPERM; + + ret =3D get_futex_key(uaddr, flags, &key, FUTEX_WRITE); + if (ret) + return ret; + + CLASS(hbr, hbr)(&key); + auto hb =3D hbr.hb; + spin_lock(&hb->lock); + top_waiter =3D futex_top_waiter(hb, &key); + + if (!top_waiter) { + spin_unlock(&hb->lock); + /* No waiters in the kernel, we can just clear FUTEX_WAITERS */ + ret =3D lock_pi_update_atomic(uaddr, uval, 0); + if (ret) { + switch (ret) { + case -EFAULT: + goto uaddr_faulted; + case -EAGAIN: + cond_resched(); + goto retry; + default: + WARN_ON_ONCE(1); + } + } + return ret; + } + + ping_state =3D top_waiter->ping_state; + ret =3D -EINVAL; + if (!ping_state) + goto out_unlock; + raw_spin_lock_irq(&ping_state->ping_mutex.wait_lock); + if (ping_state->owner !=3D current) { + raw_spin_unlock_irq(&ping_state->ping_mutex.wait_lock); + goto out_unlock; + } + get_ping_state(ping_state); + /* Leave it queued, it gets unqueued on the lock side */ + get_task_struct(top_waiter->task); + wake_q_add_safe(&wake_q, top_waiter->task); + spin_unlock(&hb->lock); + + /* + * Unconditionally set FUTEX_WAITERS. + * It will get removed by the next unlocker who notices there is + * no top_waiter. + */ + new =3D FUTEX_WAITERS; + ret =3D lock_pi_update_atomic(uaddr, uval, new); + if (ret) { + raw_spin_unlock_irq(&ping_state->ping_mutex.wait_lock); + put_ping_state(ping_state); + switch (ret) { + case -EFAULT: + goto uaddr_faulted; + case -EAGAIN: + cond_resched(); + goto retry; + default: + WARN_ON_ONCE(1); + return ret; + } + } + + ping_state_update_owner(ping_state, top_waiter->task); + WRITE_ONCE(ping_state->ping_mutex.owner, NULL); + raw_spin_unlock_irq_wake(&ping_state->ping_mutex.wait_lock, &wake_q); + put_ping_state(ping_state); + return 0; + +out_unlock: + spin_unlock(&hb->lock); + return ret; + +uaddr_faulted: + ret =3D fault_in_user_writeable(uaddr); + if (ret) + return ret; + goto retry; +} diff --git a/kernel/futex/syscalls.c b/kernel/futex/syscalls.c index 2fa19d9d008d..2e6847f0be79 100644 --- a/kernel/futex/syscalls.c +++ b/kernel/futex/syscalls.c @@ -151,6 +151,12 @@ long do_futex(u32 __user *uaddr, int op, u32 val, ktim= e_t *timeout, return futex_unlock_pi(uaddr, flags, uaddr2); case FUTEX_TRYLOCK_PI: return futex_lock_pi(uaddr, flags, NULL, 1); + case FUTEX_LOCK_PING: + return futex_lock_ping(uaddr, flags, timeout, 0); + case FUTEX_UNLOCK_PING: + return futex_unlock_ping(uaddr, flags); + case FUTEX_TRYLOCK_PING: + return futex_lock_ping(uaddr, flags, NULL, 1); case FUTEX_WAIT_REQUEUE_PI: val3 =3D FUTEX_BITSET_MATCH_ANY; return futex_wait_requeue_pi(uaddr, flags, val, timeout, val3, --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pl1-f200.google.com (mail-pl1-f200.google.com [209.85.214.200]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 767062F39C7 for ; Thu, 17 Sep 2026 04:34:01 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.214.200 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619643; cv=none; b=LS2v1Z2m8i22NNAhNyr+r3VsOJEBcPSF9a9PschcPGyUxPA0MJfivLDJLIPvcIACEBeY56DukioiabxCeLW/jddHBPIboYUFWvV6rPecIVWOHpDkr+RGK6nkNQFmfROI7ykrwGQmJI9VKjGZddUkxItnSPHrc7nwZtb4fLX8DxI= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619643; c=relaxed/simple; bh=eDxJlO1beYomrz1saE9Ne3C+Rx4WExHSJA7Wxz9vWhQ=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=Uzc1lhAngehpbZEnXY7fZiSQakofNVaPSwEje0gUw+v3kbbpzzvNtn01PrsUaQW9aNGjddcVC8JSNo82ORB6AG5rNDIgIzak57Vo0zZ8+ds48yUbeQ0lcEZE6O7DL2k6i8Q3FIjHxglRL+Ckzwhx9WwDIZu8XEXl7jVprj9haDU= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=MGGzQZNa; arc=none smtp.client-ip=209.85.214.200 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="MGGzQZNa" Received: by mail-pl1-f200.google.com with SMTP id d9443c01a7336-2d959904658so9097365ad.2 for ; Wed, 16 Sep 2026 21:34:01 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619641; x=1790224441; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=gOgTtu/ydiLy3NCjRlGseZb8fPO1g55uwKhgB7FabPk=; b=MGGzQZNaegZJEKQq7+VO14El4OH7e1LQ5puhxAJHNenQu2KgOiULVbG9+uvCIjV3iP TDZYa7QGwuljkav4brPP32G90VA135cMlfayeelELikbbcJQ9qGlEQJmmDQ94c3w6KVe rzxScbn5zUOP1b5Bz5iHHjoyfrvmjpL2Zv5QSea4J4E8aktuHqtHWm4xcDQdMbEhTmsD NAOVflSkfp0PiWsi3nDI6N1Kgsp0r8WuDaWQp4sQ3RpP92prDI808X0cZxAbz4ostUAG OgxjXdpnHfqMT3yv7a8GMrIiLnpuxfk5AmwBxaDc3o7xoRTtCTpsejk5V8Zbel73pJyJ SfJg== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619641; x=1790224441; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=gOgTtu/ydiLy3NCjRlGseZb8fPO1g55uwKhgB7FabPk=; b=bH4KQ6e7LME/FaLXxbkr3V8ud4Ksp4uETHN3NCtxfHFFsMBDW347C7iCe8tK+FVuhZ MkAe9OJQ219tGqtPu1awlp6NGASsQ3fFhQUhqqq3wPyzhWt6vLHj4cuzgvy1zY4Wz+50 SMibBOU5nsGuASs99XQZygWWJML89xQTAumWxRfGupQEkIF/cWsh11ZYGZWvvGkCAczi gD9Pyu3O6/AYRw6HNeoiOLPslBNwsbo/PdOBkniLpLTk53WpHFt01bKUHHyv99YkiCjl 0CAiRQEiSwCA4vufjF0oL2e1x/8DFreVcFfzl99efAXrycF6qicuMDnjv2rrIwft3lbg HDGQ== X-Gm-Message-State: AFuF++m4COvaeoN/XD4cpIllCdDyyHloGS9gwK2Byj56h2605Rhz5VAw b49EhBfWX/Do6wD/wY3PHJXK+d+zacq1R88N0GWrxfQN/X6AkJeSoIUBXj6mgbU3dB87chp+pdK 1IvkojtANCw3VQc94u9I2lbnFS9BrePc7wgXra7C0Grem4ZAMKVtQoOUc6db+ItW0O8WIOVGalX yAq25BaqM77tI6NDjLyrizmiwDSmgHTX0PyuBdyNyEkXrqJS47f+iCe1c= X-Received: from plhn4.prod.google.com ([2002:a17:903:1104:b0:2db:56c2:8176]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a17:902:f60c:b0:2db:3d9a:173b with SMTP id d9443c01a7336-2dd8e17e0a6mr111390485ad.9.1789619640410; Wed, 16 Sep 2026 21:34:00 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:29 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-6-suleiman@google.com> Subject: [RFC PATCH 05/12] futex: Implement exit_ping_state_list(). From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Handle the case when a task exits while holding some ping_states, unowning them and dropping their references. Signed-off-by: Suleiman Souhlal --- kernel/futex/core.c | 55 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 55 insertions(+) diff --git a/kernel/futex/core.c b/kernel/futex/core.c index 13c7ea3a26b3..56c7e2d5faac 100644 --- a/kernel/futex/core.c +++ b/kernel/futex/core.c @@ -1430,6 +1430,58 @@ static void exit_pi_state_list(struct task_struct *c= urr) static inline void exit_pi_state_list(struct task_struct *curr) { } #endif =20 +/* Similar to exit_pi_state_list() */ +static void exit_ping_state_list(struct task_struct *curr) +{ + struct list_head *next, *head =3D &curr->futex.ping_state_list; + struct futex_pi_state *ping_state; + union futex_key key =3D FUTEX_KEY_INIT; + + might_sleep(); + WARN_ON(curr !=3D current); + guard(private_hash)(current->mm); + + raw_spin_lock_irq(&curr->pi_futex_lock); + while (!list_empty(head)) { + next =3D head->next; + ping_state =3D list_entry(next, struct futex_pi_state, list); + if (1) { + CLASS(hbr, hbr)(&key); + auto hb =3D hbr.hb; + + if (!refcount_inc_not_zero(&ping_state->refcount)) { + raw_spin_unlock_irq(&curr->pi_futex_lock); + cpu_relax(); + raw_spin_lock_irq(&curr->pi_futex_lock); + continue; + } + raw_spin_unlock_irq(&curr->pi_futex_lock); + + spin_lock(&hb->lock); + raw_spin_lock_irq(&ping_state->ping_mutex.wait_lock); + raw_spin_lock(&curr->pi_futex_lock); + if (head->next !=3D next) { + raw_spin_unlock(&ping_state->ping_mutex.wait_lock); + spin_unlock(&hb->lock); + put_ping_state(ping_state); + continue; + } + + WARN_ON(ping_state->owner !=3D curr); + WARN_ON(list_empty(&ping_state->list)); + list_del_init(&ping_state->list); + ping_state->owner =3D NULL; + raw_spin_unlock(&curr->pi_futex_lock); + raw_spin_unlock_irq(&ping_state->ping_mutex.wait_lock); + spin_unlock(&hb->lock); + } + put_ping_state(ping_state); + + raw_spin_lock_irq(&curr->pi_futex_lock); + } + raw_spin_unlock_irq(&curr->pi_futex_lock); +} + bool futex_robust_list_clear_pending(void __user *pop, unsigned int flags) { bool size32bit =3D !!(flags & FLAGS_ROBUST_LIST32); @@ -1475,6 +1527,9 @@ static void futex_cleanup(struct task_struct *tsk) =20 if (unlikely(!list_empty(&tsk->futex.pi_state_list))) exit_pi_state_list(tsk); + + if (unlikely(!list_empty(&tsk->futex.ping_state_list))) + exit_ping_state_list(tsk); } =20 /** --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pg1-f198.google.com (mail-pg1-f198.google.com [209.85.215.198]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id B4EBA3A3E97 for ; Thu, 17 Sep 2026 04:34:02 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.215.198 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619644; cv=none; b=fOx1VuUCHkGOmvfAdJTPWoFXR83ftayRbY26Um/cw9lMjjVRZMkUQNJK5onCY8333sFgGnTg2tb+0tjKF7KRkqEU8VbLtnGJBPa0K/YquuwZGH9upEmMIWBQ9U43kYz5y/P7gAv/Xk341Dy8golGIKPWiW74OvgCOV4n0874Mfg= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619644; c=relaxed/simple; bh=4CRAupY4QvRM4EhlfkuMpd5r53+BH1uGNeskJjHP14Y=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=eonxpvyVbAU1yAes0GVJWvM8o2KHdJClYmKBwP+vmvvjVZSOhqoxanlhe7Fajz0x7gc5q5oFdC7irdfvh36xyHyLjv+MgzUJMwuoNhFLSOSOvNQg9i6jvzgGYWUhkc5ludxA0UJDTg6DWgW22QmdnyGZGBQpCXRxfQudtCA0ivM= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=VYy2A4Xj; arc=none smtp.client-ip=209.85.215.198 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="VYy2A4Xj" Received: by mail-pg1-f198.google.com with SMTP id 41be03b00d2f7-cc1ca15334cso548642a12.1 for ; Wed, 16 Sep 2026 21:34:02 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619642; x=1790224442; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=tJznT2p9kSqAbUumLQMmYK9dyBk6t2Lz114HtNv5sRM=; b=VYy2A4XjDp9DuqyARLHUipcFmXxQORgBW6ZRr3kpPkOE3hPCZh25rfSiSJtYcy9tab QfqNU79xvEjM/4B2olKV6z0zdDe7jm002TKDMtcOgsp0G4cI68YHCfU9jUWWqwdRC75L HkIzTUAs5ZuBoK++n/915+USBGd+ZzMWi2RdeXXXMTVFiGROw5OaFTWB8N1n77sMlwl7 2QRNWQ9oaw1b2/3tcXRvSU8/qboq6QiGJQC+nfJ2TZ1mI2QupuQOVFOQC24wUGCDr2Js SVh4niT0H1xix5bdSsRzB4erzw7DHO90ntRNB5QbbqraM+DZozBqgcfyRAe4Ae5X8ZOF ZOtw== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619642; x=1790224442; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=tJznT2p9kSqAbUumLQMmYK9dyBk6t2Lz114HtNv5sRM=; b=ZOKsD/JhQa1bNaNhbLCN+MatUcH20pVTDnt9JG54R5SD0GvLWlnKFEg0XO7sSzFdlB J92hEU2QkIEDKE1OatCw/8e+vVAI8gWVVxLOaHePkWBWq/bh0ZBERT0FRa1gSjPeSm8l IMweW59AKM5P8Z4tw3gQr8lNawZOqMyKve3dUtWFDQDIb7meT0ymaGQ/IV2CNTl8B61n r+wg2pebyL0XS7zIri78UJO2EdS9WgFURWW1Liz3flnlh5CBBAltrtozHvfjjMIWZ4Gc Oessk0EFH6IrOJ3L7LxaykOgoCeFtwyAc4imFjYpY4f2f8yIUSnciXSYzhVF3iGEiztP QAHQ== X-Gm-Message-State: AFuF++lL3i73Z9CVFlxFkSPDu6fVkCx9M3j+yxVX4XhIATzD/QlGzZEp 7eiQVFUFBfMQ0Thf6v84GQ+bPK6KUidg5uK3GQSA6x71RtYt2e9rICL3j9dAbkU8KIP4XmBg+2v eszMhpkoR6VWa5HxD7bKPdW2IVzxr6YpgO/RSH4qI8xLNSybD4smUkbUMAzlkvZOsAO7jnQMqaT ZSD+AYZMC0d9eWkatOarKBKX4V49865kVaVm2eYLnX9SYTFLsugSjG47Q= X-Received: from pgbm11-n1.prod.google.com ([2002:a05:6a02:618b:10b0:cc5:a28:4194]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a05:6a21:1193:b0:3da:ebec:69c1 with SMTP id adf61e73a8af0-3dd5f7acb7amr13562703637.24.1789619641754; Wed, 16 Sep 2026 21:34:01 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:30 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-7-suleiman@google.com> Subject: [RFC PATCH 06/12] futex: Address aborting from futex_lock_ping() while owning ping_state. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" It is possible to unsuccesfully get out of the futex_lock_ping() loop while owning the ping_state. This can happen when a waiting task gets chosen by an unlocker as the top waiter, but gets a signal or its timeout expires. When this happens, wake up the next waiter and give them the ping_state. Signed-off-by: Suleiman Souhlal --- kernel/futex/ping.c | 72 ++++++++++++++++++++++++++++++++++++++------- 1 file changed, 61 insertions(+), 11 deletions(-) diff --git a/kernel/futex/ping.c b/kernel/futex/ping.c index 3d732489513b..ebcd3c4a7793 100644 --- a/kernel/futex/ping.c +++ b/kernel/futex/ping.c @@ -105,6 +105,43 @@ static void futex_unqueue_ping(struct futex_q *q) q->ping_state =3D NULL; } =20 +/* + * We own the ping_state but weren't able to get the futex. + * Wake up the next waiter and give them ownership. + */ +static void give_ping_state_to_next_waiter(struct futex_hash_bucket *hb, + union futex_key *key, + struct futex_pi_state *ping_state) +{ + struct futex_q *top_waiter; + DEFINE_WAKE_Q(wake_q); + + raw_spin_lock_irq(&ping_state->ping_mutex.wait_lock); + /* + * Someone else got the futex and we don't need to do anything + * anymore, as it's their responsibility now. + */ + if (ping_state->owner && ping_state->owner !=3D current) + goto out; + + top_waiter =3D futex_top_waiter(hb, key); + /* + * There are no other waiters but we leave the WAITERS bit set + * (with no owner or ping_state) to be cleaned up at a later unlock, + * at the cost of an extra syscall at the next lock operation, to + * keep things simple. + */ + if (!top_waiter) + goto out; + get_ping_state(ping_state); + get_task_struct(top_waiter->task); + wake_q_add_safe(&wake_q, top_waiter->task); + ping_state_update_owner(ping_state, top_waiter->task); + +out: + raw_spin_unlock_irq_wake(&ping_state->ping_mutex.wait_lock, &wake_q); +} + /* Returns >0 if lock acquired, <0 on error */ static int futex_trylock_ping_state(u32 __user *uaddr, struct futex_pi_state *ping_state) @@ -365,18 +402,31 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int f= lags, ktime_t *time, } =20 out_unqueue: + /* + * We got a pending signal or timeout, but the futex was handed + * off to us. Fix up the return value to indicate success. + */ + if ((ret =3D=3D -EINTR || ret =3D=3D -ETIMEDOUT) && + ping_mutex_owner(&q.ping_state->ping_mutex) =3D=3D current) + ret =3D 0; + + /* + * We are the pi_state owner but don't own the futex. + * This can happen if we get picked by the previous + * owner but get out without acquiring the lock for + * some reason. + * Wake up the next waiter and give the ping_state to them. + */ if (ret !=3D 0 && q.ping_state->owner =3D=3D current) { - /* - * We are pi_state owner but don't own the futex. - * This can happen if we get picked by the previous - * owner but get out without acquiring the lock for - * some reason. - * A later commit addresses this. - */ - WARN_ON_ONCE(1); - } - /* This also puts the ping_state */ - futex_unqueue_ping(&q); + if (!plist_node_empty(&q.list)) + __futex_unqueue(&q); + give_ping_state_to_next_waiter(hb, &q.key, + q.ping_state); + put_ping_state(q.ping_state); + } else + /* This also puts the ping_state */ + futex_unqueue_ping(&q); + out_unlock: futex_q_unlock(hb); __release(q.lock_ptr); --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pj1-f70.google.com (mail-pj1-f70.google.com [209.85.216.70]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 8208C3B5E08 for ; Thu, 17 Sep 2026 04:34:04 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.216.70 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619647; cv=none; b=Jo/YkKhC+6xVBuc7ncM+oxrX8Ndst+HpWVZtzE0K/Re2YWkK9nqnQZGzdlDpvc7if2clDMT9RR5rM9OPiIOohUW+Rd+/wDjYcr3Y/DbOoEgw0YZI2J1vprzRLv61CnWTYu1KjhzfE/Nfjz1GO8QC5Tl+FrYya0GMhFddB5ehmzk= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619647; c=relaxed/simple; bh=DzM1wwGRb2tHik98ddoG6RceyOa+DELhGaixssLReuc=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=HIQzDOr8dwLg+6nekVKmRDN4gyFhEQsQK6blyVxfHgGbcZ/vhnrrl6amC/I+Cwt6WaEkxHkNXkWyj7yGlTpRrgT5Q3Ji6biYt2Onxh4wcopM0543ATdHUavSQqTo+WVlDi67LelXaBje5HLVlzywf3eAMU08QY4cG0Jj2tNuiyg= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=djDDNR8T; arc=none smtp.client-ip=209.85.216.70 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="djDDNR8T" Received: by mail-pj1-f70.google.com with SMTP id 98e67ed59e1d1-398d0010cfaso838300a91.3 for ; Wed, 16 Sep 2026 21:34:04 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619644; x=1790224444; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=ARME4+OC5I5xyK4kMYGfPWWw6m/P7auePB5jMoGWdY0=; b=djDDNR8TZ9Oz06EJdza++P41hV3ZhzOAzgT6qq55gTziYHqF1fMO9QHx3cZ2ED0vjQ YAtuTxeRyRd1H5erJ6wJf4GZtdtwuSyCbyeF4S8eEu7KGYMa1aV5QLADFsl++LKUWVk3 qSDfIJQIGr0SnXu9ezXU6LrgnoQQM444r/pfMfFyB28wkm1762q09fdw3NjfDTSG70wr LpAVDG1gbSl901XSbtPHhgvizWis2vpugtU57qK3zMolbV8i5yvzSZegjQE6pSxkGjyJ RK+vBeG6kT2fJtCa4Cd8iNzIOfcL1Rt9q/aEnRlvUfe5cS7DL3vNPdtXN514W6qOV6cP rpng== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619644; x=1790224444; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=ARME4+OC5I5xyK4kMYGfPWWw6m/P7auePB5jMoGWdY0=; b=kvbxknSVbWoA8mJgRFhntqH57fCjJoiOT88pdR1r/azFJfXjMJ+JpRnmglQPhotW/n LyHNrsNu9oFsQL2FRG7Y/isnoCWRkv905GSH+ix/L2Q1AJsBfLzFd/66WWzFMVDMLxeN ziXTRbV2/Hh7Z+g2O16vXu6gJSAOs2EWXtlHLdgG6V0VZmpaHWtT1B4hhwpuk9x4SsMh 62PqWLLW/jgi0upXtqTUjydyk6TmJDmawbyDvNhu/U7Xb/3/rjsb8zut1O3Ob8GHEBhy RBc6SXGDSArY1DUafT2URZvAUl4doIKlGX6WNQ8Q00kfSBRsCRfQay6uqP1Odfj6doVH 8p1w== X-Gm-Message-State: AFuF++l9SuViNlAnOuY853K9pSJIS5AO00wOYzfWkqQAA9IbjwQ46h94 Erqf+bEvZCjjg31UBXyIssHVwfXP/E1L1B4/lMc58n8fqJNeNYKYeA0RnSvjC2afCAeXwovLHRH RSAewY2BmutY2DnSIUnyVhm8yRnXKL7M8o5u38TkeIW3kUEE7J7NQh+0nDZt3PKyH18S/sYE4Rw SkHzszjcfyE3hzOcdJ0Kl4CjBZYFr8ibBYRIje8xkYBlOGpvHd96U640o= X-Received: from pjbmj9.prod.google.com ([2002:a17:90b:3689:b0:39d:bebd:df7e]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a17:90b:3b91:b0:381:6c5:3f63 with SMTP id 98e67ed59e1d1-39e1e2bf7a7mr15545672a91.6.1789619643055; Wed, 16 Sep 2026 21:34:03 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:31 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-8-suleiman@google.com> Subject: [RFC PATCH 07/12] futex: Make FUTEX_*_PING use Proxy Execution. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Set the locker side proxy execution bits, so that a task blocked on a PING futex can act as a donor. Signed-off-by: Suleiman Souhlal --- include/linux/futex.h | 11 +++++++++++ include/linux/sched.h | 12 ++++++++++++ kernel/futex/ping.c | 7 +++++++ kernel/sched/core.c | 7 +++++++ 4 files changed, 37 insertions(+) diff --git a/include/linux/futex.h b/include/linux/futex.h index b1b422b480fb..84eca792e44a 100644 --- a/include/linux/futex.h +++ b/include/linux/futex.h @@ -170,4 +170,15 @@ static inline struct task_struct *ping_mutex_owner(str= uct ping_mutex *ping_mutex return READ_ONCE(ping_mutex->owner); } =20 +static inline void ping_mutex_lock_wait_lock(struct ping_mutex *ping_mutex) +{ + lockdep_assert_irqs_disabled(); + raw_spin_lock(&ping_mutex->wait_lock); +} + +static inline void ping_mutex_unlock_wait_lock(struct ping_mutex *ping_mut= ex) +{ + raw_spin_unlock(&ping_mutex->wait_lock); +} + #endif /* _LINUX_FUTEX_H */ diff --git a/include/linux/sched.h b/include/linux/sched.h index a7de5c496e3c..200f41c38333 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -835,6 +835,7 @@ struct task_ipi_mask { }; enum blocked_on_type { BO_T_NONE, BO_T_MUTEX, + BO_T_PING_FUTEX, }; =20 struct blocked_on_lock { @@ -2257,6 +2258,13 @@ static inline void __set_task_blocked_on(struct task= _struct *p, void *m, p->blocked_on.type =3D type; } =20 +static inline void set_task_blocked_on(struct task_struct *p, void *m, + enum blocked_on_type type) +{ + guard(raw_spinlock_irqsave)(&p->blocked_lock); + __set_task_blocked_on(p, m, type); +} + static inline void __clear_task_blocked_on(struct task_struct *p, void *m) { /* Currently we serialize blocked_on under the task::blocked_lock */ @@ -2278,6 +2286,10 @@ static inline void clear_task_blocked_on(struct task= _struct *p, void *m) } =20 #else +static inline void set_task_blocked_on(struct task_struct *p, void *m, + enum blocked_on_type type) +{ +} static inline void __clear_task_blocked_on(struct task_struct *p, void *m) { } diff --git a/kernel/futex/ping.c b/kernel/futex/ping.c index ebcd3c4a7793..689f149f7150 100644 --- a/kernel/futex/ping.c +++ b/kernel/futex/ping.c @@ -370,6 +370,9 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int fla= gs, ktime_t *time, =20 queued =3D false; while (1) { + set_task_blocked_on(current, &q.ping_state->ping_mutex, + BO_T_PING_FUTEX); + set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE); if (!queued) { futex_queue(&q, hb, current); @@ -382,6 +385,9 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int fla= gs, ktime_t *time, =20 futex_do_wait(&q, to); =20 + clear_task_blocked_on(current, + &q.ping_state->ping_mutex); + futex_q_lockptr_lock(&q); if (to && !to->task) { ret =3D -ETIMEDOUT; @@ -511,6 +517,7 @@ int futex_unlock_ping(u32 __user *uaddr, unsigned int f= lags) /* Leave it queued, it gets unqueued on the lock side */ get_task_struct(top_waiter->task); wake_q_add_safe(&wake_q, top_waiter->task); + clear_task_blocked_on(top_waiter->task, &ping_state->ping_mutex); spin_unlock(&hb->lock); =20 /* diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 2e8fe4b9bb88..1d35b1d90ea2 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -68,6 +68,7 @@ #include #include #include +#include =20 #ifdef CONFIG_PREEMPT_DYNAMIC # ifdef CONFIG_GENERIC_IRQ_ENTRY @@ -158,6 +159,8 @@ static inline struct task_struct *__blocked_on_owner(st= ruct blocked_on_lock *bo) return NULL; case BO_T_MUTEX: return __mutex_owner(bo->lock); + case BO_T_PING_FUTEX: + return ping_mutex_owner(bo->lock); default: WARN_ON_ONCE(1); return NULL; @@ -6907,6 +6910,8 @@ lock_blocked_on_lock(struct blocked_on_lock *bo) { if (bo->type =3D=3D BO_T_MUTEX) raw_spin_lock(&((struct mutex *)bo->lock)->wait_lock); + else if (bo->type =3D=3D BO_T_PING_FUTEX) + ping_mutex_lock_wait_lock(bo->lock); else WARN_ON_ONCE(1); } @@ -6916,6 +6921,8 @@ unlock_blocked_on_lock(struct blocked_on_lock *bo) { if (bo->type =3D=3D BO_T_MUTEX) raw_spin_unlock(&((struct mutex *)bo->lock)->wait_lock); + else if (bo->type =3D=3D BO_T_PING_FUTEX) + ping_mutex_unlock_wait_lock(bo->lock); else WARN_ON_ONCE(1); } --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pl1-f199.google.com (mail-pl1-f199.google.com [209.85.214.199]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 8945A3C4B89 for ; Thu, 17 Sep 2026 04:34:05 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.214.199 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619651; cv=none; b=sFR4AHhLcEdlCWwMtTtrhYGo8Dk7fNd7cArqcRR26BK0faHZEXt3pZ3P0WW4CMhqtAP3CCiTP4+QeZpjeSw1dtok0TZVj3/repPxlRwr4nrfDmVb6C87xe/fD1tBaWJs8y/VLV90HqKtaiV7O8X4xxR8IcG6xr7Rsi31o1KrWS8= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619651; c=relaxed/simple; bh=0glP17tViegVOrpC8YhTjfFSd82Ld3O+ArMRj7ieEOM=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=d3PEqNN4MzA6D6lbIl0PPYNV5x0AjSE0fExyVXUmLyeBWGVlE9GNGOjKoY3AK9Ncr62Nfix74Tu9ZhuYhIFMmUZRaf0I6rRHGggg3mKMkKiFdBil8vLliSDEMTShh30Ti1a+gMVfoIzD9Fe/C7249BdiOEXCKoZ3/5BA5MZ63YQ= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=RAbqJs52; arc=none smtp.client-ip=209.85.214.199 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="RAbqJs52" Received: by mail-pl1-f199.google.com with SMTP id d9443c01a7336-2d6f80c76e6so8937425ad.3 for ; Wed, 16 Sep 2026 21:34:05 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619645; x=1790224445; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=10JHcTrzupSF1uhcqjTzPBoiXPS1yfbQ+6EhNUxQ2A4=; b=RAbqJs52RFEfyKIVYgt+cPUT+BYlMfrszsVtxoqvrhrKxpnvgY3zHb8oNiPrX0kYJ6 v46mI7ZWoCdDVk9Y4hMywhmk57lucQL2YypmLEsUDdTp8Jx2su9oFQcplZro7ZfEL8lY 5sOHH0MPo97IjtI6ICePGJpiHWFs5E7YOHipAKRZTFyHzaTREJHAuZH2SaBQcpGq+hYJ pXGoP9+bpdTOPZEMWeyyNTWqEkP+RqGlgcm9DGAsftFKP5+75adBzAVzNY5u+zNmqTnI H/afQf4gO1zBCeP1EEEOQH6G4X75sKKEi9K9gtYszuS6X7TyxfhaYvDu3LtBxSu1xpeO 1ipg== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619645; x=1790224445; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=10JHcTrzupSF1uhcqjTzPBoiXPS1yfbQ+6EhNUxQ2A4=; b=JHE+SHmOWU+YoxnIR+ogRofmdVMmSesb3yBRxIfFbygvU1RjdWzHHHMzLls832ZcV7 Ue5wAyZ+k+rFzC2T4129zdzkUlemoCHmWV0tKBJLU/4gO24/PwvJqVriQxCCFIaKWTzH 28pi7pDAWJO5JhR9ar7XVYCsIo6L7PUwbvVWZSCrvWtMmh4pcBX+VhqiHUsS/tudteau WYxKMxuJKgB9SFS6ihnN34mRvgE+38RdIqFRi2Mn6PmZP4nvILsKzPgZSWn5AUcN139J Imd9LSSi2BJZzktJe/G3RNhdLekVJ/ovQUwsZBT+uCptVXwrFw9cMHumaXQb0xs49xOr qZRA== X-Gm-Message-State: AFuF++kQnkWub2Tr1ggyMH0tZrrYyqXl/8WI8YHDAGakjRHWWFfVK3pF ySikycqFqoyTDjkBDaWRAjUoJ8BJ3RPiZ2t+s/GwLVJ8dYLs4I1P5hun7bQqUSObB1PFtXWtbMD ftDL6gtlVMaHVQioqahXSy7X/7VyGQ/YKt1JlZufLR/ODJ+eCvN7j7cEE9ecJCz6O2jRlxJVSAv TUSl73ZI+xRqKrZxHcKs/StssqT/F259GftU4/w2TzTEzWpt8vcFfo2G8= X-Received: from plcz17.prod.google.com ([2002:a17:903:4091:b0:2dd:1604:ef9d]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a17:902:da85:b0:2dd:89b1:7fd3 with SMTP id d9443c01a7336-2dd8dc06d6amr102021715ad.4.1789619644397; Wed, 16 Sep 2026 21:34:04 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:32 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-9-suleiman@google.com> Subject: [RFC PATCH 08/12] futex: Implement PING futex handoff. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Implement handoff for PING futexes, to prevent starvation in case of a waiter being repeatedly stolen from. Currently engages after being stolen from once, after which the next unlocker hands off the futex such that it can't be stolen. Signed-off-by: Suleiman Souhlal --- kernel/futex/futex.h | 2 ++ kernel/futex/ping.c | 45 ++++++++++++++++++++++++++++++++++---------- 2 files changed, 37 insertions(+), 10 deletions(-) diff --git a/kernel/futex/futex.h b/kernel/futex/futex.h index c0560d30aaaa..113289779dd1 100644 --- a/kernel/futex/futex.h +++ b/kernel/futex/futex.h @@ -174,6 +174,8 @@ struct futex_pi_state { =20 struct task_struct *owner; refcount_t refcount; + bool handoff; + bool pickup; =20 union futex_key key; } __randomize_layout; diff --git a/kernel/futex/ping.c b/kernel/futex/ping.c index 689f149f7150..5bf2293bf582 100644 --- a/kernel/futex/ping.c +++ b/kernel/futex/ping.c @@ -7,7 +7,8 @@ #include "futex.h" =20 static int futex_trylock_ping_state(u32 __user *uaddr, - struct futex_pi_state *ping_state); + struct futex_pi_state *ping_state, + bool handoff); =20 static void ping_state_update_owner(struct futex_pi_state *ping_state, struct task_struct *new_owner) @@ -144,7 +145,8 @@ static void give_ping_state_to_next_waiter(struct futex= _hash_bucket *hb, =20 /* Returns >0 if lock acquired, <0 on error */ static int futex_trylock_ping_state(u32 __user *uaddr, - struct futex_pi_state *ping_state) + struct futex_pi_state *ping_state, + bool handoff) { struct task_struct *owner; u32 uval, new, newtid; @@ -152,13 +154,15 @@ static int futex_trylock_ping_state(u32 __user *uaddr, =20 ret =3D 0; raw_spin_lock_irq(&ping_state->ping_mutex.wait_lock); + ret =3D futex_get_value_locked(&uval, uaddr); + if (ret) + goto err; owner =3D ping_mutex_owner(&ping_state->ping_mutex); if (owner =3D=3D NULL) { newtid =3D task_pid_vnr(current); =20 - ret =3D futex_get_value_locked(&uval, uaddr); - if (ret) - goto err; + WARN_ON_ONCE(ping_state->handoff || ping_state->pickup); + if (uval & FUTEX_TID_MASK) { ret =3D -EAGAIN; goto err; @@ -171,7 +175,19 @@ static int futex_trylock_ping_state(u32 __user *uaddr, ping_state_update_owner(ping_state, current); WRITE_ONCE(ping_state->ping_mutex.owner, current); ret =3D 1; - } + } else if (ping_state->pickup) { + if (owner !=3D current) { + ret =3D -EAGAIN; + goto err; + } + if ((uval & FUTEX_TID_MASK) !=3D task_pid_vnr(current)) { + ret =3D -EINVAL; + goto err; + } + ping_state->pickup =3D 0; + ret =3D 1; + } else if (handoff && !ping_state->handoff) + ping_state->handoff =3D 1; raw_spin_unlock_irq(&ping_state->ping_mutex.wait_lock); return ret; =20 @@ -233,7 +249,7 @@ static int futex_lock_ping_atomic(u32 __user *uaddr, _ps =3D top_waiter->ping_state; if (_ps =3D=3D NULL) return -EINVAL; - ret =3D futex_trylock_ping_state(uaddr, _ps); + ret =3D futex_trylock_ping_state(uaddr, _ps, false); if (ret > 0) { /* We stole the lock from the top waiter. */ raw_spin_lock_irq(&_ps->ping_mutex.wait_lock); @@ -297,7 +313,7 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int fla= gs, ktime_t *time, struct hrtimer_sleeper timeout, *to; struct task_struct *exiting; struct futex_q q =3D futex_q_init; - bool queued; + bool queued, should_handoff; int ret; =20 if (refill_pi_state_cache()) @@ -369,6 +385,7 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int fla= gs, ktime_t *time, } =20 queued =3D false; + should_handoff =3D false; while (1) { set_task_blocked_on(current, &q.ping_state->ping_mutex, BO_T_PING_FUTEX); @@ -398,13 +415,15 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int f= lags, ktime_t *time, goto out_unqueue; } =20 - ret =3D futex_trylock_ping_state(uaddr, q.ping_state); + ret =3D futex_trylock_ping_state(uaddr, q.ping_state, + should_handoff); if (ret > 0) { /* Got the futex */ ret =3D 0; goto out_unqueue; } else if (ret < 0) goto out_unqueue; + should_handoff =3D true; } =20 out_unqueue: @@ -526,6 +545,13 @@ int futex_unlock_ping(u32 __user *uaddr, unsigned int = flags) * no top_waiter. */ new =3D FUTEX_WAITERS; + if (ping_state->handoff) { + new |=3D task_pid_vnr(top_waiter->task); + ping_state->handoff =3D 0; + ping_state->pickup =3D 1; + WRITE_ONCE(ping_state->ping_mutex.owner, top_waiter->task); + } else + WRITE_ONCE(ping_state->ping_mutex.owner, NULL); ret =3D lock_pi_update_atomic(uaddr, uval, new); if (ret) { raw_spin_unlock_irq(&ping_state->ping_mutex.wait_lock); @@ -543,7 +569,6 @@ int futex_unlock_ping(u32 __user *uaddr, unsigned int f= lags) } =20 ping_state_update_owner(ping_state, top_waiter->task); - WRITE_ONCE(ping_state->ping_mutex.owner, NULL); raw_spin_unlock_irq_wake(&ping_state->ping_mutex.wait_lock, &wake_q); put_ping_state(ping_state); return 0; --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pf1-f197.google.com (mail-pf1-f197.google.com [209.85.210.197]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id C70323BED23 for ; Thu, 17 Sep 2026 04:34:06 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.210.197 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619651; cv=none; b=Z8cmdciA+YwKSSpxw6vzdMhdQgWSMXYLIpbCXmZU15DMIU+Tj0PDJNnUCIaQ/QfZaq6ewNrpIfY9GKEfZjaUqoa2CY6UE2nqoHELlIMfjy3THSH0e3/0mbD2r04+7+Uzq6fZx7Pi1UT71lZ9JDQLVhszmvzptVhyzvYNdKK0I7E= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619651; c=relaxed/simple; bh=EhzHdBWaVd2CA9AxJQCiWMZlyt1B78tq9q2Ffexiep4=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=OzDvIR8S6LlxAAgWo92sIibojN4lquLkAS6zqD1yqinXXsAxQfU1I7stcPIDqqQ9hnzzh+6let0s0bVSMuZtBtQenzaR3bDN58YUVEeLMMTHzqm9s97Zy0unm0jxFkx3iaJAWZa1Xas+fJLut4YlE7KHPu1uEBCbYa9hIyMuem8= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=D7ziT+dx; arc=none smtp.client-ip=209.85.210.197 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="D7ziT+dx" Received: by mail-pf1-f197.google.com with SMTP id d2e1a72fcca58-86b3c58d686so765193b3a.3 for ; Wed, 16 Sep 2026 21:34:06 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619646; x=1790224446; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=48US/8NeMBTIu8PjHXwAbDh6N6z1er052DCS5yC+V3k=; b=D7ziT+dx7hz0U0OorMZ94kqWCStA4nxePKYH+vj8imiFfD1tJA2KD5u31OqaPG0Iwg WQWStHhh+TZJxGcYQ4bK8JSnxpURuUBBTcwORq8YNljFlFRGrN++fEvj2GzeX2Kx2Z/j e3QmbM1xtpNMZjSumESJ06DUYkmwEkudAsHkBEvYXAySzsEaqCU8ql3pPm7U2O3ClGmi ogvpjWNYR9YJty4x5SSkZGcr9oJfhDjWJbq35BqmySlGgsUquhGBplwaUMw0xuQ72bVM yzSNWw3e8nYURSu/vsjjVqUxzNaq8yH+olncUnGlixQ9fRY1QlUb+QyL/zTTAPFQfloh ymvQ== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619646; x=1790224446; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=48US/8NeMBTIu8PjHXwAbDh6N6z1er052DCS5yC+V3k=; b=Oc0F6KpGBaIZVx59edB6lO+BhhUIhaLl+HffEF/BLZJuWRuiIf43FslFHUK6T/jGhd df52AHW+wmPtPvLBhNRBJaYyc0BMYS1PijRh1jK2Rjfrr77M/Lz5QQE0BFGR2berqI+G nkkXHVHE7VvpMcf5Y5sDT4XtcREKSQAEybgo/kfsew9On9s49IQOoTfYeJQb2e002iKk gEAL5JRxXKlO0W5XP/uilVu9UYHK95zqH3e8jnuzCoTlkhbZ+Fs6nHP4n4j5meHBTp6S Kh5X9KqO+zEm6xjx+fDXVFEd4HF/yJFshWezILVlAu4AhFNNye96Cb22+3Xt7jwZ4LAU Iy2Q== X-Gm-Message-State: AFuF++nPSutiNo2W+qF5G9I5MXAjezumlzvJ/qcWFZvBK7nG9HBwoOcb 2ofOmFIyvAT99uE24z5N9Gbef+pkYhTgmZ6uTN3XPkwuF+dASXP1Ns+xEYdrdzwpQlLJN3tjlKz lRQZVeDH9saxF055zAlF9Klrp0022yD6+SlGPAngPJNCu1ygv5KNtNdeuW+X7CwiF4M71D4Jg0x S2UFvlCOSBS1MRvRgaMw8eJFMgxSBWdmFq/7eInnxTTEbmeyZUHQJfQCY= X-Received: from pfbcz3.prod.google.com ([2002:aa7:9303:0:b0:848:41c3:4371]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a05:6a00:6087:b0:872:20f2:3294 with SMTP id d2e1a72fcca58-8723dd0f2b6mr11493394b3a.21.1789619645832; Wed, 16 Sep 2026 21:34:05 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:33 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-10-suleiman@google.com> Subject: [RFC PATCH 09/12] futex: Wake up donor in PING futex unlock. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Now that PING blockers can act as donors, it makes sense to wake up the donor instead of the top_waiter when unlocking. Signed-off-by: Suleiman Souhlal --- kernel/futex/ping.c | 56 +++++++++++++++++++++++++++++++++++++-------- 1 file changed, 46 insertions(+), 10 deletions(-) diff --git a/kernel/futex/ping.c b/kernel/futex/ping.c index 5bf2293bf582..df701b437f16 100644 --- a/kernel/futex/ping.c +++ b/kernel/futex/ping.c @@ -478,10 +478,46 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int f= lags, ktime_t *time, return ret; } =20 +#ifdef CONFIG_SCHED_PROXY_EXEC +static inline +struct task_struct *ping_current_proxy_donor(struct futex_pi_state *ping_s= tate) +{ + struct task_struct *ret =3D NULL; + + if (sched_proxy_exec()) { + struct task_struct *donor; + + raw_spin_lock(¤t->blocked_lock); + donor =3D current->blocked_donor; + if (donor) { + void *ping_lock =3D (void *)&ping_state->ping_mutex; + + raw_spin_lock_nested(&donor->blocked_lock, + SINGLE_DEPTH_NESTING); + if (__get_task_blocked_on(donor) =3D=3D ping_lock) { + ret =3D get_task_struct(donor); + __clear_task_blocked_on(donor, ping_lock); + current->blocked_donor =3D NULL; + } + raw_spin_unlock(&donor->blocked_lock); + } + raw_spin_unlock(¤t->blocked_lock); + } + return ret; +} +#else +static inline +struct task_struct *ping_current_proxy_donor(struct futex_pi_state *ping_s= tate) +{ + return NULL; +} +#endif + int futex_unlock_ping(u32 __user *uaddr, unsigned int flags) { struct futex_pi_state *ping_state; u32 new, uval, vpid =3D task_pid_vnr(current); + struct task_struct *next; union futex_key key =3D FUTEX_KEY_INIT; struct futex_q *top_waiter; DEFINE_WAKE_Q(wake_q); @@ -528,15 +564,15 @@ int futex_unlock_ping(u32 __user *uaddr, unsigned int= flags) if (!ping_state) goto out_unlock; raw_spin_lock_irq(&ping_state->ping_mutex.wait_lock); - if (ping_state->owner !=3D current) { - raw_spin_unlock_irq(&ping_state->ping_mutex.wait_lock); - goto out_unlock; - } + + next =3D ping_current_proxy_donor(ping_state); get_ping_state(ping_state); /* Leave it queued, it gets unqueued on the lock side */ - get_task_struct(top_waiter->task); - wake_q_add_safe(&wake_q, top_waiter->task); - clear_task_blocked_on(top_waiter->task, &ping_state->ping_mutex); + if (next =3D=3D NULL) { + next =3D get_task_struct(top_waiter->task); + clear_task_blocked_on(next, &ping_state->ping_mutex); + } + wake_q_add_safe(&wake_q, next); spin_unlock(&hb->lock); =20 /* @@ -546,10 +582,10 @@ int futex_unlock_ping(u32 __user *uaddr, unsigned int= flags) */ new =3D FUTEX_WAITERS; if (ping_state->handoff) { - new |=3D task_pid_vnr(top_waiter->task); + new |=3D task_pid_vnr(next); ping_state->handoff =3D 0; ping_state->pickup =3D 1; - WRITE_ONCE(ping_state->ping_mutex.owner, top_waiter->task); + WRITE_ONCE(ping_state->ping_mutex.owner, next); } else WRITE_ONCE(ping_state->ping_mutex.owner, NULL); ret =3D lock_pi_update_atomic(uaddr, uval, new); @@ -568,7 +604,7 @@ int futex_unlock_ping(u32 __user *uaddr, unsigned int f= lags) } } =20 - ping_state_update_owner(ping_state, top_waiter->task); + ping_state_update_owner(ping_state, next); raw_spin_unlock_irq_wake(&ping_state->ping_mutex.wait_lock, &wake_q); put_ping_state(ping_state); return 0; --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pj1-f72.google.com (mail-pj1-f72.google.com [209.85.216.72]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id A0D433A3E97 for ; Thu, 17 Sep 2026 04:34:10 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.216.72 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619652; cv=none; b=sY6kPfSHMymx3EvsbSpZkt29kZYVId6mcuh6yCIIa9ppa2Fj1fxO/3IAVlNw2FbMotYJcW4BLygjjkaNvOKTbJgjTfP7C23i84Tg7TVoSekMUJbmvaJpU1m1EJcg6byz+QsWQduNUxT5aleBppXN7bTWchEZqzsEDGfRM5BiO7U= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619652; c=relaxed/simple; bh=FlAJUxDZTy5DSJRi9+xcyW9t9UxoCG7ph8kRU5Qs0dk=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=qQvCQLIKtyMsAO97OG5SuaBgurMl1qVKXhOORJs1a39d28WMnfz3pu3x9Er96IWt6Z3GUYr2rXrR0vw3YUuWfrznIjcVWTi6T3AWaYja0eCw07uL2e5qTyXkCFUExv/sz0R9EPMy2bTS07eFYm9iQgBxkSQCHzoJt2CDc9ncZAE= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=RK14xgq9; arc=none smtp.client-ip=209.85.216.72 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="RK14xgq9" Received: by mail-pj1-f72.google.com with SMTP id 98e67ed59e1d1-39b77130a7fso793864a91.2 for ; Wed, 16 Sep 2026 21:34:10 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619648; x=1790224448; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=wrAa7nuWDHelhYsLr/ekNyZ1GNe9/NxFjXsAX7Nn//k=; b=RK14xgq9211rmze2EJ+F1XiinhPrCBbqcrcHxErR7I9qGYtetoX8IU9Erqz7r8SRIH ujke+e+L1Qp2Cb1xymUkTrpxLXLs5km7fVViCQZzFR6rxsfFvPwJMDYzbDjPmir+GtG9 WQwi3pjkyFVMnF2JT5/hBTxDKJV2WBlsoOdCISB8yt4ZLxfKgWMyxhARzLkKH/K9MUcs CTZjjkkrKfJKcKKuFBxpewwh76t1KaDJtJZULRXCPwcxyDuEXqRsLswfJb/r+YXXq0ic sNqf84OJlaNlMPu7ADYWa87r4SxieVLDtso4yGkjU0hB5fBeHVG1CvdeyiQBMDtV4jfv H/lA== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619648; x=1790224448; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=wrAa7nuWDHelhYsLr/ekNyZ1GNe9/NxFjXsAX7Nn//k=; b=WI0n7J+ATZdlLCB2oloXJE6hoE0AJA0wiqDbayQHw8Qdj0KAwZdYTskjnLEaga/OwL jR6no1SQpcZXdD25jVq1Hs+oe50YtJkF243QCeU7Yh25GoYCnMDbCNGrBiYIOqP2vZkR 5ORUW2ETd6cQbhExn3Kus+6mvTOsJHIEUpJMEWee+/ygNUZ6c5G2kmA+876dZL66Wn9I T8LrlxWVWwLlEnq7cNNpKpjaTqFq3ygwXJwzed2LBT7VNe2vGGUtQFYDtcJzE07/Z6qI xuNzuxn4WH5MEqr713xT5rtH2B1I46saj3NN53Wa4zLoTm73xQvIcXvj604rjCAiboEB bNxA== X-Gm-Message-State: AFuF++n07Ulz433fBN2f5NfhUw/C2+WrLrGTiCB8g5ahjMBFdM3NQtQc B7DUQwFN/nesID9YZ8qM2CvWWjo/XVKe77rH29oCIbtVMBf3cVP9XULZjTspgWdOV7W764g0S2Q gXVSuH7Wg9FbwCxrqIuZhpTzLOWSrbwMlmo9e6jtkQbrjsX714nfI36YI+guFPogFVreL4S0Jlz R6HHqprFYMP4tVpv+j8ssvlxGrthpzKplxfW0tk4s5TQ1HbRoTBYodyeo= X-Received: from pjbmv12.prod.google.com ([2002:a17:90b:198c:b0:39e:ac:d0ad]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a17:90a:6c8f:b0:39e:345b:3322 with SMTP id 98e67ed59e1d1-39e345b3c87mr3835430a91.23.1789619647121; Wed, 16 Sep 2026 21:34:07 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:34 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-11-suleiman@google.com> Subject: [RFC PATCH 10/12] futex: Optimistic spinning for PING futexes. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" This allows a thread to spin on the owner of the futex instead of unconditionally spinning, if the owner is currently running on another CPU. TODO: Currently allows every task to attempt to spin at the same time. In the future, spinning will probably either use an osq or only the top waiter will be allowed to wait. We've seen some long latencies when using an osq that we still need to investigate. Signed-off-by: Suleiman Souhlal --- kernel/futex/ping.c | 80 +++++++++++++++++++++++++++++++++++++-------- 1 file changed, 67 insertions(+), 13 deletions(-) diff --git a/kernel/futex/ping.c b/kernel/futex/ping.c index df701b437f16..e5765b2d3834 100644 --- a/kernel/futex/ping.c +++ b/kernel/futex/ping.c @@ -301,6 +301,50 @@ static int futex_lock_ping_atomic(u32 __user *uaddr, return attach_to_pi_owner(uaddr, newval, key, ps, exiting, true); } =20 +/* + * Returns >0 on successfully having taken the lock, <0 on error + */ +static int ping_spin_or_trylock(u32 __user *uaddr, + struct futex_pi_state *ping_state, + bool handoff) +{ + struct task_struct *owner =3D ping_mutex_owner(&ping_state->ping_mutex); + struct task_struct *new; + int ret; + + if (owner && !owner_on_cpu(owner)) + return 0; + + /* + * XXX We'll want to try to limit spinning either with an osq or + * by only allowing the top waiter to spin. + */ + + while (1) { + new =3D ping_mutex_owner(&ping_state->ping_mutex); + ret =3D 0; + if (!owner || !new || new =3D=3D current) { + ret =3D futex_trylock_ping_state(uaddr, ping_state, + handoff); + if (ret !=3D 0) + break; + /* Spin on new owner if we didn't get the lock */ + owner =3D ping_mutex_owner(&ping_state->ping_mutex); + goto next; + } + if (new !=3D owner) { + owner =3D ping_mutex_owner(&ping_state->ping_mutex); + goto next; + } + if (!owner_on_cpu(owner) || need_resched()) + break; +next: + cpu_relax(); + } + + return ret; +} + /* * Return values: * < 0: error. @@ -387,9 +431,6 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int fla= gs, ktime_t *time, queued =3D false; should_handoff =3D false; while (1) { - set_task_blocked_on(current, &q.ping_state->ping_mutex, - BO_T_PING_FUTEX); - set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE); if (!queued) { futex_queue(&q, hb, current); @@ -400,6 +441,26 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int fl= ags, ktime_t *time, __release(q->lock_ptr); } =20 + preempt_disable(); + set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE); + ret =3D ping_spin_or_trylock(uaddr, q.ping_state, + should_handoff); + if (ret > 0) { + /* Got the futex */ + ret =3D 0; + preempt_enable(); + futex_q_lockptr_lock(&q); + goto out_unqueue; + } else if (ret < 0) { + preempt_enable(); + futex_q_lockptr_lock(&q); + goto out_unqueue; + } + preempt_enable(); + + set_task_blocked_on(current, &q.ping_state->ping_mutex, + BO_T_PING_FUTEX); + futex_do_wait(&q, to); =20 clear_task_blocked_on(current, @@ -414,19 +475,12 @@ int futex_lock_ping(u32 __user *uaddr, unsigned int f= lags, ktime_t *time, ret =3D -EINTR; goto out_unqueue; } - - ret =3D futex_trylock_ping_state(uaddr, q.ping_state, - should_handoff); - if (ret > 0) { - /* Got the futex */ - ret =3D 0; - goto out_unqueue; - } else if (ret < 0) - goto out_unqueue; should_handoff =3D true; } =20 out_unqueue: + __set_current_state(TASK_RUNNING); + /* * We got a pending signal or timeout, but the futex was handed * off to us. Fix up the return value to indicate success. @@ -581,7 +635,7 @@ int futex_unlock_ping(u32 __user *uaddr, unsigned int f= lags) * no top_waiter. */ new =3D FUTEX_WAITERS; - if (ping_state->handoff) { + if (ping_state->handoff) { /* Don't handoff to donor */ new |=3D task_pid_vnr(next); ping_state->handoff =3D 0; ping_state->pickup =3D 1; --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pf1-f198.google.com (mail-pf1-f198.google.com [209.85.210.198]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id CF67F3C2B95 for ; Thu, 17 Sep 2026 04:34:09 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.210.198 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619651; cv=none; b=Cu2aSqEebvUljdQ6OnhbdRC4/YjMG19j1joveqgAeA9VGkfH9hpMA+D+izCVy9P+1DVjvGdZXrU++xrxLdakUMZaNNF5/paPYcOe/03m42cZ/eydT1TVri+paAUfJHIUYc55C6ONyiU0C1783lPlcvFVZikhC93byAggCHWm9Y4= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619651; c=relaxed/simple; bh=UoT8lmgVGUsVGRmIzQcQADrCNDlYHVGbUsQfdt5hgd0=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=WCtfQFC0BGimIInP1eWk70GEXWQYaQOQWULMHriz9hMT649qGYA1du22uztzwKpk4/llhHZa253G11pGEovj2AcX1h6MG1EpYn939D38n68pvnexeAOTDQuYDGcrrpRLfa2xGKH8HBOfhLmdudVCTy0O8U3l9MaVLrp6obZJrbk= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=Tmw6n+zp; arc=none smtp.client-ip=209.85.210.198 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="Tmw6n+zp" Received: by mail-pf1-f198.google.com with SMTP id d2e1a72fcca58-8688139460bso526325b3a.0 for ; Wed, 16 Sep 2026 21:34:09 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619649; x=1790224449; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:from:to:cc:subject:date:message-id:reply-to :content-type; bh=WjRJ7qIcoyYP80tPA6wEvSZrR9vPCGZkyHgHo5ANiJY=; b=Tmw6n+zpZu9RdbbbEan/2F55NeEHNnQO2/cjoTtzKeX+hMawJ2s5I3MhVQ+bMrd393 hzu56a+drw5AtiNtgpu3Cj5XGO4Mf2cZ5zo3ySBuUb0j8ly+8BmrCiVxP2eXu26I57M+ sOPb6hlrucnOAAg+vEHs3sLcRnNancw3q9oTGl4enKB/1R4FCgBmkVbX2dDB/abHqLu/ GSoo3QOBFvXZrhRd0Jx+OirN1rtB7/ph0XPJJKZ482bB/SqBniLa60KprRpv1fu5K8y4 RQRzhuiR54bhbfKmvf7lCfp5cugO8syhfhCnYsSiIy8VJvGHBevu0ASufW6t69riCKsC RYYw== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619649; x=1790224449; h=content-type:cc:to:from:subject:message-id:references:mime-version :in-reply-to:date:x-gm-message-state:from:to:cc:subject:date :message-id:reply-to:content-type; bh=WjRJ7qIcoyYP80tPA6wEvSZrR9vPCGZkyHgHo5ANiJY=; b=kKTt0EInU0knZ+FFA8jFuQtCvnQXpNY+eMZo6aiRGcuclJ2s8VXmIp7gy7ub6eaBfP XdCX1N01OAIfRAp0YSvhPxiuBG6BozNakIhyzKvOztCy6sHXAaEvMFFr0Px0EPHu4uYg IeavZfAXTFs3CYs0xTNM2k8arNTZyiXCl4EFp6qq6tRijShvgGAHi0vCWrii+Cc/x2JJ mt83t3I4VOH3wU/5nx2SplIN5h4bgIn6S3YGUFqAsY/24KzQ2L92wjTOXgupRWIaQ+Is AxaxfxZ/MwbsHCDyQjiiP6msepcTC0B60ppvJHYNdj7q4ukFc89UfkIx1YMK25/GsqjQ Nyeg== X-Gm-Message-State: AFuF++nE2cYw3yr/tnO81t1ndbHb/YpdxhtP+CeWJSMihrZA4mfAUeli vZ2KUdGPrO6vAXORtu2Bc+E4UsIDBTmOswuobkWjFzHzLEBtAy7MoeFVV9O98UhElTbmDoLAAI2 Ov5XH/DZfmhWLYsPk+EUqLwGvfi0KrG29jZSVPAAA548W8ZuyqghANnK7lJir1NhYQdGKaZ23d6 EdTCiovNur3yW5cLEz2kV6vMXcdaUblBJCVP2i/3VSr42mRhLcF3JnXg0= X-Received: from pgy13.prod.google.com ([2002:a63:184d:0:b0:cc5:120f:c9c0]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a05:6a20:3d95:b0:3d1:d188:b0fe with SMTP id adf61e73a8af0-3dd5f5d62e6mr12887707637.13.1789619648572; Wed, 16 Sep 2026 21:34:08 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:35 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-12-suleiman@google.com> Subject: [RFC PATCH 11/12] futex: Allow userspace stealing for PING futexes. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Similarly to how the kernel part of PING locking can steal the futex from the top waiter, it is also possible for userspace to also take advantage of this and try to steal without going to the kernel. When a new (contending) locker notices that the lock has been stolen from userspace, the ownership of the ping_state and the ping_mutex are fixed up to the real owner. TODO: Verify that the case where the ping_state owner exits with the futex having been user stolen is handled correctly. In other words, when ping_state->owner =3D task getting killed, ping_mutex.owner =3D NULL and uval =3D real owner (set by userspace). Right now, it seems like we might be doing the wrong thing in such cases. exit_ping_state_list() needs to detect such situations (by walking the uvals?) and fixup the ownerships. Signed-off-by: Suleiman Souhlal --- kernel/futex/core.c | 8 ++++++++ kernel/futex/futex.h | 2 ++ kernel/futex/pi.c | 12 ++++++++++-- kernel/futex/ping.c | 38 +++++++++++++++++++++++++++++++++++--- 4 files changed, 55 insertions(+), 5 deletions(-) diff --git a/kernel/futex/core.c b/kernel/futex/core.c index 56c7e2d5faac..023531c4b45a 100644 --- a/kernel/futex/core.c +++ b/kernel/futex/core.c @@ -1445,6 +1445,14 @@ static void exit_ping_state_list(struct task_struct = *curr) while (!list_empty(head)) { next =3D head->next; ping_state =3D list_entry(next, struct futex_pi_state, list); + /* + * XXX In the case when we are ping_state owner but + * the futex is actually owned by someone who stole it + * from userspace, is setting ping_state.owner =3D NULL here + * enough? Probably not, otherwise the ping_state could leak + * if they also exit before unlocking or someone else + * fixing up the ownership. + */ if (1) { CLASS(hbr, hbr)(&key); auto hb =3D hbr.hb; diff --git a/kernel/futex/futex.h b/kernel/futex/futex.h index 113289779dd1..922d130c5b44 100644 --- a/kernel/futex/futex.h +++ b/kernel/futex/futex.h @@ -413,6 +413,8 @@ extern int attach_to_pi_owner(u32 __user *uaddr, u32 uv= al, union futex_key *key, bool ping); extern void get_ping_state(struct futex_pi_state *ping_state); extern void put_ping_state(struct futex_pi_state *ping_state); +extern int fixup_ping_owner_after_user_steal(struct futex_pi_state *ping_s= tate, + u32 __user *uaddr, u32 uval); =20 /* * Express the locking dependencies for lockdep: diff --git a/kernel/futex/pi.c b/kernel/futex/pi.c index e7e6e347f97d..dfd2da5188b0 100644 --- a/kernel/futex/pi.c +++ b/kernel/futex/pi.c @@ -348,8 +348,16 @@ int attach_to_pi_state(u32 __user *uaddr, u32 uval, * state exists then the owner TID must be the same as the * user space TID. [9/10] */ - if (pid !=3D task_pid_vnr(pi_state->owner)) - goto out_einval; + if (pid !=3D task_pid_vnr(pi_state->owner)) { + if (!ping) { + goto out_einval; + } else { + ret =3D fixup_ping_owner_after_user_steal(pi_state, + uaddr, uval); + if (ret) + goto out_error; + } + } =20 out_attach: if (!ping) { diff --git a/kernel/futex/ping.c b/kernel/futex/ping.c index e5765b2d3834..4cef1c46b4c6 100644 --- a/kernel/futex/ping.c +++ b/kernel/futex/ping.c @@ -96,6 +96,22 @@ put_ping_state(struct futex_pi_state *ping_state) } } =20 +int fixup_ping_owner_after_user_steal(struct futex_pi_state *ping_state, + u32 __user *uaddr, u32 uval) +{ + struct task_struct *p; + + p =3D find_get_task_by_vpid(uval & FUTEX_TID_MASK); + if (p =3D=3D NULL) + return pi_handle_exit_race(uaddr, uval); + if (unlikely(p->flags & PF_KTHREAD)) + return -EPERM; + ping_state_update_owner(ping_state, p); + WRITE_ONCE(ping_state->ping_mutex.owner, p); + + return 0; +} + static void futex_unqueue_ping(struct futex_q *q) { if (!plist_node_empty(&q->list)) @@ -149,6 +165,7 @@ static int futex_trylock_ping_state(u32 __user *uaddr, bool handoff) { struct task_struct *owner; + pid_t pid; u32 uval, new, newtid; int ret; =20 @@ -163,8 +180,17 @@ static int futex_trylock_ping_state(u32 __user *uaddr, =20 WARN_ON_ONCE(ping_state->handoff || ping_state->pickup); =20 - if (uval & FUTEX_TID_MASK) { - ret =3D -EAGAIN; + /* + * No owner but a userspace TID means that it got stolen + * from userspace. + * Fix up the ownership. + */ + pid =3D uval & FUTEX_TID_MASK; + if (pid) { + ret =3D fixup_ping_owner_after_user_steal(ping_state, + uaddr, uval); + if (ret =3D=3D 0) + ret =3D -EAGAIN; goto err; } new =3D newtid | FUTEX_WAITERS; @@ -203,7 +229,7 @@ static int futex_trylock_ping_state(u32 __user *uaddr, case -EINVAL: break; default: - WARN_ON(1); + break; } return ret; } @@ -619,6 +645,12 @@ int futex_unlock_ping(u32 __user *uaddr, unsigned int = flags) goto out_unlock; raw_spin_lock_irq(&ping_state->ping_mutex.wait_lock); =20 + /* + * If we're unlocking a lock we stole from userspace, + * it's possible that we don't own the ping_state or the + * ping_mutex. But we'll give them to the next task here anyway. + */ + next =3D ping_current_proxy_donor(ping_state); get_ping_state(ping_state); /* Leave it queued, it gets unqueued on the lock side */ --=20 2.55.0.1082.g2b9226bbc0-goog From nobody Fri Sep 25 03:16:24 2026 Received: from mail-pf1-f197.google.com (mail-pf1-f197.google.com [209.85.210.197]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 992683C109A for ; Thu, 17 Sep 2026 04:34:14 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.210.197 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619665; cv=none; b=YSkbVKooqTjqkM3IjkRW9ZejtjZdy9Fy9wk+tZ0jLzqRgk3ehEY1wSuyHhFav8UA8TuwUuwDgEaPpkcfDoEE2LvumYIo6pVe88IPMv0VM4yCkIE/qUwK9ztV3sFt1+NwqfGr2+P29AF7P6xEFprq6DsVv6ceKCFfzfiFWPhut5A= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1789619665; c=relaxed/simple; bh=eehDtZ7OlP60D9pMffxyT/ekD05+h9cg0eM+4fu18fw=; h=Date:In-Reply-To:Mime-Version:References:Message-ID:Subject:From: To:Cc:Content-Type; b=QwXiTCLvbunPe7Ekp4QR2nOSSlfc+r+zKMM549ey+EFVxOSNVu9mr4EfyrUHl7h+4Q5bm9xdPVMwe/zwn+PArDDNBINwdfgAEfCnWFrAvdkaTGiowHhoicr8++O18Qiu6/z4pc/+ZiiXafgRmfL2BfEQMAD6wVcDMW/DVSTMMew= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=vRyir6W3; arc=none smtp.client-ip=209.85.210.197 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--suleiman.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="vRyir6W3" Received: by mail-pf1-f197.google.com with SMTP id d2e1a72fcca58-87088f3b83dso651998b3a.1 for ; Wed, 16 Sep 2026 21:34:14 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1789619654; x=1790224454; darn=vger.kernel.org; h=content-transfer-encoding:content-type:cc:to:from:subject :message-id:references:mime-version:in-reply-to:date:from:to:cc :subject:date:message-id:reply-to:content-type; bh=PUi0w321u1M0bPfHbTN2MqImPn3RRghNR/0evAaIXTo=; b=vRyir6W3HrN7FMZSKhZe0dn+JEXtS0dVk7TCkntPnBBNh2aF+okc62+PFJopiPYbE0 dNZzT2jexieJSY6/aoeWy79UlR8uVNuPGmOs8LsAwGXeM6vN82rZwFy5vX19BUWqIRHo ScWwhHob1xjTH5DBui6ep49YGpYbEJa7fUFQ43IuaFEOAPsYhBbnx2I+UpAdnoZ9p/dj pZqhEEuejp+bp2TF8QBvE0Y1BQrig2yimWo0sCvn1ToqxpsAnRCO+54vBevgZl955nX7 xWb8e2NasFGFZFmITIBJS1h5d36OMvZ8OkhbIWauuxLPu/4PhQ/l7rg15dj7xVdSuzW5 KjXQ== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20260707; t=1789619654; x=1790224454; h=content-transfer-encoding:content-type:cc:to:from:subject :message-id:references:mime-version:in-reply-to:date :x-gm-message-state:from:to:cc:subject:date:message-id:reply-to :content-type; bh=PUi0w321u1M0bPfHbTN2MqImPn3RRghNR/0evAaIXTo=; b=tqWqZKnbA7GcF1DOYiSNxFn6bNQqEuS3xY9ZUqiNbNd5g/Af2xL/z2vDAkgDpfrMtd Bh/u+E57JSDLM5dN1WQLyChWWhDW3pdKCs62YXcrIwN6P0o+ZrAFvxljg8ws1ng1TVjR KHyIHaKNcQuXq+ZoT3O8+QCpqEyQKCeuPaGyR3X2acrSDTf3oMgij+4u3FBz5iufKJbY Qd6RBxvhy9TPbJUQpJ62lABsgb0wQVomUpEuMRBF0Eak/9TtT1tP+5mY9G/mQ9wiwWn1 h0sTxB5dyWzv4moIrLbqzT4TFxThZTCEZfkC62gLtdzA6f1GbdbNFV1lCJ3LQNg+uHtZ z/1w== X-Gm-Message-State: AFuF++lsoJA1xbQS4KyqCgUFHsseKyoXruLI4mYjkTlpHgYS6dD3ND2I gw6kg604T1OMjKvKLIoEL/UZs2jb0RkLec2EilBqGwufZ+bXLVBwYwCUQvviavqJpvJehhBjLm+ N3LWSlYS/5f9Gvl1bwvkeGWDUvW/zdbgeIWZyvP484GZINIp3UPabkM/WGeLlAf0x1Vkedh80CN hMs+uIL7/kzmHkT1Ui8eQ4FLpNVbL7K0FVcVwK1kbNqLX+NlgzCddRS6Y= X-Received: from pfnp14.prod.google.com ([2002:aa7:860e:0:b0:873:3b26:abee]) (user=suleiman job=prod-delivery.src-stubby-dispatcher) by 2002:a05:6a00:3a02:b0:86c:a9a8:844f with SMTP id d2e1a72fcca58-872363f6650mr10686252b3a.1.1789619650038; Wed, 16 Sep 2026 21:34:10 -0700 (PDT) Date: Thu, 17 Sep 2026 04:33:36 +0000 In-Reply-To: <20260917043339.2093426-1-suleiman@google.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260917043339.2093426-1-suleiman@google.com> X-Mailer: git-send-email 2.55.0.1082.g2b9226bbc0-goog Message-ID: <20260917043339.2093426-13-suleiman@google.com> Subject: [RFC PATCH 12/12] tools/testing/futex: Add ping_bench, a tool for benchmarking futexes. From: Suleiman Souhlal To: linux-kernel@vger.kernel.org Cc: Suleiman Souhlal , Thomas Gleixner , Ingo Molnar , Peter Zijlstra , Darren Hart , Davidlohr Bueso , "=?UTF-8?q?Andr=C3=A9=20Almeida?=" , Juri Lelli , Vincent Guittot , Dietmar Eggemann , Steven Rostedt , Ben Segall , Mel Gorman , Valentin Schneider , K Prateek Nayak , zhidao su , John Stultz , Qais Yousef , ssouhlal@FreeBSD.org Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" It can generate metrics across a number of configurations, measuring how long locking takes. It prints out lock call durations for the foreground thread to stdout (unless -q is passed), for analysis with other tools such as ministat or turning into histograms. Some parameters: -a: Print out durations for all threads instead of only for the foreground thread. -q: Don't print out locking durations. -f: Type of futex ("f" FUTEX_WAIT, "p" FUTEX_LOCK_PI, "n" FUTEX_LOCK_PING, "N" FUTEX_LOCK_PING with userspace stealing, "m" pthread_mutex_t). -t: Number of threads acquiring/releasing the lock. -n: Number of iterations the threads will take the lock. -s: How frequently a thread tries to take the lock. -S: Duration in usec to sleep instead of spinning after unlocking (use instead of -s). -w: Lock hold time. -b: Number of =E2=80=9Cbusy=E2=80=9D cpu spinner threads. -r: Number of threads using RT prio. -p: Use nice() biasing (foreground thread gets more cpu time, background threads get less - allows for priority inversions). Co-developed-by: John Stultz Signed-off-by: John Stultz Signed-off-by: Suleiman Souhlal --- tools/testing/futex/Makefile | 13 + tools/testing/futex/ping_bench.c | 428 +++++++++++++++++++++++++++++++ 2 files changed, 441 insertions(+) create mode 100644 tools/testing/futex/Makefile create mode 100644 tools/testing/futex/ping_bench.c diff --git a/tools/testing/futex/Makefile b/tools/testing/futex/Makefile new file mode 100644 index 000000000000..cbb2deb30923 --- /dev/null +++ b/tools/testing/futex/Makefile @@ -0,0 +1,13 @@ +# SPDX-License-Identifier: GPL-2.0 + +.PHONY: clean + +TARGETS =3D ping_bench +CFLAGS =3D -O -Wall -g +OFILES =3D ping_bench.o +TARGETS =3D ping_bench + +ping_bench: $(OFILES) + +clean: + $(RM) $(TARGETS) $(OFILES) diff --git a/tools/testing/futex/ping_bench.c b/tools/testing/futex/ping_be= nch.c new file mode 100644 index 000000000000..418e147174d0 --- /dev/null +++ b/tools/testing/futex/ping_bench.c @@ -0,0 +1,428 @@ +// SPDX-License-Identifier: GPL-2.0-only + +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#define MAX_THR 512 + +#define FUTEX_LOCK_PING 14 +#define FUTEX_UNLOCK_PING 15 + +#define READ_ONCE(x) (*(volatile typeof(x) *)&(x)) + +struct thread { + pthread_t pthr; + uint64_t *dur; + int id; +}; + +static struct thread bthr[MAX_THR]; +static struct thread thr[MAX_THR]; +static pthread_barrier_t bar; + +uint32_t _lock; +pthread_mutex_t mtx =3D PTHREAD_MUTEX_INITIALIZER; +void *lock =3D &_lock; + +static long counter; +static int num_thr; +static int num_rt; +static int busy_thr; +static uint64_t num_iter =3D 10000; +static __thread pid_t tid; +static int work_times =3D 100000; +static uint64_t sleep_dur =3D 1000; +static bool do_sleep; +static bool print_all; +static bool quiet; +static bool bias; + +enum futex_type { + FUTEX, + FUTEX_PI, + FUTEX_PING, + FUTEX_PING_USTEAL, + MUTEX, +}; +static enum futex_type futex_type =3D FUTEX_PING; + +extern char *optarg; +extern int optind; + +int trace_marker_fd; + +static void +init_trace_marker(void) +{ + trace_marker_fd =3D open("/sys/kernel/tracing/trace_marker", O_WRONLY); + if (trace_marker_fd < 0) + perror("Failed to open trace_marker"); +} + +static int +write_trace_marker(const char *format, ...) +{ + char buffer[256]; + va_list args; + int len; + + if (trace_marker_fd <=3D 0) + return -1; + + va_start(args, format); + len =3D vsnprintf(buffer, sizeof(buffer), format, args); + va_end(args); + + if (len > 0) + write(trace_marker_fd, buffer, len); + + return (0); +} + +static inline uint64_t +now_ns(void) +{ + struct timespec ts; + + if (clock_gettime(CLOCK_MONOTONIC, &ts) !=3D 0) + err(1, "clock_gettime"); + + return (ts.tv_sec * 1000000000UL + ts.tv_nsec); +} + +static int +futex(uint32_t *uaddr, int op, uint32_t val, struct timespec *to) +{ + return (syscall(SYS_futex, uaddr, op, val, to)); +} + +static void +futex_lock(void *lock) +{ + pthread_mutex_t *mu; + uint32_t *fu, old; + int ret; + + fu =3D lock; + mu =3D lock; + switch (futex_type) { + case FUTEX: + while (1) { + old =3D 0; + if (atomic_compare_exchange_strong(fu, &old, 1)) + return; + if (futex(fu, FUTEX_WAIT, 1, NULL) !=3D 0 && errno !=3D + EAGAIN) + err(1, "FUTEX_WAIT"); + } + break; + case FUTEX_PI: + old =3D 0; + if (atomic_compare_exchange_strong(fu, &old, tid)) + return; + if ((ret =3D futex(fu, FUTEX_LOCK_PI, 0, NULL)) !=3D 0) + errx(1, "FUTEX_LOCK_PI %s", strerror(ret)); + break; + case FUTEX_PING: + case FUTEX_PING_USTEAL: + old =3D 0; + if (atomic_compare_exchange_strong(fu, &old, tid)) + return; + old =3D FUTEX_WAITERS; + if (futex_type =3D=3D FUTEX_PING_USTEAL && + atomic_compare_exchange_strong(fu, &old, tid | + FUTEX_WAITERS)) + return; + if ((ret =3D futex(fu, FUTEX_LOCK_PING, 0, NULL)) !=3D 0) + err(1, "FUTEX_LOCK_PING"); + break; + case MUTEX: + if (pthread_mutex_lock(mu) < 0) + err(1, "pthread_mutex_lock"); + break; + } +} + +static void +futex_unlock(void *lock) +{ + pthread_mutex_t *mu; + uint32_t *fu, old; + + fu =3D lock; + mu =3D lock; + switch (futex_type) { + case FUTEX: + old =3D 1; + if (atomic_compare_exchange_strong(fu, &old, 0)) + if (futex(fu, FUTEX_WAKE, 1, NULL) < 0) + err(1, "FUTEX_WAKE"); + break; + case FUTEX_PI: + old =3D tid; + if (atomic_compare_exchange_strong(fu, &old, 0)) + return; + if (futex(fu, FUTEX_UNLOCK_PI, 0, NULL) !=3D 0) + err(1, "FUTEX_UNLOCK_PI t %x old %x", tid, + READ_ONCE(*fu)); + break; + case FUTEX_PING: + case FUTEX_PING_USTEAL: + old =3D tid; + if (atomic_compare_exchange_strong(fu, &old, 0)) + return; + if (futex(fu, FUTEX_UNLOCK_PING, 0, NULL) !=3D 0) + err(1, "FUTEX_UNLOCK_PING t %x old %x", tid, + READ_ONCE(*fu)); + break; + case MUTEX: + if (pthread_mutex_unlock(mu) < 0) + err(1, "pthread_mutex_unlock"); + break; + } +} + +atomic_int stop_spinners =3D 0; + +static void * +func(void *p) +{ + struct thread *thr; + uint64_t end, start; + uint64_t i, j, _num_iter; + uint64_t my_sleep_dur =3D sleep_dur; + uint64_t my_work_times =3D work_times; + struct timespec ts; + + tid =3D gettid(); + + thr =3D p; + thr->dur =3D malloc(num_iter * sizeof(uint64_t)); + _num_iter =3D num_iter; + + if (thr->id =3D=3D 0) { + prctl(PR_SET_NAME, "foreground", 0, 0, 0); + if (bias) { + nice(-5); + my_sleep_dur *=3D 10; + if (my_work_times) + my_work_times /=3D 10; + } + } else { + prctl(PR_SET_NAME, "background", 0, 0, 0); + if (bias) { + if (my_sleep_dur) + my_sleep_dur /=3D 10; + my_work_times *=3D 10; + nice(19); + } + } + + ts.tv_sec =3D my_sleep_dur / 1000000; + ts.tv_nsec =3D (my_sleep_dur % 1000000) * 1000; + + + if (thr->id < num_rt) { + struct sched_param param =3D { .sched_priority =3D 10 }; + + if (sched_setscheduler(0, SCHED_FIFO, ¶m) !=3D 0) + err(1, "sched_setscheduler"); + } + + pthread_barrier_wait(&bar); + + for (i =3D 0; i < _num_iter; i++) { + if (thr->id =3D=3D 0) + write_trace_marker("B|%lu|Locking", (unsigned long)tid); + + start =3D now_ns(); + futex_lock(lock); + end =3D now_ns(); + + if (thr->id =3D=3D 0) + write_trace_marker("E|%lu|Locking", (unsigned long)tid); + thr->dur[i] =3D end - start; + + for (j =3D 0; j < my_work_times; j++) + __asm __volatile("" ::: "memory"); + + counter++; + futex_unlock(lock); + + if (atomic_load(&stop_spinners)) + break; + + if (sleep_dur) { + if (do_sleep) + clock_nanosleep(CLOCK_MONOTONIC, 0, &ts, 0); + else + for (j =3D 0; j < my_sleep_dur; j++) + __asm __volatile("" ::: "memory"); + } + } + + if (thr->id =3D=3D 0) + atomic_store(&stop_spinners, 1); + return (NULL); +} + +static void * +busy(void *p) +{ + prctl(PR_SET_NAME, "spinner", 0, 0, 0); + if (0 && num_rt) { + struct sched_param param =3D { .sched_priority =3D 1 }; + + if (sched_setscheduler(0, SCHED_FIFO, ¶m) !=3D 0) + err(1, "sched_setscheduler"); + } + + pthread_barrier_wait(&bar); + + while (!atomic_load(&stop_spinners)) + __asm __volatile("" ::: "memory"); + + return (NULL); +} + + +static void +usage(char *a) +{ + fprintf(stderr, "Usage: %s [-a] [-b n] [-c] [-f f/m/n/N/p] [-n n] [-q]" + " [-r n] [-s n] [-S n] [-t n] [-w n]\n", a); + exit(1); +} + +int +main(int argc, char **argv) +{ + int c, i, j, ret; + + while ((c =3D getopt(argc, argv, "ab:f:n:pqr:S:s:t:w:")) !=3D -1) { + switch (c) { + case 'a': + print_all =3D 1; + break; + case 'f': + switch (*optarg) { + case 'f': + futex_type =3D FUTEX; + break; + case 'm': + futex_type =3D MUTEX; + lock =3D &mtx; + break; + case 'n': + futex_type =3D FUTEX_PING; + break; + case 'N': + futex_type =3D FUTEX_PING_USTEAL; + break; + case 'p': + futex_type =3D FUTEX_PI; + break; + default: + usage(argv[0]); + } + break; + case 'n': + num_iter =3D atoi(optarg); + break; + case 'q': + quiet =3D 1; + break; + case 'p': + bias =3D 1; + break; + case 'r': + num_rt =3D atoi(optarg); + break; + case 'S': + do_sleep =3D 1; + /* Fallthrough */ + case 's': + sleep_dur =3D atoi(optarg); + break; + case 't': + num_thr =3D atoi(optarg); + break; + case 'b': + busy_thr =3D atoi(optarg); + break; + case 'w': + work_times =3D atoi(optarg); + break; + default: + usage(argv[0]); + } + } + + init_trace_marker(); + + if (num_thr < num_rt) + num_thr =3D num_rt; + + num_thr -=3D num_rt; + if (num_thr > MAX_THR) + num_thr =3D MAX_THR; + + tid =3D gettid(); + + ret =3D pthread_barrier_init(&bar, NULL, num_thr + num_rt + busy_thr); + if (ret !=3D 0) + errx(1, "pthread_barrier_init %s", strerror(ret)); + + for (i =3D 0; i < num_thr + num_rt; i++) { + thr[i].id =3D i; + if ((ret =3D pthread_create(&thr[i].pthr, NULL, func, &thr[i])) + !=3D 0) + errx(1, "pthread_create %d: %s", i, strerror(ret)); + } +=09 + for (i =3D 0; i < busy_thr; i++) { + if ((ret =3D pthread_create(&bthr[i].pthr, NULL, busy, &bthr[i])) + !=3D 0) + errx(1, "pthread_create %d: %s", i, strerror(ret)); + } + + for (i =3D 0; i < busy_thr; i++) + if ((ret =3D pthread_join(bthr[i].pthr, NULL)) !=3D 0) + errx(1, "pthread_join: %s\n", strerror(ret)); + + for (i =3D 0; i < num_thr + num_rt; i++) + if ((ret =3D pthread_join(thr[i].pthr, NULL)) !=3D 0) + errx(1, "pthread_join: %s\n", strerror(ret)); + + if (!quiet) { + /* skip the first run */ + if (print_all) + for (i =3D 0; i < num_thr; i++) + for (j =3D 1; j < num_iter; j++) + printf("%lu\n", thr[0].dur[j]); + else + for (j =3D 1; j < num_iter; j++) + printf("%lu\n", thr[0].dur[j]); + } + + return (0); +} --=20 2.55.0.1082.g2b9226bbc0-goog