From nobody Fri Oct 2 09:22:29 2026 Received: from mailtransmit05.runbox.com (mailtransmit05.runbox.com [185.226.149.38]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id B23A43AC0CB; Mon, 3 Aug 2026 09:01:51 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=185.226.149.38 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785747717; cv=none; b=hlYlKfsFblmS6G8j2Pkt2f3Hs7E/0vx1fKzh1zs3vVZ2BHWFl+FZ6rjjx69MXQnmQx4OQEMt8M7brIEBqI9qlGTpdO9Disnrw4WhlQAsg4xRS+BFVDOeugejb4vFIbe1B6YRp7YOrt6yVWYtctshXT6CiT5drTwghBQpf9PmscA= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785747717; c=relaxed/simple; bh=Z9oJOKfNkMISTXBIhGkL4Euhy5QznNcG5muaM0JGi2c=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=ultD8gvT4c/GboEb5E5NMbrvBjMN6aX3orrhzrf10PtwHWQ/B3Myc+cO9fhfrkEen+iYnIqJXfDFw2DSfJdWbGpMoGJcrhSqSBjfZACa0wo/sQgwqoRsi0iYeppRo4gM8P3VfN/G6nHGpA4PX2oh4YyBst7l2VVCdY3to7zdgs4= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=rbox.co; spf=pass smtp.mailfrom=rbox.co; dkim=pass (2048-bit key) header.d=rbox.co header.i=@rbox.co header.b=gTwEnd3z; arc=none smtp.client-ip=185.226.149.38 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=rbox.co Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=rbox.co Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=rbox.co header.i=@rbox.co header.b="gTwEnd3z" Received: from mailtransmit02.runbox ([10.9.9.162] helo=aibo.runbox.com) by mailtransmit05.runbox.com with esmtps (TLS1.2) tls TLS_ECDHE_RSA_WITH_AES_128_GCM_SHA256 (Exim 4.93) (envelope-from ) id 1wqoYB-003g8t-2e; Mon, 03 Aug 2026 11:01:35 +0200 DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=rbox.co; s=selector2; h=Cc:To:In-Reply-To:References:Message-Id: Content-Transfer-Encoding:Content-Type:MIME-Version:Subject:Date:From; bh=pfSMmgoKMBdfU0Z5WR10/bxb8fMdxnB7dgEW4lMNPXo=; b=gTwEnd3zDCZo3WCrhJdOL9TN5m pJobiDQ8x+5Hsh616A4UmOEt1T+G3+EO2KnBNIIvdVABZGThGuxBXvpqyNGldLoy7m158gq8rBedd aDaCPrLZBp+C5J2mShAZz36sLCnXkL4ekA7BRT7MapealYMIJ4Ug6VQ/JVB/MFMkyPQsib+opqwIi /Vkogq2wK96uL70B6vsVLhJlW0JK8v0Gmo9IwXy0XA+DD+F9h1ISuFcrq3FWOK2Nhu1W4mBJXXfMT +hxlmuR3jIkeykOA66LO1AqwbmfZ+Kw1KzWAPtcx9mQUmlE+sSxLTBpXozZ+/1bIPXxUZzVDWy1Vl 9/W6Spyw==; Received: from [10.9.9.74] (helo=submission03.runbox) by mailtransmit02.runbox with esmtp (Exim 4.86_2) (envelope-from ) id 1wqoY9-0006d2-If; Mon, 03 Aug 2026 11:01:33 +0200 Received: by submission03.runbox with esmtpsa [Authenticated ID (604044)] (TLS1.2:ECDHE_SECP256R1__RSA_PSS_RSAE_SHA256__AES_256_GCM:256) (Exim 4.95) id 1wqoXs-00GDCj-KX; Mon, 03 Aug 2026 11:01:17 +0200 From: Michal Luczaj Date: Mon, 03 Aug 2026 11:00:42 +0200 Subject: [PATCH bpf v2 1/2] bpf: Extract shared reqsk-to-listener upgrade Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260803-sockmap-lookup-tcp-leak-v2-1-306e025bfe66@rbox.co> References: <20260803-sockmap-lookup-tcp-leak-v2-0-306e025bfe66@rbox.co> In-Reply-To: <20260803-sockmap-lookup-tcp-leak-v2-0-306e025bfe66@rbox.co> To: Alexei Starovoitov , Daniel Borkmann , Andrii Nakryiko , Eduard Zingerman , Kumar Kartikeya Dwivedi , Martin KaFai Lau , Song Liu , Yonghong Song , Jiri Olsa , Emil Tsalapatis , John Fastabend , Stanislav Fomichev , "David S. Miller" , Eric Dumazet , Jakub Kicinski , Paolo Abeni , Simon Horman , Kuniyuki Iwashima , Willem de Bruijn , Jakub Sitnicki , Jiayuan Chen , Joe Stringer Cc: Michal Luczaj , bpf@vger.kernel.org, netdev@vger.kernel.org, linux-kernel@vger.kernel.org X-Mailer: b4 0.15.2 __bpf_sk_lookup() and bpf_sk_lookup() duplicate the same sk_to_full_sk() reqsk-to-listener upgrade. Extract it into a helper. Leave the currently unreachable WARN_ONCE as a defensive assert. No functional change. Signed-off-by: Michal Luczaj Reviewed-by: Emil Tsalapatis Reviewed-by: Jakub Sitnicki --- net/core/filter.c | 58 +++++++++++++++++++++++++--------------------------= ---- 1 file changed, 26 insertions(+), 32 deletions(-) diff --git a/net/core/filter.c b/net/core/filter.c index 11bb0d236822..fede810ef37f 100644 --- a/net/core/filter.c +++ b/net/core/filter.c @@ -7079,6 +7079,28 @@ __bpf_skc_lookup(struct sk_buff *skb, struct bpf_soc= k_tuple *tuple, u32 len, return sk; } =20 +static struct sock * +bpf_sk_lookup_full_sk(struct sock *sk) +{ + struct sock *sk2 =3D sk_to_full_sk(sk); + + /* + * sk_to_full_sk() may return sk->rsk_listener, make sure the original + * sk sock refcnt is decremented to prevent a request_sock leak. + */ + if (sk2 !=3D sk) { + sock_gen_put(sk); + /* Ensure there is no need to bump sk2 refcnt. */ + if (unlikely(sk2 && !sock_flag(sk2, SOCK_RCU_FREE))) { + WARN_ONCE(1, "Found non-RCU, unreferenced socket!"); + return NULL; + } + sk =3D sk2; + } + + return sk; +} + static struct sock * __bpf_sk_lookup(struct sk_buff *skb, struct bpf_sock_tuple *tuple, u32 len, struct net *caller_net, u32 ifindex, u8 proto, u64 netns_id, @@ -7088,22 +7110,8 @@ __bpf_sk_lookup(struct sk_buff *skb, struct bpf_sock= _tuple *tuple, u32 len, ifindex, proto, netns_id, flags, sdif); =20 - if (sk) { - struct sock *sk2 =3D sk_to_full_sk(sk); - - /* sk_to_full_sk() may return (sk)->rsk_listener, so make sure the origi= nal sk - * sock refcnt is decremented to prevent a request_sock leak. - */ - if (sk2 !=3D sk) { - sock_gen_put(sk); - /* Ensure there is no need to bump sk2 refcnt */ - if (unlikely(sk2 && !sock_flag(sk2, SOCK_RCU_FREE))) { - WARN_ONCE(1, "Found non-RCU, unreferenced socket!"); - return NULL; - } - sk =3D sk2; - } - } + if (sk) + sk =3D bpf_sk_lookup_full_sk(sk); =20 return sk; } @@ -7134,22 +7142,8 @@ bpf_sk_lookup(struct sk_buff *skb, struct bpf_sock_t= uple *tuple, u32 len, struct sock *sk =3D bpf_skc_lookup(skb, tuple, len, proto, netns_id, flags); =20 - if (sk) { - struct sock *sk2 =3D sk_to_full_sk(sk); - - /* sk_to_full_sk() may return (sk)->rsk_listener, so make sure the origi= nal sk - * sock refcnt is decremented to prevent a request_sock leak. - */ - if (sk2 !=3D sk) { - sock_gen_put(sk); - /* Ensure there is no need to bump sk2 refcnt */ - if (unlikely(sk2 && !sock_flag(sk2, SOCK_RCU_FREE))) { - WARN_ONCE(1, "Found non-RCU, unreferenced socket!"); - return NULL; - } - sk =3D sk2; - } - } + if (sk) + sk =3D bpf_sk_lookup_full_sk(sk); =20 return sk; } --=20 2.55.0 From nobody Fri Oct 2 09:22:29 2026 Received: from mailtransmit04.runbox.com (mailtransmit04.runbox.com [185.226.149.37]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 87A6A3BBFBA; Mon, 3 Aug 2026 09:01:44 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=185.226.149.37 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785747709; cv=none; b=cHct2Z1SmZhEm4qmDxFr5le26gdgtoL+JF2FrpJl37H0Qvh8m2RyePjdiZ5h8XQB2eqfoK/Tx4Ty83honkWz5u4tcy2PpDwIfWfHnoFDc9p6pHwIaTz+SxHUjXboVMVczy/e6lWlphppeFz9i5+8hBewdJiQuehk+9dsI0mTdb0= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785747709; c=relaxed/simple; bh=52iuD3i7TRN4nBU/20dqpqs3iGShQDt+XaFa/feofZ4=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=ME6N299bro/pk1AjjP5PAPZk6Jjn9xzA+joqRjjvTNQyNKxSnYK39PgeEDCGU3HVjs8OdCvT1sajDLXwIKoN0aepEMbRyi11fUIQa/qiEpM5DEbleH0ZIxWPDL9WmDLWZRn11EHigNSFMGYpsS1z+Pw8FQ+90cXY1eTgYfJucus= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=rbox.co; spf=pass smtp.mailfrom=rbox.co; dkim=pass (2048-bit key) header.d=rbox.co header.i=@rbox.co header.b=KS4U5oJw; arc=none smtp.client-ip=185.226.149.37 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=rbox.co Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=rbox.co Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=rbox.co header.i=@rbox.co header.b="KS4U5oJw" Received: from mailtransmit03.runbox ([10.9.9.163] helo=aibo.runbox.com) by mailtransmit04.runbox.com with esmtps (TLS1.2) tls TLS_ECDHE_RSA_WITH_AES_128_GCM_SHA256 (Exim 4.93) (envelope-from ) id 1wqoYE-001fLN-B8; Mon, 03 Aug 2026 11:01:38 +0200 DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=rbox.co; s=selector2; h=Cc:To:In-Reply-To:References:Message-Id: Content-Transfer-Encoding:Content-Type:MIME-Version:Subject:Date:From; bh=gEL7vrretHDhWO9eEJSfhdss43QrZ/6YR3FxQETnLjY=; b=KS4U5oJw5WzBNCffDf/Y7782E3 8J87X9/mfUQNzfgZflruji669E8nBv0WIpbk4J1i/rx+CTTJwTbTOrmqd+g56xTk6NXYnrUyHek5k 8giywGWK6lxKEwsXwN0ZNljeYoiChrquTu9F4/X6WKfXhXqCnQqV94e7XDaFoPyXuQbVQKE4BNcE2 KY8Tc3X4n8/BqraDcPVUMUGZfammxjtlrRVdqmUIyEIogNamTeXIGknRhEQPQmgm7RAEsAG7z1mju 4z3uN03KyFH5XWbIUDAAvi+YlwqmU6SzzxuoDkNt/9YXx9q4VV3rWVP4EHVVkKuN7gRUzh3aQbcqO 8inBhKrA==; Received: from [10.9.9.74] (helo=submission03.runbox) by mailtransmit03.runbox with esmtp (Exim 4.86_2) (envelope-from ) id 1wqoYD-0003xn-Pd; Mon, 03 Aug 2026 11:01:38 +0200 Received: by submission03.runbox with esmtpsa [Authenticated ID (604044)] (TLS1.2:ECDHE_SECP256R1__RSA_PSS_RSAE_SHA256__AES_256_GCM:256) (Exim 4.95) id 1wqoXu-00GDCj-CY; Mon, 03 Aug 2026 11:01:18 +0200 From: Michal Luczaj Date: Mon, 03 Aug 2026 11:00:43 +0200 Subject: [PATCH bpf v2 2/2] bpf: Unconditionally take socket references in lookup helpers Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260803-sockmap-lookup-tcp-leak-v2-2-306e025bfe66@rbox.co> References: <20260803-sockmap-lookup-tcp-leak-v2-0-306e025bfe66@rbox.co> In-Reply-To: <20260803-sockmap-lookup-tcp-leak-v2-0-306e025bfe66@rbox.co> To: Alexei Starovoitov , Daniel Borkmann , Andrii Nakryiko , Eduard Zingerman , Kumar Kartikeya Dwivedi , Martin KaFai Lau , Song Liu , Yonghong Song , Jiri Olsa , Emil Tsalapatis , John Fastabend , Stanislav Fomichev , "David S. Miller" , Eric Dumazet , Jakub Kicinski , Paolo Abeni , Simon Horman , Kuniyuki Iwashima , Willem de Bruijn , Jakub Sitnicki , Jiayuan Chen , Joe Stringer Cc: Michal Luczaj , bpf@vger.kernel.org, netdev@vger.kernel.org, linux-kernel@vger.kernel.org, Sashiko X-Mailer: b4 0.15.2 Lookup helpers gate whether to acquire a socket reference on sk_is_refcounted(), a check re-evaluated at release. An established socket refcounted at acquire time can gain SOCK_RCU_FREE via connect(AF_UNSPEC)+listen() before release runs; the release-side re-check then reads sk_is_refcounted() =3D=3D false and skips the put. The reference leaks. Make acquire and release unconditional and symmetric: always take a reference, always put it. Adapt sk_select_reuseport(). Fixes: 6acc9b432e67 ("bpf: Add helper to retrieve socket in BPF") Fixes: 64d85290d79c ("bpf: Allow bpf_map_lookup_elem for SOCKMAP and SOCKHA= SH") Reported-by: Sashiko Closes: https://lore.kernel.org/bpf/20260701235552.2B0AA1F00A3F@smtp.kernel= .org/ Signed-off-by: Michal Luczaj Reviewed-by: Emil Tsalapatis --- TC bpf_sk_assign() has the same issue; it takes a reference only when sk_is_refcounted() is true at assign time, but sock_pfree() (the skb destructor it installs) re-checks sk_is_refcounted() independently at release time. The same connect(AF_UNSPEC)+listen() transition leaks the socket here too. I'd welcome suggestions on the right way to handle this. --- net/core/filter.c | 27 ++++++++++++++++++--------- net/core/sock_map.c | 4 ++-- 2 files changed, 20 insertions(+), 11 deletions(-) diff --git a/net/core/filter.c b/net/core/filter.c index fede810ef37f..d71e069f669a 100644 --- a/net/core/filter.c +++ b/net/core/filter.c @@ -7032,6 +7032,14 @@ static struct sock *sk_lookup(struct net *net, struc= t bpf_sock_tuple *tuple, WARN_ONCE(1, "Found non-RCU, unreferenced socket!"); sk =3D NULL; } + + /* + * Always take a reference, even if the lookup skipped one; + * bpf_sk_release() always puts one. + */ + if (sk && !refcounted && !refcount_inc_not_zero(&sk->sk_refcnt)) + sk =3D NULL; + return sk; } =20 @@ -7090,11 +7098,16 @@ bpf_sk_lookup_full_sk(struct sock *sk) */ if (sk2 !=3D sk) { sock_gen_put(sk); - /* Ensure there is no need to bump sk2 refcnt. */ if (unlikely(sk2 && !sock_flag(sk2, SOCK_RCU_FREE))) { WARN_ONCE(1, "Found non-RCU, unreferenced socket!"); return NULL; } + /* + * sk2 is RCU-free, but take a reference anyway; + * bpf_sk_release() puts. + */ + if (sk2 && !refcount_inc_not_zero(&sk2->sk_refcnt)) + sk2 =3D NULL; sk =3D sk2; } =20 @@ -7279,7 +7292,7 @@ static const struct bpf_func_proto bpf_tc_sk_lookup_u= dp_proto =3D { =20 BPF_CALL_1(bpf_sk_release, struct sock *, sk) { - if (sk && sk_is_refcounted(sk)) + if (sk) sock_gen_put(sk); return 0; } @@ -11571,11 +11584,13 @@ BPF_CALL_4(sk_select_reuseport, struct sk_reusepo= rt_kern *, reuse_kern, bool is_sockarray =3D map->map_type =3D=3D BPF_MAP_TYPE_REUSEPORT_SOCKARR= AY; struct sock_reuseport *reuse; struct sock *selected_sk; - int err; + int err =3D 0; =20 selected_sk =3D map->ops->map_lookup_elem(map, key); if (!selected_sk) return -ENOENT; + if (!is_sockarray) + sock_put(selected_sk); =20 reuse =3D rcu_dereference(selected_sk->sk_reuseport_cb); if (!reuse) { @@ -11605,13 +11620,7 @@ BPF_CALL_4(sk_select_reuseport, struct sk_reusepor= t_kern *, reuse_kern, } =20 reuse_kern->selected_sk =3D selected_sk; - - return 0; error: - /* Lookup in sock_map can return TCP ESTABLISHED sockets. */ - if (sk_is_refcounted(selected_sk)) - sock_put(selected_sk); - return err; } =20 diff --git a/net/core/sock_map.c b/net/core/sock_map.c index 9efbd8ca7db8..92a006fd3368 100644 --- a/net/core/sock_map.c +++ b/net/core/sock_map.c @@ -392,7 +392,7 @@ static void *sock_map_lookup(struct bpf_map *map, void = *key) sk =3D __sock_map_lookup_elem(map, *(u32 *)key); if (!sk) return NULL; - if (sk_is_refcounted(sk) && !refcount_inc_not_zero(&sk->sk_refcnt)) + if (!refcount_inc_not_zero(&sk->sk_refcnt)) return NULL; return sk; } @@ -1218,7 +1218,7 @@ static void *sock_hash_lookup(struct bpf_map *map, vo= id *key) sk =3D __sock_hash_lookup_elem(map, key); if (!sk) return NULL; - if (sk_is_refcounted(sk) && !refcount_inc_not_zero(&sk->sk_refcnt)) + if (!refcount_inc_not_zero(&sk->sk_refcnt)) return NULL; return sk; } --=20 2.55.0