From nobody Fri Jul 24 22:17:37 2026 Received: from mail-wm1-f71.google.com (mail-wm1-f71.google.com [209.85.128.71]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 061D82FD1B3 for ; Wed, 22 Jul 2026 21:09:31 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.85.128.71 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784754573; cv=none; b=RWoEdfMv+IqqD/VvvpkMZvdMlmvE6o4sdWB041lzU4dK3NkzbtdPDqTusf32IC9rKp8QfegsdzMZ30JQqSKJTppRqKi0/80JYV58mEr2AC3OB3RZ9z6DpO0ranDUP4JUjyjpixX8njVVqLmTQix5hdmaYzjikfX3SaJh1Bsh35M= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784754573; c=relaxed/simple; bh=t5RwgUTaH5M9kRWyGURkC8HeC0l9B/s9piR2QGnoAy4=; h=Date:Mime-Version:Message-ID:Subject:From:To:Cc:Content-Type; b=dTaSg0soTBScoYxWxHTLbxbe1gALQKnyUT5k74094ltB5BihQPVSY+ExgUgtHExoGHWP55GemmDelt4kqrBadiWUXBqf21WWIGIBZ5glPpcLl8FU3K2FFMZUgmdHQSTAN7eacw4u922Lwjq+1HWBheUCfBJ4VC929PfAG6XpiSc= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com; spf=pass smtp.mailfrom=flex--aliceryhl.bounces.google.com; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b=I0RVAVEU; arc=none smtp.client-ip=209.85.128.71 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=reject dis=none) header.from=google.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=flex--aliceryhl.bounces.google.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=google.com header.i=@google.com header.b="I0RVAVEU" Received: by mail-wm1-f71.google.com with SMTP id 5b1f17b1804b1-49561facb1dso27798205e9.3 for ; Wed, 22 Jul 2026 14:09:31 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=google.com; s=20251104; t=1784754570; x=1785359370; darn=vger.kernel.org; h=content-type:cc:to:from:subject:message-id:mime-version:date:from :to:cc:subject:date:message-id:reply-to:content-type; bh=nNQEV2stmZS3ZSr7vZKBKxIMqHGZ/9vvFA0yE3lwS2k=; b=I0RVAVEUZLzdTBNxFysoe5yb3XjEwvLXRBWe8qDmjHAkcozhjKNrWKUwOqXcDTbQAO GeFkGxmTuokkMEXglYVqdhn38syyq7di7+AqcUN+ZaKZl7u77lIY03xgtSmfH91mQ9Y6 Q9dG+xxCib3OnFLB8d7DEJetQ2pqMa5kr9tMjAn+IdHRiUuNT7HCQFdxSZqsMgHXm9cM fT+kgzXVEHSnNApbl0M8X0pHYc6qDT2ZIK7y9Mtj+ci8k0nIYvk/HVgS9IHUbMyG2XPb zebkUqNRAScO+90LXg0SdhhkyH54t200HnJ72bTOlUbHOU3TVnwYhmU4Fs/YfwqQkL0r o8cA== X-Google-DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=1e100.net; s=20251104; t=1784754570; x=1785359370; h=content-type:cc:to:from:subject:message-id:mime-version:date :x-gm-message-state:from:to:cc:subject:date:message-id:reply-to :content-type; bh=nNQEV2stmZS3ZSr7vZKBKxIMqHGZ/9vvFA0yE3lwS2k=; b=jroUTJyT6HXTMbMb73BQn77nOEPQg3LL3O1eZb2zcbQH6Y3QIkSwYNCssv15yO1uJ8 SqbiSfqogFJxZ+ijIXikAscP8FCiWNIDhLsvdQLo3aykemg8ABzvrFO7BWrNpe4t0qr6 GNgEpoXNfTluz0a2kTwWk5BcPucAIHkYdur7TaTyGL0ChW7EcnOCxgyUN1p9EaRKOwlg 18t5uWca4l5JrKlv0Q9Xzw9CKDUMdscusb/VbC3a8wvINbJZTTUY5k6zrUaAEZI1M4Qc J9Uu9h1t+w+hSXsv8UOBlBjgz4yeH/749b81FnF+HHRnoIcWlJWC2h9w/3TsmAsY4LoI GWXQ== X-Forwarded-Encrypted: i=1; AHgh+RpmMLf8grq/VayRkyzAlbISK/PtJN+om5lb5AxVu6dBXY2rhgdzSZVYBFjFj7UJnFVx24CgAq8kJsAzcfg=@vger.kernel.org X-Gm-Message-State: AOJu0YxlNCjhL5D5UvSBWf1Le53Nekk9Fd/jvKKdHssGFykKTuOUQaqS MLj+W6FM0SipEp1l0pK84mEuL7r7pXc/uU3BlSGYoO0hTLu9j9pYTmzXnNyL6D1iVbtT1914lwB UNsqCUShOknBO44RP0g== X-Received: from wmbjj16.prod.google.com ([2002:a05:600c:6a10:b0:495:5422:15f2]) (user=aliceryhl job=prod-delivery.src-stubby-dispatcher) by 2002:a7b:cb83:0:b0:495:6b55:f938 with SMTP id 5b1f17b1804b1-49573cc6d02mr2294815e9.10.1784754569961; Wed, 22 Jul 2026 14:09:29 -0700 (PDT) Date: Wed, 22 Jul 2026 21:09:22 +0000 Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 X-B4-Tracking: v=1; b=H4sIAIExYWoC/13MQQqDMBCF4avIrJuSidTarrxHcRGSFw1YI4lIi +TuTYVuuvyHed9OCdEj0b3aKWLzyYe5hDpVZEY9DxDeliYlVSOvfBEWDlGY8FwmrBCOLfRNO65 1S2W0RDj/OsBHX3r0aQ3xffgbf68/qvmnNhYsDCTq1tbSmqYbQhgmnMsH9TnnD5ohQh2tAAAA X-Change-Id: 20260715-defer-complete-f1dea9af13a8 X-Developer-Key: i=aliceryhl@google.com; a=openpgp; fpr=49F6C1FAA74960F43A5B86A1EE7A392FDE96209F X-Developer-Signature: v=1; a=openpgp-sha256; l=11530; i=aliceryhl@google.com; h=from:subject:message-id; bh=t5RwgUTaH5M9kRWyGURkC8HeC0l9B/s9piR2QGnoAy4=; b=owEBbQKS/ZANAwAKAQRYvu5YxjlGAcsmYgBqYTGE4zFUf7aPm0FHBsO7mI8c+9XLvGapkpu9r 4dWP571irOJAjMEAAEKAB0WIQSDkqKUTWQHCvFIvbIEWL7uWMY5RgUCamExhAAKCRAEWL7uWMY5 RrFDEACc+Ju1ZctpRLjgT1APtZckSMKu30tezbOzA1ioFm5MkQlW01nqz5tcOMI3t/pFdp+dEGl S8vDWwatVBaIyy618jI0Q3zW5qcu1OE+65QFkltezs7VBLqMa+2xHsdR65MfO92aD1kqqawQujC txI886x2400sbwL+pvAdUjfMRron21KB+FcnC+pYHM+O+SDxH064Oua/pTiW/ITDPz77gjVqpVI X5OHkuTRaMpQa/9HxyzhsQxia3FGDB/jeQEn0rhod4MGqxZjgPL1YIaHuv8RSwfVDSC776qvElg 8u6BxMDCT99speKd5LJ7pMo+q+bcBn7VYn5JQkjurgNNFeoqEV/QtBxwoYhxv3+B6XwRz54CX+w UBaQq/qvxJF7wApvKUsjaLRSHLpVov4PSME+iWG0ZpCkazMMWfDh2In3SlT3ARHK7lj9cgVQOmP G8MJKqoUjcXGSf7D5X0Lkk7B4ZLHf6JrzOCJ24Mpp7KUhOyrJEIAfn1CtjL/nsZEs/ekCv9+aP1 WZSnbhpzr0Y+Qz9lOPRmNCmCuZCHeOn5k47603/XQ55gphDqJHLzZ9F7fWcsgLo8d4hlzClPJU1 3agNNgUgXuA9UUnly9iTZRg4Lm3HSTS5sE+H8AKcEYgxbU+LIU9r0PMwKgYKX53KEUeMJl/WK7C plCa9HD4HgMbejw== X-Mailer: b4 0.14.3 Message-ID: <20260722-defer-complete-v2-1-6c67af0e2ac2@google.com> Subject: [PATCH v2] rust_binder: add TF_DEFER_COMPLETE flag for avoiding userspace roundtrip From: Alice Ryhl To: Greg Kroah-Hartman , Todd Kjos , Carlos Llamas Cc: Miguel Ojeda , Boqun Feng , Gary Guo , "=?utf-8?q?Bj=C3=B6rn_Roy_Baron?=" , Benno Lossin , Andreas Hindborg , Trevor Gross , Danilo Krummrich , Daniel Almeida , Tamir Duberstein , Alexandre Courbot , "=?utf-8?q?Onur_=C3=96zkan?=" , rust-for-linux@vger.kernel.org, linux-kernel@vger.kernel.org, Alice Ryhl Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Outgoing transactions are able to send a message and wait for its reply in a single ioctl. Why not avoid a userspace roundtrip by applying the same logic for replying to incoming messages and waiting for the next incoming message? Generally, when you send a reply using BC_REPLY, the kernel sends BR_TRANSACTION_COMPLETE as a reply to BC_REPLY right away. The BR_TRANSACTION_COMPLETE command indicates that it's safe for userspace to free any resources associated with this message (such as embedded fds or Binder nodes). However, the BR_TRANSACTION_COMPLETE message is problematic because after BC_REPLY is issued, there will be a pending message for userspace. The kernel will refuse to sleep for incoming messages in this scenario. The way this is handled for outgoing transaction is through a mechanism known as deferred delivery of BR_TRANSACTION_COMPLETE. The idea is that when you send an outgoing transaction, then we do not return to userspace right away if BR_TRANSACTION_COMPLETE is the only pending message. This patch adds a new flag called TF_DEFER_COMPLETE that lets userspace opt-in to the same deferred delivery mechanism for BR_TRANSACTION_COMPLETE when using BC_REPLY. Given this new uapi, we can adjust sendReply in userspace libbinder so that it writes the BC_REPLY command into mOut but does not flush the buffer to the kernel. Then, userspace simply continues running until it returns all the way out to the top-level joinThreadPool() loop, which calls into the kernel to get the next incoming transaction. At this point, mOut is flushed, sending the reply. The same ioctl then proceeds to sleep for an incoming message. Userspace only actually specifies TF_DEFER_COMPLETE when the Parcel does not contain fds or refcounts on binder objects. This is because otherwise said fd or binder node will not be freed until the binder thread receives another incoming transaction, which could be a long time. In the case of fds, this is especially important because delaying fclose() can result in processes hanging because they read from a pipe that isn't being closed due to fclose() not getting called. Note that even if TF_DEFER_COMPLETE is not specified for this transaction, it can still be useful to defer the BC_REPLY command, as it can still avoid a userspace roundtrip when a new incoming transaction is available right away. Observing the cuttlefish logs while booting with this change shows that there were 4297 opportunities for this optimization to kick in (that is, boot invoked BC_REPLY 4297 times). Out of those, 3441 binder ioctls sent and received a transaction in the same ioctl. This indicates that we successfully eliminated a syscall on the server side for 80% of incoming transactions. Generally, this means that a server is now able to handle incoming messages using one syscall per incoming message (for each incoming transaction, the syscall handles one BC_FREE_BUFFER and BC_REPLY command, and then waits for the next incoming transaction). Signed-off-by: Alice Ryhl --- Changes in v2: - Return deferred thread work instead of the process global work if there is work in the process global list. - Link to v1: https://lore.kernel.org/r/20260716-defer-complete-v1-1-ce0e38= d30dc6@google.com --- drivers/android/binder/defs.rs | 3 +- drivers/android/binder/process.rs | 8 ++++++ drivers/android/binder/thread.rs | 55 +++++++++++++++++++++++++++++++--= ---- include/uapi/linux/android/binder.h | 1 + 4 files changed, 57 insertions(+), 10 deletions(-) diff --git a/drivers/android/binder/defs.rs b/drivers/android/binder/defs.rs index 8ac9bdd7a499..cc4becd6e168 100644 --- a/drivers/android/binder/defs.rs +++ b/drivers/android/binder/defs.rs @@ -77,7 +77,8 @@ macro_rules! pub_no_prefix { TF_ONE_WAY, TF_ACCEPT_FDS, TF_CLEAR_BUF, - TF_UPDATE_TXN + TF_UPDATE_TXN, + TF_DEFER_COMPLETE, ); =20 pub(crate) use uapi::{ diff --git a/drivers/android/binder/process.rs b/drivers/android/binder/pro= cess.rs index 1778628d8acd..4f23a7cf7352 100644 --- a/drivers/android/binder/process.rs +++ b/drivers/android/binder/process.rs @@ -686,8 +686,16 @@ pub(crate) fn get_work(&self) -> Option> { pub(crate) fn get_work_or_register<'a>( &'a self, thread: &'a Arc, + thread_has_deferred_work: bool, ) -> GetWorkOrRegister<'a> { let mut inner =3D self.inner.lock(); + + if thread_has_deferred_work && !inner.work.is_empty() { + if let Some(work) =3D thread.pop_work_even_if_deferred() { + return GetWorkOrRegister::Work(work); + } + } + // Try to get work from the process queue. if let Some(work) =3D inner.work.pop_front() { return GetWorkOrRegister::Work(work); diff --git a/drivers/android/binder/thread.rs b/drivers/android/binder/thre= ad.rs index a51821dde0ad..c4d67b9ef39b 100644 --- a/drivers/android/binder/thread.rs +++ b/drivers/android/binder/thread.rs @@ -572,9 +572,21 @@ fn get_work_local(self: &Arc, wait: bool) -> Res= ult, wait: bool) -> Result>> { + let thread_has_deferred_work; + // Try to get work from the thread's work queue, using only a loca= l lock. { let mut inner =3D self.inner.lock(); + + // The process_work_list boolean is used to make us go to slee= p even if there is work + // in the thread todo-list, but it doesn't apply to the proces= s todo-list. Furthermore, + // work in the thread todo-list must still be delivered before= the process list. + // + // Thus, in some scenarios we must return the thread work now = even if we were requested + // to wait. Adjust `process_work_list` to `true` accordingly. + inner.process_work_list |=3D inner.looper_need_return; + inner.process_work_list |=3D !wait; + if let Some(work) =3D inner.pop_work() { return Ok(Some(work)); } @@ -582,18 +594,26 @@ fn get_work(self: &Arc, wait: bool) -> Result return Ok(Some(work)), GetWorkOrRegister::Register(reg) =3D> reg, }; @@ -609,14 +629,18 @@ fn get_work(self: &Arc, wait: bool) -> Result Ok(Some(work)), None if signal_pending =3D> Err(EINTR), None =3D> Ok(None), @@ -674,6 +698,12 @@ pub(crate) fn push_return_work(&self, reply: u32) { self.inner.lock().push_return_work(reply); } =20 + pub(crate) fn pop_work_even_if_deferred(&self) -> Option> { + let mut thread_inner =3D self.inner.lock(); + thread_inner.process_work_list =3D true; + thread_inner.pop_work() + } + fn translate_object( &self, obj_index: usize, @@ -1398,8 +1428,15 @@ fn reply_inner(self: &Arc, info: &mut Transact= ionInfo) -> BinderResult { let process =3D orig.from.process.clone(); let allow_fds =3D orig.flags & TF_ACCEPT_FDS !=3D 0; let reply =3D Transaction::new_reply(self, process, info, allo= w_fds)?; - // Not notifying: Reply to current thread. - let _ =3D self.inner.lock().push_work(completion); + { + // This performs a deferred push so that `read` can wait f= or the next incoming + // transaction without a userspace roundtrip. + let mut inner =3D self.inner.lock(); + inner.push_work_deferred(completion); + // However, if `TF_DEFER_COMPLETE` is not set, then set `p= rocess_work_list` to make + // the push non-deferred. This forces a userspace roundtri= p. + inner.process_work_list |=3D info.flags & TF_DEFER_COMPLET= E =3D=3D 0; + } orig.from.deliver_reply(Ok(reply), &orig, None); Ok(()) })() diff --git a/include/uapi/linux/android/binder.h b/include/uapi/linux/andro= id/binder.h index 701cad36de43..96e5b0184a1b 100644 --- a/include/uapi/linux/android/binder.h +++ b/include/uapi/linux/android/binder.h @@ -296,6 +296,7 @@ enum transaction_flags { TF_ACCEPT_FDS =3D 0x10, /* allow replies with file descriptors */ TF_CLEAR_BUF =3D 0x20, /* clear buffer on txn complete */ TF_UPDATE_TXN =3D 0x40, /* update the outdated pending async txn */ + TF_DEFER_COMPLETE =3D 0x80, /* defer transaction complete to userspace */ }; =20 struct binder_transaction_data { --- base-commit: 2cedf2272f1bb42471e646868ac572cc5752bd91 change-id: 20260715-defer-complete-f1dea9af13a8 Best regards, --=20 Alice Ryhl