Series comparison

-[PULL for-5.0 0/3] Block patches
+[PULL 0/2] Block patches
-The following changes since commit 8bac3ba57eecc466b7e73dabf7d19328a59f684e:
+The following changes since commit ca61fa4b803e5d0abaf6f1ceb690f23bb78a4def:
-  Merge remote-tracking branch 'remotes/rth/tags/pull-rx-20200408' into staging (2020-04-09 13:23:30 +0100)
+  Merge remote-tracking branch 'remotes/quic/tags/pull-hex-20211006' into staging (2021-10-06 12:11:14 -0700)
 are available in the Git repository at:
-  https://github.com/stefanha/qemu.git tags/block-pull-request
+  https://gitlab.com/stefanha/qemu.git tags/block-pull-request
-for you to fetch changes up to 5710a3e09f9b85801e5ce70797a4a511e5fc9e2c:
+for you to fetch changes up to 1cc7eada97914f090125e588497986f6f7900514:
-  async: use explicit memory barriers (2020-04-09 16:17:14 +0100)
+  iothread: use IOThreadParamInfo in iothread_[set|get]_param() (2021-10-07 15:29:50 +0100)
 ----------------------------------------------------------------
 Pull request
-Fixes for QEMU on aarch64 ARM hosts and fdmon-io_uring.
 ----------------------------------------------------------------
-Paolo Bonzini (2):
+Stefano Garzarella (2):
-  aio-wait: delegate polling of main AioContext if BQL not held
+  iothread: rename PollParamInfo to IOThreadParamInfo
-  async: use explicit memory barriers
+  iothread: use IOThreadParamInfo in iothread_[set|get]_param()
-Stefan Hajnoczi (1):
+ iothread.c | 28 +++++++++++++++-------------
-  aio-posix: signal-proof fdmon-io_uring
+file changed, 15 insertions(+), 13 deletions(-)
  include/block/aio-wait.h | 22 ++++++++++++++++++++++
  include/block/aio.h      | 29 ++++++++++-------------------
  util/aio-posix.c         | 16 ++++++++++++++--
  util/aio-win32.c         | 17 ++++++++++++++---
  util/async.c             | 16 ++++++++++++----
  util/fdmon-io_uring.c    | 10 ++++++++--
 files changed, 80 insertions(+), 30 deletions(-)
 --
-.25.1
+.31.1

-[PULL for-5.0 1/3] aio-posix: signal-proof fdmon-io_uring
+Deleted patch
-The io_uring_enter(2) syscall returns with errno=EINTR when interrupted
-by a signal.  Retry the syscall in this case.
-It's essential to do this in the io_uring_submit_and_wait() case.  My
-interpretation of the Linux v5.5 io_uring_enter(2) code is that it
-shouldn't affect the io_uring_submit() case, but there is no guarantee
-this will always be the case.  Let's check for -EINTR around both APIs.
-Note that the liburing APIs have -errno return values.
-Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
-Reviewed-by: Stefano Garzarella <sgarzare@redhat.com>
-Reviewed-by: Philippe Mathieu-Daudé <philmd@redhat.com>
-Message-id: 20200408091139.273851-1-stefanha@redhat.com
-Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
----
- util/fdmon-io_uring.c | 10 ++++++++--
-file changed, 8 insertions(+), 2 deletions(-)
-diff --git a/util/fdmon-io_uring.c b/util/fdmon-io_uring.c
-index XXXXXXX..XXXXXXX 100644
---- a/util/fdmon-io_uring.c
-+++ b/util/fdmon-io_uring.c
-@@ -XXX,XX +XXX,XX @@ static struct io_uring_sqe *get_sqe(AioContext *ctx)
-     }
-     /* No free sqes left, submit pending sqes first */
--    ret = io_uring_submit(ring);
-+    do {
-+        ret = io_uring_submit(ring);
-+    } while (ret == -EINTR);
-+
-     assert(ret > 1);
-     sqe = io_uring_get_sqe(ring);
-     assert(sqe);
-@@ -XXX,XX +XXX,XX @@ static int fdmon_io_uring_wait(AioContext *ctx, AioHandlerList *ready_list,
-     fill_sq_ring(ctx);
--    ret = io_uring_submit_and_wait(&ctx->fdmon_io_uring, wait_nr);
-+    do {
-+        ret = io_uring_submit_and_wait(&ctx->fdmon_io_uring, wait_nr);
-+    } while (ret == -EINTR);
-+
-     assert(ret >= 0);
-     return process_cq_ring(ctx, ready_list);
---
-.25.1

-[PULL for-5.0 2/3] aio-wait: delegate polling of main AioContext if BQL not held
+[PULL 1/2] iothread: rename PollParamInfo to IOThreadParamInfo
-From: Paolo Bonzini <pbonzini@redhat.com>
+From: Stefano Garzarella <sgarzare@redhat.com>
-Any thread that is not a iothread returns NULL for qemu_get_current_aio_context().
+Commit 1793ad0247 ("iothread: add aio-max-batch parameter") added
-As a result, it would also return true for
+a new parameter (aio-max-batch) to IOThread and used PollParamInfo
-in_aio_context_home_thread(qemu_get_aio_context()), causing
+structure to handle it.
 AIO_WAIT_WHILE to invoke aio_poll() directly.  This is incorrect
 if the BQL is not held, because aio_poll() does not expect to
 run concurrently from multiple threads, and it can actually
 happen when savevm writes to the vmstate file from the
 migration thread.
-Therefore, restrict in_aio_context_home_thread to return true
+Since it is not a parameter of the polling mechanism, we rename the
-for the main AioContext only if the BQL is held.
+structure to a more generic IOThreadParamInfo.
-The function is moved to aio-wait.h because it is mostly used
+Suggested-by: Kevin Wolf <kwolf@redhat.com>
-there and to avoid a circular reference between main-loop.h
+Signed-off-by: Stefano Garzarella <sgarzare@redhat.com>
-and block/aio.h.
+Reviewed-by: Philippe Mathieu-Daudé <philmd@redhat.com>
+Message-id: 20210727145936.147032-2-sgarzare@redhat.com
 Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>
 Message-Id: <20200407140746.8041-5-pbonzini@redhat.com>
 Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
 ---
- include/block/aio-wait.h | 22 ++++++++++++++++++++++
+ iothread.c | 14 +++++++-------
- include/block/aio.h      | 29 ++++++++++-------------------
+file changed, 7 insertions(+), 7 deletions(-)
 files changed, 32 insertions(+), 19 deletions(-)
-diff --git a/include/block/aio-wait.h b/include/block/aio-wait.h
+diff --git a/iothread.c b/iothread.c
 index XXXXXXX..XXXXXXX 100644
---- a/include/block/aio-wait.h
+--- a/iothread.c
-+++ b/include/block/aio-wait.h
++++ b/iothread.c
-@@ -XXX,XX +XXX,XX @@
+@@ -XXX,XX +XXX,XX @@ static void iothread_complete(UserCreatable *obj, Error **errp)
- #define QEMU_AIO_WAIT_H
+ typedef struct {
+     const char *name;
- #include "block/aio.h"
+     ptrdiff_t offset; /* field's byte offset in IOThread struct */
-+#include "qemu/main-loop.h"
+-} PollParamInfo;
++} IOThreadParamInfo;
- /**
-  * AioWait:
+-static PollParamInfo poll_max_ns_info = {
-@@ -XXX,XX +XXX,XX @@ void aio_wait_kick(void);
++static IOThreadParamInfo poll_max_ns_info = {
-  */
+     "poll-max-ns", offsetof(IOThread, poll_max_ns),
- void aio_wait_bh_oneshot(AioContext *ctx, QEMUBHFunc *cb, void *opaque);
+ };
+-static PollParamInfo poll_grow_info = {
-+/**
++static IOThreadParamInfo poll_grow_info = {
-+ * in_aio_context_home_thread:
+     "poll-grow", offsetof(IOThread, poll_grow),
-+ * @ctx: the aio context
+ };
-+ *
+-static PollParamInfo poll_shrink_info = {
-+ * Return whether we are running in the thread that normally runs @ctx.  Note
++static IOThreadParamInfo poll_shrink_info = {
-+ * that acquiring/releasing ctx does not affect the outcome, each AioContext
+     "poll-shrink", offsetof(IOThread, poll_shrink),
-+ * still only has one home thread that is responsible for running it.
+ };
-+ */
+-static PollParamInfo aio_max_batch_info = {
-+static inline bool in_aio_context_home_thread(AioContext *ctx)
++static IOThreadParamInfo aio_max_batch_info = {
-+{
+     "aio-max-batch", offsetof(IOThread, aio_max_batch),
-+    if (ctx == qemu_get_current_aio_context()) {
+ };
-+        return true;
-+    }
+@@ -XXX,XX +XXX,XX @@ static void iothread_get_param(Object *obj, Visitor *v,
-+
+         const char *name, void *opaque, Error **errp)
-+    if (ctx == qemu_get_aio_context()) {
+ {
-+        return qemu_mutex_iothread_locked();
+     IOThread *iothread = IOTHREAD(obj);
-+    } else {
+-    PollParamInfo *info = opaque;
-+        return false;
++    IOThreadParamInfo *info = opaque;
-+    }
+     int64_t *field = (void *)iothread + info->offset;
-+}
-+
+     visit_type_int64(v, name, field, errp);
- #endif /* QEMU_AIO_WAIT_H */
+@@ -XXX,XX +XXX,XX @@ static bool iothread_set_param(Object *obj, Visitor *v,
-diff --git a/include/block/aio.h b/include/block/aio.h
+         const char *name, void *opaque, Error **errp)
-index XXXXXXX..XXXXXXX 100644
+ {
---- a/include/block/aio.h
+     IOThread *iothread = IOTHREAD(obj);
-+++ b/include/block/aio.h
+-    PollParamInfo *info = opaque;
-@@ -XXX,XX +XXX,XX @@ struct AioContext {
++    IOThreadParamInfo *info = opaque;
-     AioHandlerList deleted_aio_handlers;
+     int64_t *field = (void *)iothread + info->offset;
+     int64_t value;
-     /* Used to avoid unnecessary event_notifier_set calls in aio_notify;
 -     * accessed with atomic primitives.  If this field is 0, everything
 -     * (file descriptors, bottom halves, timers) will be re-evaluated
 -     * before the next blocking poll(), thus the event_notifier_set call
 -     * can be skipped.  If it is non-zero, you may need to wake up a
 -     * concurrent aio_poll or the glib main event loop, making
 -     * event_notifier_set necessary.
 +     * only written from the AioContext home thread, or under the BQL in
 +     * the case of the main AioContext.  However, it is read from any
 +     * thread so it is still accessed with atomic primitives.
 +     *
 +     * If this field is 0, everything (file descriptors, bottom halves,
 +     * timers) will be re-evaluated before the next blocking poll() or
 +     * io_uring wait; therefore, the event_notifier_set call can be
 +     * skipped.  If it is non-zero, you may need to wake up a concurrent
 +     * aio_poll or the glib main event loop, making event_notifier_set
 +     * necessary.
       *
       * Bit 0 is reserved for GSource usage of the AioContext, and is 1
       * between a call to aio_ctx_prepare and the next call to aio_ctx_check.
@@ -XXX,XX +XXX,XX @@ void aio_co_enter(AioContext *ctx, struct Coroutine *co);
   */
  AioContext *qemu_get_current_aio_context(void);
 -/**
 - * in_aio_context_home_thread:
 - * @ctx: the aio context
 - *
 - * Return whether we are running in the thread that normally runs @ctx.  Note
 - * that acquiring/releasing ctx does not affect the outcome, each AioContext
 - * still only has one home thread that is responsible for running it.
 - */
 -static inline bool in_aio_context_home_thread(AioContext *ctx)
 -{
 -    return ctx == qemu_get_current_aio_context();
 -}
 -
  /**
   * aio_context_setup:
   * @ctx: the aio context
 --
-.25.1
+.31.1

-[PULL for-5.0 3/3] async: use explicit memory barriers
+[PULL 2/2] iothread: use IOThreadParamInfo in iothread_[set|get]_param()
-From: Paolo Bonzini <pbonzini@redhat.com>
+From: Stefano Garzarella <sgarzare@redhat.com>
-When using C11 atomics, non-seqcst reads and writes do not participate
+Commit 0445409d74 ("iothread: generalize
-in the total order of seqcst operations.  In util/async.c and util/aio-posix.c,
+iothread_set_param/iothread_get_param") moved common code to set and
-in particular, the pattern that we use
+get IOThread parameters in two new functions.
-          write ctx->notify_me                 write bh->scheduled
+These functions are called inside callbacks, so we don't need to use an
-          read bh->scheduled                   read ctx->notify_me
+opaque pointer. Let's replace `void *opaque` parameter with
-          if !bh->scheduled, sleep             if ctx->notify_me, notify
+`IOThreadParamInfo *info`.
-needs to use seqcst operations for both the write and the read.  In
+Suggested-by: Kevin Wolf <kwolf@redhat.com>
-general this is something that we do not want, because there can be
+Signed-off-by: Stefano Garzarella <sgarzare@redhat.com>
-many sources that are polled in addition to bottom halves.  The
+Reviewed-by: Philippe Mathieu-Daudé <philmd@redhat.com>
-alternative is to place a seqcst memory barrier between the write
+Message-id: 20210727145936.147032-3-sgarzare@redhat.com
 and the read.  This also comes with a disadvantage, in that the
 memory barrier is implicit on strongly-ordered architectures and
 it wastes a few dozen clock cycles.
 Fortunately, ctx->notify_me is never written concurrently by two
 threads, so we can assert that and relax the writes to ctx->notify_me.
 The resulting solution works and performs well on both aarch64 and x86.
 Note that the atomic_set/atomic_read combination is not an atomic
 read-modify-write, and therefore it is even weaker than C11 ATOMIC_RELAXED;
 on x86, ATOMIC_RELAXED compiles to a locked operation.
 Analyzed-by: Ying Fang <fangying1@huawei.com>
 Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>
 Tested-by: Ying Fang <fangying1@huawei.com>
 Message-Id: <20200407140746.8041-6-pbonzini@redhat.com>
 Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
 ---
- util/aio-posix.c | 16 ++++++++++++++--
+ iothread.c | 18 ++++++++++--------
- util/aio-win32.c | 17 ++++++++++++++---
+file changed, 10 insertions(+), 8 deletions(-)
  util/async.c     | 16 ++++++++++++----
 files changed, 40 insertions(+), 9 deletions(-)
-diff --git a/util/aio-posix.c b/util/aio-posix.c
+diff --git a/iothread.c b/iothread.c
 index XXXXXXX..XXXXXXX 100644
---- a/util/aio-posix.c
+--- a/iothread.c
-+++ b/util/aio-posix.c
++++ b/iothread.c
-@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
+@@ -XXX,XX +XXX,XX @@ static IOThreadParamInfo aio_max_batch_info = {
-     int64_t timeout;
+ };
-     int64_t start = 0;
+ static void iothread_get_param(Object *obj, Visitor *v,
-+    /*
+-        const char *name, void *opaque, Error **errp)
-+     * There cannot be two concurrent aio_poll calls for the same AioContext (or
++        const char *name, IOThreadParamInfo *info, Error **errp)
-+     * an aio_poll concurrent with a GSource prepare/check/dispatch callback).
+ {
-+     * We rely on this below to avoid slow locked accesses to ctx->notify_me.
+     IOThread *iothread = IOTHREAD(obj);
-+     */
+-    IOThreadParamInfo *info = opaque;
-     assert(in_aio_context_home_thread(ctx));
+     int64_t *field = (void *)iothread + info->offset;
-     /* aio_notify can avoid the expensive event_notifier_set if
+     visit_type_int64(v, name, field, errp);
-@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
+ }
-      * so disable the optimization now.
-      */
+ static bool iothread_set_param(Object *obj, Visitor *v,
-     if (blocking) {
+-        const char *name, void *opaque, Error **errp)
--        atomic_add(&ctx->notify_me, 2);
++        const char *name, IOThreadParamInfo *info, Error **errp)
-+        atomic_set(&ctx->notify_me, atomic_read(&ctx->notify_me) + 2);
+ {
-+        /*
+     IOThread *iothread = IOTHREAD(obj);
-+         * Write ctx->notify_me before computing the timeout
+-    IOThreadParamInfo *info = opaque;
-+         * (reading bottom half flags, etc.).  Pairs with
+     int64_t *field = (void *)iothread + info->offset;
-+         * smp_mb in aio_notify().
+     int64_t value;
-+         */
-+        smp_mb();
+@@ -XXX,XX +XXX,XX @@ static bool iothread_set_param(Object *obj, Visitor *v,
  static void iothread_get_poll_param(Object *obj, Visitor *v,
          const char *name, void *opaque, Error **errp)
  {
 +    IOThreadParamInfo *info = opaque;
 -    iothread_get_param(obj, v, name, opaque, errp);
 +    iothread_get_param(obj, v, name, info, errp);
  }
  static void iothread_set_poll_param(Object *obj, Visitor *v,
          const char *name, void *opaque, Error **errp)
  {
      IOThread *iothread = IOTHREAD(obj);
 +    IOThreadParamInfo *info = opaque;
 -    if (!iothread_set_param(obj, v, name, opaque, errp)) {
 +    if (!iothread_set_param(obj, v, name, info, errp)) {
          return;
      }
-     qemu_lockcnt_inc(&ctx->list_lock);
+@@ -XXX,XX +XXX,XX @@ static void iothread_set_poll_param(Object *obj, Visitor *v,
-@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
+ static void iothread_get_aio_param(Object *obj, Visitor *v,
          const char *name, void *opaque, Error **errp)
  {
 +    IOThreadParamInfo *info = opaque;
 -    iothread_get_param(obj, v, name, opaque, errp);
 +    iothread_get_param(obj, v, name, info, errp);
  }
  static void iothread_set_aio_param(Object *obj, Visitor *v,
          const char *name, void *opaque, Error **errp)
  {
      IOThread *iothread = IOTHREAD(obj);
 +    IOThreadParamInfo *info = opaque;
 -    if (!iothread_set_param(obj, v, name, opaque, errp)) {
 +    if (!iothread_set_param(obj, v, name, info, errp)) {
          return;
      }
-     if (blocking) {
--        atomic_sub(&ctx->notify_me, 2);
-+        /* Finish the poll before clearing the flag.  */
-+        atomic_store_release(&ctx->notify_me, atomic_read(&ctx->notify_me) - 2);
-         aio_notify_accept(ctx);
-     }
-diff --git a/util/aio-win32.c b/util/aio-win32.c
-index XXXXXXX..XXXXXXX 100644
---- a/util/aio-win32.c
-+++ b/util/aio-win32.c
-@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
-     int count;
-     int timeout;
-+    /*
-+     * There cannot be two concurrent aio_poll calls for the same AioContext (or
-+     * an aio_poll concurrent with a GSource prepare/check/dispatch callback).
-+     * We rely on this below to avoid slow locked accesses to ctx->notify_me.
-+     */
-+    assert(in_aio_context_home_thread(ctx));
-     progress = false;
-     /* aio_notify can avoid the expensive event_notifier_set if
-@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
-      * so disable the optimization now.
-      */
-     if (blocking) {
--        atomic_add(&ctx->notify_me, 2);
-+        atomic_set(&ctx->notify_me, atomic_read(&ctx->notify_me) + 2);
-+        /*
-+         * Write ctx->notify_me before computing the timeout
-+         * (reading bottom half flags, etc.).  Pairs with
-+         * smp_mb in aio_notify().
-+         */
-+        smp_mb();
-     }
-     qemu_lockcnt_inc(&ctx->list_lock);
-@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
-         ret = WaitForMultipleObjects(count, events, FALSE, timeout);
-         if (blocking) {
-             assert(first);
--            assert(in_aio_context_home_thread(ctx));
--            atomic_sub(&ctx->notify_me, 2);
-+            atomic_store_release(&ctx->notify_me, atomic_read(&ctx->notify_me) - 2);
-             aio_notify_accept(ctx);
-         }
-diff --git a/util/async.c b/util/async.c
-index XXXXXXX..XXXXXXX 100644
---- a/util/async.c
-+++ b/util/async.c
-@@ -XXX,XX +XXX,XX @@ aio_ctx_prepare(GSource *source, gint    *timeout)
- {
-     AioContext *ctx = (AioContext *) source;
--    atomic_or(&ctx->notify_me, 1);
-+    atomic_set(&ctx->notify_me, atomic_read(&ctx->notify_me) | 1);
-+
-+    /*
-+     * Write ctx->notify_me before computing the timeout
-+     * (reading bottom half flags, etc.).  Pairs with
-+     * smp_mb in aio_notify().
-+     */
-+    smp_mb();
-     /* We assume there is no timeout already supplied */
-     *timeout = qemu_timeout_ns_to_ms(aio_compute_timeout(ctx));
-@@ -XXX,XX +XXX,XX @@ aio_ctx_check(GSource *source)
-     QEMUBH *bh;
-     BHListSlice *s;
--    atomic_and(&ctx->notify_me, ~1);
-+    /* Finish computing the timeout before clearing the flag.  */
-+    atomic_store_release(&ctx->notify_me, atomic_read(&ctx->notify_me) & ~1);
-     aio_notify_accept(ctx);
-     QSLIST_FOREACH_RCU(bh, &ctx->bh_list, next) {
-@@ -XXX,XX +XXX,XX @@ LuringState *aio_get_linux_io_uring(AioContext *ctx)
- void aio_notify(AioContext *ctx)
- {
-     /* Write e.g. bh->scheduled before reading ctx->notify_me.  Pairs
--     * with atomic_or in aio_ctx_prepare or atomic_add in aio_poll.
-+     * with smp_mb in aio_ctx_prepare or aio_poll.
-      */
-     smp_mb();
--    if (ctx->notify_me) {
-+    if (atomic_read(&ctx->notify_me)) {
-         event_notifier_set(&ctx->notifier);
-         atomic_mb_set(&ctx->notified, true);
-     }
 --
-.25.1
+.31.1

The following changes since commit 8bac3ba57eecc466b7e73dabf7d19328a59f684e:

Merge remote-tracking branch 'remotes/rth/tags/pull-rx-20200408' into staging (2020-04-09 13:23:30 +0100)

are available in the Git repository at:

https://github.com/stefanha/qemu.git tags/block-pull-request

for you to fetch changes up to 5710a3e09f9b85801e5ce70797a4a511e5fc9e2c:

async: use explicit memory barriers (2020-04-09 16:17:14 +0100)

----------------------------------------------------------------
Pull request

Fixes for QEMU on aarch64 ARM hosts and fdmon-io_uring.

----------------------------------------------------------------

Paolo Bonzini (2):
  aio-wait: delegate polling of main AioContext if BQL not held
  async: use explicit memory barriers

Stefan Hajnoczi (1):
  aio-posix: signal-proof fdmon-io_uring

include/block/aio-wait.h | 22 ++++++++++++++++++++++
 include/block/aio.h      | 29 ++++++++++-------------------
 util/aio-posix.c         | 16 ++++++++++++++--
 util/aio-win32.c         | 17 ++++++++++++++---
 util/async.c             | 16 ++++++++++++----
 util/fdmon-io_uring.c    | 10 ++++++++--
 6 files changed, 80 insertions(+), 30 deletions(-)

-- 
2.25.1

The io_uring_enter(2) syscall returns with errno=EINTR when interrupted
by a signal.  Retry the syscall in this case.

It's essential to do this in the io_uring_submit_and_wait() case.  My
interpretation of the Linux v5.5 io_uring_enter(2) code is that it
shouldn't affect the io_uring_submit() case, but there is no guarantee
this will always be the case.  Let's check for -EINTR around both APIs.

Note that the liburing APIs have -errno return values.

Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
Reviewed-by: Stefano Garzarella <sgarzare@redhat.com>
Reviewed-by: Philippe Mathieu-Daudé <philmd@redhat.com>
Message-id: 20200408091139.273851-1-stefanha@redhat.com
Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
---
 util/fdmon-io_uring.c | 10 ++++++++--
 1 file changed, 8 insertions(+), 2 deletions(-)

diff --git a/util/fdmon-io_uring.c b/util/fdmon-io_uring.c
index XXXXXXX..XXXXXXX 100644
--- a/util/fdmon-io_uring.c
+++ b/util/fdmon-io_uring.c
@@ -XXX,XX +XXX,XX @@ static struct io_uring_sqe *get_sqe(AioContext *ctx)
     }
 
     /* No free sqes left, submit pending sqes first */
-    ret = io_uring_submit(ring);
+    do {
+        ret = io_uring_submit(ring);
+    } while (ret == -EINTR);
+
     assert(ret > 1);
     sqe = io_uring_get_sqe(ring);
     assert(sqe);
@@ -XXX,XX +XXX,XX @@ static int fdmon_io_uring_wait(AioContext *ctx, AioHandlerList *ready_list,
 
     fill_sq_ring(ctx);
 
-    ret = io_uring_submit_and_wait(&ctx->fdmon_io_uring, wait_nr);
+    do {
+        ret = io_uring_submit_and_wait(&ctx->fdmon_io_uring, wait_nr);
+    } while (ret == -EINTR);
+
     assert(ret >= 0);
 
     return process_cq_ring(ctx, ready_list);
-- 
2.25.1

From: Paolo Bonzini <pbonzini@redhat.com>

Any thread that is not a iothread returns NULL for qemu_get_current_aio_context().
As a result, it would also return true for
in_aio_context_home_thread(qemu_get_aio_context()), causing
AIO_WAIT_WHILE to invoke aio_poll() directly.  This is incorrect
if the BQL is not held, because aio_poll() does not expect to
run concurrently from multiple threads, and it can actually
happen when savevm writes to the vmstate file from the
migration thread.

Therefore, restrict in_aio_context_home_thread to return true
for the main AioContext only if the BQL is held.

The function is moved to aio-wait.h because it is mostly used
there and to avoid a circular reference between main-loop.h
and block/aio.h.

Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>
Message-Id: <20200407140746.8041-5-pbonzini@redhat.com>
Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
---
 include/block/aio-wait.h | 22 ++++++++++++++++++++++
 include/block/aio.h      | 29 ++++++++++-------------------
 2 files changed, 32 insertions(+), 19 deletions(-)

diff --git a/include/block/aio-wait.h b/include/block/aio-wait.h
index XXXXXXX..XXXXXXX 100644
--- a/include/block/aio-wait.h
+++ b/include/block/aio-wait.h
@@ -XXX,XX +XXX,XX @@
 #define QEMU_AIO_WAIT_H
 
 #include "block/aio.h"
+#include "qemu/main-loop.h"
 
 /**
  * AioWait:
@@ -XXX,XX +XXX,XX @@ void aio_wait_kick(void);
  */
 void aio_wait_bh_oneshot(AioContext *ctx, QEMUBHFunc *cb, void *opaque);
 
+/**
+ * in_aio_context_home_thread:
+ * @ctx: the aio context
+ *
+ * Return whether we are running in the thread that normally runs @ctx.  Note
+ * that acquiring/releasing ctx does not affect the outcome, each AioContext
+ * still only has one home thread that is responsible for running it.
+ */
+static inline bool in_aio_context_home_thread(AioContext *ctx)
+{
+    if (ctx == qemu_get_current_aio_context()) {
+        return true;
+    }
+
+    if (ctx == qemu_get_aio_context()) {
+        return qemu_mutex_iothread_locked();
+    } else {
+        return false;
+    }
+}
+
 #endif /* QEMU_AIO_WAIT_H */
diff --git a/include/block/aio.h b/include/block/aio.h
index XXXXXXX..XXXXXXX 100644
--- a/include/block/aio.h
+++ b/include/block/aio.h
@@ -XXX,XX +XXX,XX @@ struct AioContext {
     AioHandlerList deleted_aio_handlers;
 
     /* Used to avoid unnecessary event_notifier_set calls in aio_notify;
-     * accessed with atomic primitives.  If this field is 0, everything
-     * (file descriptors, bottom halves, timers) will be re-evaluated
-     * before the next blocking poll(), thus the event_notifier_set call
-     * can be skipped.  If it is non-zero, you may need to wake up a
-     * concurrent aio_poll or the glib main event loop, making
-     * event_notifier_set necessary.
+     * only written from the AioContext home thread, or under the BQL in
+     * the case of the main AioContext.  However, it is read from any
+     * thread so it is still accessed with atomic primitives.
+     *
+     * If this field is 0, everything (file descriptors, bottom halves,
+     * timers) will be re-evaluated before the next blocking poll() or
+     * io_uring wait; therefore, the event_notifier_set call can be
+     * skipped.  If it is non-zero, you may need to wake up a concurrent
+     * aio_poll or the glib main event loop, making event_notifier_set
+     * necessary.
      *
      * Bit 0 is reserved for GSource usage of the AioContext, and is 1
      * between a call to aio_ctx_prepare and the next call to aio_ctx_check.
@@ -XXX,XX +XXX,XX @@ void aio_co_enter(AioContext *ctx, struct Coroutine *co);
  */
 AioContext *qemu_get_current_aio_context(void);
 
-/**
- * in_aio_context_home_thread:
- * @ctx: the aio context
- *
- * Return whether we are running in the thread that normally runs @ctx.  Note
- * that acquiring/releasing ctx does not affect the outcome, each AioContext
- * still only has one home thread that is responsible for running it.
- */
-static inline bool in_aio_context_home_thread(AioContext *ctx)
-{
-    return ctx == qemu_get_current_aio_context();
-}
-
 /**
  * aio_context_setup:
  * @ctx: the aio context
-- 
2.25.1

From: Paolo Bonzini <pbonzini@redhat.com>

When using C11 atomics, non-seqcst reads and writes do not participate
in the total order of seqcst operations.  In util/async.c and util/aio-posix.c,
in particular, the pattern that we use

write ctx->notify_me                 write bh->scheduled
          read bh->scheduled                   read ctx->notify_me
          if !bh->scheduled, sleep             if ctx->notify_me, notify

needs to use seqcst operations for both the write and the read.  In
general this is something that we do not want, because there can be
many sources that are polled in addition to bottom halves.  The
alternative is to place a seqcst memory barrier between the write
and the read.  This also comes with a disadvantage, in that the
memory barrier is implicit on strongly-ordered architectures and
it wastes a few dozen clock cycles.

Fortunately, ctx->notify_me is never written concurrently by two
threads, so we can assert that and relax the writes to ctx->notify_me.
The resulting solution works and performs well on both aarch64 and x86.

Note that the atomic_set/atomic_read combination is not an atomic
read-modify-write, and therefore it is even weaker than C11 ATOMIC_RELAXED;
on x86, ATOMIC_RELAXED compiles to a locked operation.

Analyzed-by: Ying Fang <fangying1@huawei.com>
Signed-off-by: Paolo Bonzini <pbonzini@redhat.com>
Tested-by: Ying Fang <fangying1@huawei.com>
Message-Id: <20200407140746.8041-6-pbonzini@redhat.com>
Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
---
 util/aio-posix.c | 16 ++++++++++++++--
 util/aio-win32.c | 17 ++++++++++++++---
 util/async.c     | 16 ++++++++++++----
 3 files changed, 40 insertions(+), 9 deletions(-)

diff --git a/util/aio-posix.c b/util/aio-posix.c
index XXXXXXX..XXXXXXX 100644
--- a/util/aio-posix.c
+++ b/util/aio-posix.c
@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
     int64_t timeout;
     int64_t start = 0;
 
+    /*
+     * There cannot be two concurrent aio_poll calls for the same AioContext (or
+     * an aio_poll concurrent with a GSource prepare/check/dispatch callback).
+     * We rely on this below to avoid slow locked accesses to ctx->notify_me.
+     */
     assert(in_aio_context_home_thread(ctx));
 
     /* aio_notify can avoid the expensive event_notifier_set if
@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
      * so disable the optimization now.
      */
     if (blocking) {
-        atomic_add(&ctx->notify_me, 2);
+        atomic_set(&ctx->notify_me, atomic_read(&ctx->notify_me) + 2);
+        /*
+         * Write ctx->notify_me before computing the timeout
+         * (reading bottom half flags, etc.).  Pairs with
+         * smp_mb in aio_notify().
+         */
+        smp_mb();
     }
 
     qemu_lockcnt_inc(&ctx->list_lock);
@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
     }
 
     if (blocking) {
-        atomic_sub(&ctx->notify_me, 2);
+        /* Finish the poll before clearing the flag.  */
+        atomic_store_release(&ctx->notify_me, atomic_read(&ctx->notify_me) - 2);
         aio_notify_accept(ctx);
     }
 
diff --git a/util/aio-win32.c b/util/aio-win32.c
index XXXXXXX..XXXXXXX 100644
--- a/util/aio-win32.c
+++ b/util/aio-win32.c
@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
     int count;
     int timeout;
 
+    /*
+     * There cannot be two concurrent aio_poll calls for the same AioContext (or
+     * an aio_poll concurrent with a GSource prepare/check/dispatch callback).
+     * We rely on this below to avoid slow locked accesses to ctx->notify_me.
+     */
+    assert(in_aio_context_home_thread(ctx));
     progress = false;
 
     /* aio_notify can avoid the expensive event_notifier_set if
@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
      * so disable the optimization now.
      */
     if (blocking) {
-        atomic_add(&ctx->notify_me, 2);
+        atomic_set(&ctx->notify_me, atomic_read(&ctx->notify_me) + 2);
+        /*
+         * Write ctx->notify_me before computing the timeout
+         * (reading bottom half flags, etc.).  Pairs with
+         * smp_mb in aio_notify().
+         */
+        smp_mb();
     }
 
     qemu_lockcnt_inc(&ctx->list_lock);
@@ -XXX,XX +XXX,XX @@ bool aio_poll(AioContext *ctx, bool blocking)
         ret = WaitForMultipleObjects(count, events, FALSE, timeout);
         if (blocking) {
             assert(first);
-            assert(in_aio_context_home_thread(ctx));
-            atomic_sub(&ctx->notify_me, 2);
+            atomic_store_release(&ctx->notify_me, atomic_read(&ctx->notify_me) - 2);
             aio_notify_accept(ctx);
         }
 
diff --git a/util/async.c b/util/async.c
index XXXXXXX..XXXXXXX 100644
--- a/util/async.c
+++ b/util/async.c
@@ -XXX,XX +XXX,XX @@ aio_ctx_prepare(GSource *source, gint    *timeout)
 {
     AioContext *ctx = (AioContext *) source;
 
-    atomic_or(&ctx->notify_me, 1);
+    atomic_set(&ctx->notify_me, atomic_read(&ctx->notify_me) | 1);
+
+    /*
+     * Write ctx->notify_me before computing the timeout
+     * (reading bottom half flags, etc.).  Pairs with
+     * smp_mb in aio_notify().
+     */
+    smp_mb();
 
     /* We assume there is no timeout already supplied */
     *timeout = qemu_timeout_ns_to_ms(aio_compute_timeout(ctx));
@@ -XXX,XX +XXX,XX @@ aio_ctx_check(GSource *source)
     QEMUBH *bh;
     BHListSlice *s;
 
-    atomic_and(&ctx->notify_me, ~1);
+    /* Finish computing the timeout before clearing the flag.  */
+    atomic_store_release(&ctx->notify_me, atomic_read(&ctx->notify_me) & ~1);
     aio_notify_accept(ctx);
 
     QSLIST_FOREACH_RCU(bh, &ctx->bh_list, next) {
@@ -XXX,XX +XXX,XX @@ LuringState *aio_get_linux_io_uring(AioContext *ctx)
 void aio_notify(AioContext *ctx)
 {
     /* Write e.g. bh->scheduled before reading ctx->notify_me.  Pairs
-     * with atomic_or in aio_ctx_prepare or atomic_add in aio_poll.
+     * with smp_mb in aio_ctx_prepare or aio_poll.
      */
     smp_mb();
-    if (ctx->notify_me) {
+    if (atomic_read(&ctx->notify_me)) {
         event_notifier_set(&ctx->notifier);
         atomic_mb_set(&ctx->notified, true);
     }
-- 
2.25.1

From: Stefano Garzarella <sgarzare@redhat.com>

Commit 1793ad0247 ("iothread: add aio-max-batch parameter") added
a new parameter (aio-max-batch) to IOThread and used PollParamInfo
structure to handle it.

Since it is not a parameter of the polling mechanism, we rename the
structure to a more generic IOThreadParamInfo.

Suggested-by: Kevin Wolf <kwolf@redhat.com>
Signed-off-by: Stefano Garzarella <sgarzare@redhat.com>
Reviewed-by: Philippe Mathieu-Daudé <philmd@redhat.com>
Message-id: 20210727145936.147032-2-sgarzare@redhat.com
Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
---
 iothread.c | 14 +++++++-------
 1 file changed, 7 insertions(+), 7 deletions(-)

diff --git a/iothread.c b/iothread.c
index XXXXXXX..XXXXXXX 100644
--- a/iothread.c
+++ b/iothread.c
@@ -XXX,XX +XXX,XX @@ static void iothread_complete(UserCreatable *obj, Error **errp)
 typedef struct {
     const char *name;
     ptrdiff_t offset; /* field's byte offset in IOThread struct */
-} PollParamInfo;
+} IOThreadParamInfo;
 
-static PollParamInfo poll_max_ns_info = {
+static IOThreadParamInfo poll_max_ns_info = {
     "poll-max-ns", offsetof(IOThread, poll_max_ns),
 };
-static PollParamInfo poll_grow_info = {
+static IOThreadParamInfo poll_grow_info = {
     "poll-grow", offsetof(IOThread, poll_grow),
 };
-static PollParamInfo poll_shrink_info = {
+static IOThreadParamInfo poll_shrink_info = {
     "poll-shrink", offsetof(IOThread, poll_shrink),
 };
-static PollParamInfo aio_max_batch_info = {
+static IOThreadParamInfo aio_max_batch_info = {
     "aio-max-batch", offsetof(IOThread, aio_max_batch),
 };
 
@@ -XXX,XX +XXX,XX @@ static void iothread_get_param(Object *obj, Visitor *v,
         const char *name, void *opaque, Error **errp)
 {
     IOThread *iothread = IOTHREAD(obj);
-    PollParamInfo *info = opaque;
+    IOThreadParamInfo *info = opaque;
     int64_t *field = (void *)iothread + info->offset;
 
     visit_type_int64(v, name, field, errp);
@@ -XXX,XX +XXX,XX @@ static bool iothread_set_param(Object *obj, Visitor *v,
         const char *name, void *opaque, Error **errp)
 {
     IOThread *iothread = IOTHREAD(obj);
-    PollParamInfo *info = opaque;
+    IOThreadParamInfo *info = opaque;
     int64_t *field = (void *)iothread + info->offset;
     int64_t value;
 
-- 
2.31.1

From: Stefano Garzarella <sgarzare@redhat.com>

Commit 0445409d74 ("iothread: generalize
iothread_set_param/iothread_get_param") moved common code to set and
get IOThread parameters in two new functions.

These functions are called inside callbacks, so we don't need to use an
opaque pointer. Let's replace `void *opaque` parameter with
`IOThreadParamInfo *info`.

Suggested-by: Kevin Wolf <kwolf@redhat.com>
Signed-off-by: Stefano Garzarella <sgarzare@redhat.com>
Reviewed-by: Philippe Mathieu-Daudé <philmd@redhat.com>
Message-id: 20210727145936.147032-3-sgarzare@redhat.com
Signed-off-by: Stefan Hajnoczi <stefanha@redhat.com>
---
 iothread.c | 18 ++++++++++--------
 1 file changed, 10 insertions(+), 8 deletions(-)

diff --git a/iothread.c b/iothread.c
index XXXXXXX..XXXXXXX 100644
--- a/iothread.c
+++ b/iothread.c
@@ -XXX,XX +XXX,XX @@ static IOThreadParamInfo aio_max_batch_info = {
 };
 
 static void iothread_get_param(Object *obj, Visitor *v,
-        const char *name, void *opaque, Error **errp)
+        const char *name, IOThreadParamInfo *info, Error **errp)
 {
     IOThread *iothread = IOTHREAD(obj);
-    IOThreadParamInfo *info = opaque;
     int64_t *field = (void *)iothread + info->offset;
 
     visit_type_int64(v, name, field, errp);
 }
 
 static bool iothread_set_param(Object *obj, Visitor *v,
-        const char *name, void *opaque, Error **errp)
+        const char *name, IOThreadParamInfo *info, Error **errp)
 {
     IOThread *iothread = IOTHREAD(obj);
-    IOThreadParamInfo *info = opaque;
     int64_t *field = (void *)iothread + info->offset;
     int64_t value;
 
@@ -XXX,XX +XXX,XX @@ static bool iothread_set_param(Object *obj, Visitor *v,
 static void iothread_get_poll_param(Object *obj, Visitor *v,
         const char *name, void *opaque, Error **errp)
 {
+    IOThreadParamInfo *info = opaque;
 
-    iothread_get_param(obj, v, name, opaque, errp);
+    iothread_get_param(obj, v, name, info, errp);
 }
 
 static void iothread_set_poll_param(Object *obj, Visitor *v,
         const char *name, void *opaque, Error **errp)
 {
     IOThread *iothread = IOTHREAD(obj);
+    IOThreadParamInfo *info = opaque;
 
-    if (!iothread_set_param(obj, v, name, opaque, errp)) {
+    if (!iothread_set_param(obj, v, name, info, errp)) {
         return;
     }
 
@@ -XXX,XX +XXX,XX @@ static void iothread_set_poll_param(Object *obj, Visitor *v,
 static void iothread_get_aio_param(Object *obj, Visitor *v,
         const char *name, void *opaque, Error **errp)
 {
+    IOThreadParamInfo *info = opaque;
 
-    iothread_get_param(obj, v, name, opaque, errp);
+    iothread_get_param(obj, v, name, info, errp);
 }
 
 static void iothread_set_aio_param(Object *obj, Visitor *v,
         const char *name, void *opaque, Error **errp)
 {
     IOThread *iothread = IOTHREAD(obj);
+    IOThreadParamInfo *info = opaque;
 
-    if (!iothread_set_param(obj, v, name, opaque, errp)) {
+    if (!iothread_set_param(obj, v, name, info, errp)) {
         return;
     }
 
-- 
2.31.1