Series comparison

-[PULL 00/34] tcg patch queue
+[PULL 00/24] tcg patch queue
-This is mostly my code_gen_buffer cleanup, plus a few other random
+The following changes since commit 6587b0c1331d427b0939c37e763842550ed581db:
 changes thrown in.  Including a fix for a recent float32_exp2 bug.
+  Merge remote-tracking branch 'remotes/ericb/tags/pull-nbd-2021-10-15' into staging (2021-10-15 14:16:28 -0700)
 r~
 The following changes since commit 894fc4fd670aaf04a67dc7507739f914ff4bacf2:
   Merge remote-tracking branch 'remotes/jasowang/tags/net-pull-request' into staging (2021-06-11 09:21:48 +0100)
 are available in the Git repository at:
-  https://gitlab.com/rth7680/qemu.git tags/pull-tcg-20210611
+  https://gitlab.com/rth7680/qemu.git tags/pull-tcg-20211016
-for you to fetch changes up to 60afaddc208d34f6dc86dd974f6e02724fba6eb6:
+for you to fetch changes up to 995b87dedc78b0467f5f18bbc3546072ba97516a:
-  docs/devel: Explain in more detail the TB chaining mechanisms (2021-06-11 09:41:25 -0700)
+  Revert "cpu: Move cpu_common_props to hw/core/cpu.c" (2021-10-15 16:39:15 -0700)
 ----------------------------------------------------------------
-Clean up code_gen_buffer allocation.
+Move gdb singlestep to generic code
-Add tcg_remove_ops_after.
+Fix cpu_common_props
 Fix tcg_constant_* documentation.
 Improve TB chaining documentation.
 Fix float32_exp2.
 ----------------------------------------------------------------
-Jose R. Ziviani (1):
+Richard Henderson (24):
-      tcg/arm: Fix tcg_out_op function signature
+      accel/tcg: Handle gdb singlestep in cpu_tb_exec
       target/alpha: Drop checks for singlestep_enabled
       target/avr: Drop checks for singlestep_enabled
       target/cris: Drop checks for singlestep_enabled
       target/hexagon: Drop checks for singlestep_enabled
       target/arm: Drop checks for singlestep_enabled
       target/hppa: Drop checks for singlestep_enabled
       target/i386: Check CF_NO_GOTO_TB for dc->jmp_opt
       target/i386: Drop check for singlestep_enabled
       target/m68k: Drop checks for singlestep_enabled
       target/microblaze: Check CF_NO_GOTO_TB for DISAS_JUMP
       target/microblaze: Drop checks for singlestep_enabled
       target/mips: Fix single stepping
       target/mips: Drop exit checks for singlestep_enabled
       target/openrisc: Drop checks for singlestep_enabled
       target/ppc: Drop exit checks for singlestep_enabled
       target/riscv: Remove dead code after exception
       target/riscv: Remove exit_tb and lookup_and_goto_ptr
       target/rx: Drop checks for singlestep_enabled
       target/s390x: Drop check for singlestep_enabled
       target/sh4: Drop check for singlestep_enabled
       target/tricore: Drop check for singlestep_enabled
       target/xtensa: Drop check for singlestep_enabled
       Revert "cpu: Move cpu_common_props to hw/core/cpu.c"
-Luis Pires (1):
+ include/hw/core/cpu.h                          |  1 +
-      docs/devel: Explain in more detail the TB chaining mechanisms
+ target/i386/helper.h                           |  1 -
  target/rx/helper.h                             |  1 -
  target/sh4/helper.h                            |  1 -
  target/tricore/helper.h                        |  1 -
  accel/tcg/cpu-exec.c                           | 11 ++++
  cpu.c                                          | 21 ++++++++
  hw/core/cpu-common.c                           | 17 +-----
  target/alpha/translate.c                       | 13 ++---
  target/arm/translate-a64.c                     | 10 +---
  target/arm/translate.c                         | 36 +++----------
  target/avr/translate.c                         | 19 ++-----
  target/cris/translate.c                        | 16 ------
  target/hexagon/translate.c                     | 12 +----
  target/hppa/translate.c                        | 17 ++----
  target/i386/tcg/misc_helper.c                  |  8 ---
  target/i386/tcg/translate.c                    |  9 ++--
  target/m68k/translate.c                        | 44 ++++-----------
  target/microblaze/translate.c                  | 18 ++-----
  target/mips/tcg/translate.c                    | 75 ++++++++++++--------------
  target/openrisc/translate.c                    | 18 ++-----
  target/ppc/translate.c                         | 38 +++----------
  target/riscv/translate.c                       | 27 +---------
  target/rx/op_helper.c                          |  8 ---
  target/rx/translate.c                          | 12 +----
  target/s390x/tcg/translate.c                   |  8 +--
  target/sh4/op_helper.c                         |  5 --
  target/sh4/translate.c                         | 14 ++---
  target/tricore/op_helper.c                     |  7 ---
  target/tricore/translate.c                     | 14 +----
  target/xtensa/translate.c                      | 25 +++------
  target/riscv/insn_trans/trans_privileged.c.inc | 10 ++--
  target/riscv/insn_trans/trans_rvi.c.inc        |  8 ++-
  target/riscv/insn_trans/trans_rvv.c.inc        |  2 +-
 files changed, 141 insertions(+), 386 deletions(-)
-Richard Henderson (32):
-      meson: Split out tcg/meson.build
-      meson: Split out fpu/meson.build
-      tcg: Re-order tcg_region_init vs tcg_prologue_init
-      tcg: Remove error return from tcg_region_initial_alloc__locked
-      tcg: Split out tcg_region_initial_alloc
-      tcg: Split out tcg_region_prologue_set
-      tcg: Split out region.c
-      accel/tcg: Inline cpu_gen_init
-      accel/tcg: Move alloc_code_gen_buffer to tcg/region.c
-      accel/tcg: Rename tcg_init to tcg_init_machine
-      tcg: Create tcg_init
-      accel/tcg: Merge tcg_exec_init into tcg_init_machine
-      accel/tcg: Use MiB in tcg_init_machine
-      accel/tcg: Pass down max_cpus to tcg_init
-      tcg: Introduce tcg_max_ctxs
-      tcg: Move MAX_CODE_GEN_BUFFER_SIZE to tcg-target.h
-      tcg: Replace region.end with region.total_size
-      tcg: Rename region.start to region.after_prologue
-      tcg: Tidy tcg_n_regions
-      tcg: Tidy split_cross_256mb
-      tcg: Move in_code_gen_buffer and tests to region.c
-      tcg: Allocate code_gen_buffer into struct tcg_region_state
-      tcg: Return the map protection from alloc_code_gen_buffer
-      tcg: Sink qemu_madvise call to common code
-      util/osdep: Add qemu_mprotect_rw
-      tcg: Round the tb_size default from qemu_get_host_physmem
-      tcg: Merge buffer protection and guard page protection
-      tcg: When allocating for !splitwx, begin with PROT_NONE
-      tcg: Move tcg_init_ctx and tcg_ctx from accel/tcg/
-      tcg: Introduce tcg_remove_ops_after
-      tcg: Fix documentation for tcg_constant_* vs tcg_temp_free_*
-      softfloat: Fix tp init in float32_exp2
- docs/devel/tcg.rst        | 101 ++++-
- meson.build               |  12 +-
- accel/tcg/internal.h      |   2 +
- include/qemu/osdep.h      |   1 +
- include/sysemu/tcg.h      |   2 -
- include/tcg/tcg.h         |  28 +-
- tcg/aarch64/tcg-target.h  |   1 +
- tcg/arm/tcg-target.h      |   1 +
- tcg/i386/tcg-target.h     |   2 +
- tcg/mips/tcg-target.h     |   6 +
- tcg/ppc/tcg-target.h      |   2 +
- tcg/riscv/tcg-target.h    |   1 +
- tcg/s390/tcg-target.h     |   3 +
- tcg/sparc/tcg-target.h    |   1 +
- tcg/tcg-internal.h        |  40 ++
- tcg/tci/tcg-target.h      |   1 +
- accel/tcg/tcg-all.c       |  32 +-
- accel/tcg/translate-all.c | 439 +-------------------
- bsd-user/main.c           |   3 +-
- fpu/softfloat.c           |   2 +-
- linux-user/main.c         |   1 -
- tcg/region.c              | 999 ++++++++++++++++++++++++++++++++++++++++++++++
- tcg/tcg.c                 | 649 +++---------------------------
- util/osdep.c              |   9 +
- tcg/arm/tcg-target.c.inc  |   3 +-
- fpu/meson.build           |   1 +
- tcg/meson.build           |  14 +
-files changed, 1266 insertions(+), 1090 deletions(-)
- create mode 100644 tcg/tcg-internal.h
- create mode 100644 tcg/region.c
- create mode 100644 fpu/meson.build
- create mode 100644 tcg/meson.build

-[PULL 01/34] meson: Split out tcg/meson.build
+Deleted patch
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- meson.build     |  8 +-------
- tcg/meson.build | 13 +++++++++++++
-files changed, 14 insertions(+), 7 deletions(-)
- create mode 100644 tcg/meson.build
-diff --git a/meson.build b/meson.build
-index XXXXXXX..XXXXXXX 100644
---- a/meson.build
-+++ b/meson.build
-@@ -XXX,XX +XXX,XX @@ common_ss.add(capstone)
- specific_ss.add(files('cpu.c', 'disas.c', 'gdbstub.c'), capstone)
- specific_ss.add(when: 'CONFIG_TCG', if_true: files(
-   'fpu/softfloat.c',
--  'tcg/optimize.c',
--  'tcg/tcg-common.c',
--  'tcg/tcg-op-gvec.c',
--  'tcg/tcg-op-vec.c',
--  'tcg/tcg-op.c',
--  'tcg/tcg.c',
- ))
--specific_ss.add(when: 'CONFIG_TCG_INTERPRETER', if_true: files('tcg/tci.c'))
- # Work around a gcc bug/misfeature wherein constant propagation looks
- # through an alias:
-@@ -XXX,XX +XXX,XX @@ subdir('net')
- subdir('replay')
- subdir('semihosting')
- subdir('hw')
-+subdir('tcg')
- subdir('accel')
- subdir('plugins')
- subdir('bsd-user')
-diff --git a/tcg/meson.build b/tcg/meson.build
-new file mode 100644
-index XXXXXXX..XXXXXXX
---- /dev/null
-+++ b/tcg/meson.build
-@@ -XXX,XX +XXX,XX @@
-+tcg_ss = ss.source_set()
-+
-+tcg_ss.add(files(
-+  'optimize.c',
-+  'tcg.c',
-+  'tcg-common.c',
-+  'tcg-op.c',
-+  'tcg-op-gvec.c',
-+  'tcg-op-vec.c',
-+))
-+tcg_ss.add(when: 'CONFIG_TCG_INTERPRETER', if_true: files('tci.c'))
-+
-+specific_ss.add_all(when: 'CONFIG_TCG', if_true: tcg_ss)
---
-.25.1

-[PULL 33/34] softfloat: Fix tp init in float32_exp2
+[PULL 01/24] accel/tcg: Handle gdb singlestep in cpu_tb_exec
-Typo in the conversion to FloatParts64.
+Currently the change in cpu_tb_exec is masked by the debug exception
 being raised by the translators.  But this allows us to remove that code.
-Fixes: 572c4d862ff2
-Fixes: Coverity CID 1457457
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Message-Id: <20210607223812.110596-1-richard.henderson@linaro.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- fpu/softfloat.c | 2 +-
+ accel/tcg/cpu-exec.c | 11 +++++++++++
-file changed, 1 insertion(+), 1 deletion(-)
+file changed, 11 insertions(+)
-diff --git a/fpu/softfloat.c b/fpu/softfloat.c
+diff --git a/accel/tcg/cpu-exec.c b/accel/tcg/cpu-exec.c
 index XXXXXXX..XXXXXXX 100644
---- a/fpu/softfloat.c
+--- a/accel/tcg/cpu-exec.c
-+++ b/fpu/softfloat.c
++++ b/accel/tcg/cpu-exec.c
-@@ -XXX,XX +XXX,XX @@ float32 float32_exp2(float32 a, float_status *status)
+@@ -XXX,XX +XXX,XX @@ cpu_tb_exec(CPUState *cpu, TranslationBlock *itb, int *tb_exit)
+             cc->set_pc(cpu, last_tb->pc);
-     float_raise(float_flag_inexact, status);
+         }
+     }
--    float64_unpack_canonical(&xnp, float64_ln2, status);
++
-+    float64_unpack_canonical(&tp, float64_ln2, status);
++    /*
-     xp = *parts_mul(&xp, &tp, status);
++     * If gdb single-step, and we haven't raised another exception,
-     xnp = xp;
++     * raise a debug exception.  Single-step with another exception
 +     * is handled in cpu_handle_exception.
 +     */
 +    if (unlikely(cpu->singlestep_enabled) && cpu->exception_index == -1) {
 +        cpu->exception_index = EXCP_DEBUG;
 +        cpu_loop_exit(cpu);
 +    }
 +
      return last_tb;
  }
 --
 .25.1

-[PULL 02/34] meson: Split out fpu/meson.build
+[PULL 02/24] target/alpha: Drop checks for singlestep_enabled
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
+GDB single-stepping is now handled generically.
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- meson.build     | 4 +---
+ target/alpha/translate.c | 13 +++----------
- fpu/meson.build | 1 +
+file changed, 3 insertions(+), 10 deletions(-)
 files changed, 2 insertions(+), 3 deletions(-)
  create mode 100644 fpu/meson.build
-diff --git a/meson.build b/meson.build
+diff --git a/target/alpha/translate.c b/target/alpha/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/meson.build
+--- a/target/alpha/translate.c
-+++ b/meson.build
++++ b/target/alpha/translate.c
-@@ -XXX,XX +XXX,XX @@ subdir('softmmu')
+@@ -XXX,XX +XXX,XX @@ static void alpha_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
+         tcg_gen_movi_i64(cpu_pc, ctx->base.pc_next);
- common_ss.add(capstone)
+         /* FALLTHRU */
- specific_ss.add(files('cpu.c', 'disas.c', 'gdbstub.c'), capstone)
+     case DISAS_PC_UPDATED:
--specific_ss.add(when: 'CONFIG_TCG', if_true: files(
+-        if (!ctx->base.singlestep_enabled) {
--  'fpu/softfloat.c',
+-            tcg_gen_lookup_and_goto_ptr();
--))
+-            break;
+-        }
- # Work around a gcc bug/misfeature wherein constant propagation looks
+-        /* FALLTHRU */
- # through an alias:
++        tcg_gen_lookup_and_goto_ptr();
-@@ -XXX,XX +XXX,XX @@ subdir('replay')
++        break;
- subdir('semihosting')
+     case DISAS_PC_UPDATED_NOCHAIN:
- subdir('hw')
+-        if (ctx->base.singlestep_enabled) {
- subdir('tcg')
+-            gen_excp_1(EXCP_DEBUG, 0);
-+subdir('fpu')
+-        } else {
- subdir('accel')
+-            tcg_gen_exit_tb(NULL, 0);
- subdir('plugins')
+-        }
- subdir('bsd-user')
++        tcg_gen_exit_tb(NULL, 0);
-diff --git a/fpu/meson.build b/fpu/meson.build
+         break;
-new file mode 100644
+     default:
-index XXXXXXX..XXXXXXX
+         g_assert_not_reached();
 --- /dev/null
 +++ b/fpu/meson.build
@@ -0,0 +1 @@
 +specific_ss.add(when: 'CONFIG_TCG', if_true: files('softfloat.c'))
 --
 .25.1

-[PULL 03/34] tcg: Re-order tcg_region_init vs tcg_prologue_init
+Deleted patch
-Instead of delaying tcg_region_init until after tcg_prologue_init
-is complete, do tcg_region_init first and let tcg_prologue_init
-shrink the first region by the size of the generated prologue.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- accel/tcg/tcg-all.c       | 11 ---------
- accel/tcg/translate-all.c |  3 +++
- bsd-user/main.c           |  1 -
- linux-user/main.c         |  1 -
- tcg/tcg.c                 | 52 ++++++++++++++-------------------------
-files changed, 22 insertions(+), 46 deletions(-)
-diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
-index XXXXXXX..XXXXXXX 100644
---- a/accel/tcg/tcg-all.c
-+++ b/accel/tcg/tcg-all.c
-@@ -XXX,XX +XXX,XX @@ static int tcg_init(MachineState *ms)
-     tcg_exec_init(s->tb_size * 1024 * 1024, s->splitwx_enabled);
-     mttcg_enabled = s->mttcg_enabled;
--
--    /*
--     * Initialize TCG regions only for softmmu.
--     *
--     * This needs to be done later for user mode, because the prologue
--     * generation needs to be delayed so that GUEST_BASE is already set.
--     */
--#ifndef CONFIG_USER_ONLY
--    tcg_region_init();
--#endif /* !CONFIG_USER_ONLY */
--
-     return 0;
- }
-diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
-index XXXXXXX..XXXXXXX 100644
---- a/accel/tcg/translate-all.c
-+++ b/accel/tcg/translate-all.c
-@@ -XXX,XX +XXX,XX @@ void tcg_exec_init(unsigned long tb_size, int splitwx)
-                                splitwx, &error_fatal);
-     assert(ok);
-+    /* TODO: allocating regions is hand-in-glove with code_gen_buffer. */
-+    tcg_region_init();
-+
- #if defined(CONFIG_SOFTMMU)
-     /* There's no guest base to take into account, so go ahead and
-        initialize the prologue now.  */
-diff --git a/bsd-user/main.c b/bsd-user/main.c
-index XXXXXXX..XXXXXXX 100644
---- a/bsd-user/main.c
-+++ b/bsd-user/main.c
-@@ -XXX,XX +XXX,XX @@ int main(int argc, char **argv)
-      * the real value of GUEST_BASE into account.
-      */
-     tcg_prologue_init(tcg_ctx);
--    tcg_region_init();
-     /* build Task State */
-     memset(ts, 0, sizeof(TaskState));
-diff --git a/linux-user/main.c b/linux-user/main.c
-index XXXXXXX..XXXXXXX 100644
---- a/linux-user/main.c
-+++ b/linux-user/main.c
-@@ -XXX,XX +XXX,XX @@ int main(int argc, char **argv, char **envp)
-        generating the prologue until now so that the prologue can take
-        the real value of GUEST_BASE into account.  */
-     tcg_prologue_init(tcg_ctx);
--    tcg_region_init();
-     target_cpu_copy_regs(env, regs);
-diff --git a/tcg/tcg.c b/tcg/tcg.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/tcg.c
-+++ b/tcg/tcg.c
-@@ -XXX,XX +XXX,XX @@ TranslationBlock *tcg_tb_alloc(TCGContext *s)
- void tcg_prologue_init(TCGContext *s)
- {
--    size_t prologue_size, total_size;
--    void *buf0, *buf1;
-+    size_t prologue_size;
-     /* Put the prologue at the beginning of code_gen_buffer.  */
--    buf0 = s->code_gen_buffer;
--    total_size = s->code_gen_buffer_size;
--    s->code_ptr = buf0;
--    s->code_buf = buf0;
-+    tcg_region_assign(s, 0);
-+    s->code_ptr = s->code_gen_ptr;
-+    s->code_buf = s->code_gen_ptr;
-     s->data_gen_ptr = NULL;
--    /*
--     * The region trees are not yet configured, but tcg_splitwx_to_rx
--     * needs the bounds for an assert.
--     */
--    region.start = buf0;
--    region.end = buf0 + total_size;
--
- #ifndef CONFIG_TCG_INTERPRETER
--    tcg_qemu_tb_exec = (tcg_prologue_fn *)tcg_splitwx_to_rx(buf0);
-+    tcg_qemu_tb_exec = (tcg_prologue_fn *)tcg_splitwx_to_rx(s->code_ptr);
- #endif
--    /* Compute a high-water mark, at which we voluntarily flush the buffer
--       and start over.  The size here is arbitrary, significantly larger
--       than we expect the code generation for any one opcode to require.  */
--    s->code_gen_highwater = s->code_gen_buffer + (total_size - TCG_HIGHWATER);
--
- #ifdef TCG_TARGET_NEED_POOL_LABELS
-     s->pool_labels = NULL;
- #endif
-@@ -XXX,XX +XXX,XX @@ void tcg_prologue_init(TCGContext *s)
-     }
- #endif
--    buf1 = s->code_ptr;
-+    prologue_size = tcg_current_code_size(s);
-+
- #ifndef CONFIG_TCG_INTERPRETER
--    flush_idcache_range((uintptr_t)tcg_splitwx_to_rx(buf0), (uintptr_t)buf0,
--                        tcg_ptr_byte_diff(buf1, buf0));
-+    flush_idcache_range((uintptr_t)tcg_splitwx_to_rx(s->code_buf),
-+                        (uintptr_t)s->code_buf, prologue_size);
- #endif
--    /* Deduct the prologue from the buffer.  */
--    prologue_size = tcg_current_code_size(s);
--    s->code_gen_ptr = buf1;
--    s->code_gen_buffer = buf1;
--    s->code_buf = buf1;
--    total_size -= prologue_size;
--    s->code_gen_buffer_size = total_size;
-+    /* Deduct the prologue from the first region.  */
-+    region.start = s->code_ptr;
--    tcg_register_jit(tcg_splitwx_to_rx(s->code_gen_buffer), total_size);
-+    /* Recompute boundaries of the first region. */
-+    tcg_region_assign(s, 0);
-+
-+    tcg_register_jit(tcg_splitwx_to_rx(region.start),
-+                     region.end - region.start);
- #ifdef DEBUG_DISAS
-     if (qemu_loglevel_mask(CPU_LOG_TB_OUT_ASM)) {
-         FILE *logfile = qemu_log_lock();
-         qemu_log("PROLOGUE: [size=%zu]\n", prologue_size);
-         if (s->data_gen_ptr) {
--            size_t code_size = s->data_gen_ptr - buf0;
-+            size_t code_size = s->data_gen_ptr - s->code_gen_ptr;
-             size_t data_size = prologue_size - code_size;
-             size_t i;
--            log_disas(buf0, code_size);
-+            log_disas(s->code_gen_ptr, code_size);
-             for (i = 0; i < data_size; i += sizeof(tcg_target_ulong)) {
-                 if (sizeof(tcg_target_ulong) == 8) {
-@@ -XXX,XX +XXX,XX @@ void tcg_prologue_init(TCGContext *s)
-                 }
-             }
-         } else {
--            log_disas(buf0, prologue_size);
-+            log_disas(s->code_gen_ptr, prologue_size);
-         }
-         qemu_log("\n");
-         qemu_log_flush();
---
-.25.1

-[PULL 15/34] tcg: Introduce tcg_max_ctxs
+[PULL 03/24] target/avr: Drop checks for singlestep_enabled
-Finish the divorce of tcg/ from hw/, and do not take
+GDB single-stepping is now handled generically.
 the max cpu value from MachineState; just remember what
 we were passed in tcg_init.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
+Tested-by: Michael Rolnik <mrolnik@gmail.com>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
+Reviewed-by: Michael Rolnik <mrolnik@gmail.com>
 Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/tcg-internal.h |  3 ++-
+ target/avr/translate.c | 19 ++++---------------
- tcg/region.c       |  6 +++---
+file changed, 4 insertions(+), 15 deletions(-)
  tcg/tcg.c          | 23 ++++++++++-------------
 files changed, 15 insertions(+), 17 deletions(-)
-diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
+diff --git a/target/avr/translate.c b/target/avr/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/tcg-internal.h
+--- a/target/avr/translate.c
-+++ b/tcg/tcg-internal.h
++++ b/target/avr/translate.c
-@@ -XXX,XX +XXX,XX @@
+@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
- #define TCG_HIGHWATER 1024
+         tcg_gen_exit_tb(tb, n);
+     } else {
- extern TCGContext **tcg_ctxs;
+         tcg_gen_movi_i32(cpu_pc, dest);
--extern unsigned int n_tcg_ctxs;
+-        if (ctx->base.singlestep_enabled) {
-+extern unsigned int tcg_cur_ctxs;
+-            gen_helper_debug(cpu_env);
-+extern unsigned int tcg_max_ctxs;
+-        } else {
+-            tcg_gen_lookup_and_goto_ptr();
- void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus);
+-        }
- bool tcg_region_alloc(TCGContext *s);
++        tcg_gen_lookup_and_goto_ptr();
 diff --git a/tcg/region.c b/tcg/region.c
 index XXXXXXX..XXXXXXX 100644
 --- a/tcg/region.c
 +++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ void tcg_region_initial_alloc(TCGContext *s)
  /* Call from a safe-work context */
  void tcg_region_reset_all(void)
  {
 -    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 +    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
      unsigned int i;
      qemu_mutex_lock(&region.lock);
@@ -XXX,XX +XXX,XX @@ void tcg_region_prologue_set(TCGContext *s)
   */
  size_t tcg_code_size(void)
  {
 -    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 +    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
      unsigned int i;
      size_t total;
@@ -XXX,XX +XXX,XX @@ size_t tcg_code_capacity(void)
  size_t tcg_tb_phys_invalidate_count(void)
  {
 -    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 +    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
      unsigned int i;
      size_t total = 0;
 diff --git a/tcg/tcg.c b/tcg/tcg.c
 index XXXXXXX..XXXXXXX 100644
 --- a/tcg/tcg.c
 +++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@
  #define NO_CPU_IO_DEFS
  #include "exec/exec-all.h"
 -
 -#if !defined(CONFIG_USER_ONLY)
 -#include "hw/boards.h"
 -#endif
 -
  #include "tcg/tcg-op.h"
  #if UINTPTR_MAX == UINT32_MAX
@@ -XXX,XX +XXX,XX @@ static int tcg_out_ldst_finalize(TCGContext *s);
  #endif
  TCGContext **tcg_ctxs;
 -unsigned int n_tcg_ctxs;
 +unsigned int tcg_cur_ctxs;
 +unsigned int tcg_max_ctxs;
  TCGv_env cpu_env = 0;
  const void *tcg_code_gen_epilogue;
  uintptr_t tcg_splitwx_diff;
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
  #else
  void tcg_register_thread(void)
  {
 -    MachineState *ms = MACHINE(qdev_get_machine());
      TCGContext *s = g_malloc(sizeof(*s));
      unsigned int i, n;
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
      }
+     ctx->base.is_jmp = DISAS_NORETURN;
-     /* Claim an entry in tcg_ctxs */
+ }
--    n = qatomic_fetch_inc(&n_tcg_ctxs);
+@@ -XXX,XX +XXX,XX @@ static void avr_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
--    g_assert(n < ms->smp.max_cpus);
+         tcg_gen_movi_tl(cpu_pc, ctx->npc);
-+    n = qatomic_fetch_inc(&tcg_cur_ctxs);
+         /* fall through */
-+    g_assert(n < tcg_max_ctxs);
+     case DISAS_LOOKUP:
-     qatomic_set(&tcg_ctxs[n], s);
+-        if (!ctx->base.singlestep_enabled) {
+-            tcg_gen_lookup_and_goto_ptr();
-     if (n > 0) {
+-            break;
-@@ -XXX,XX +XXX,XX @@ static void tcg_context_init(unsigned max_cpus)
+-        }
-      */
+-        /* fall through */
- #ifdef CONFIG_USER_ONLY
++        tcg_gen_lookup_and_goto_ptr();
-     tcg_ctxs = &tcg_ctx;
++        break;
--    n_tcg_ctxs = 1;
+     case DISAS_EXIT:
-+    tcg_cur_ctxs = 1;
+-        if (ctx->base.singlestep_enabled) {
-+    tcg_max_ctxs = 1;
+-            gen_helper_debug(cpu_env);
- #else
+-        } else {
--    tcg_ctxs = g_new(TCGContext *, max_cpus);
+-            tcg_gen_exit_tb(NULL, 0);
-+    tcg_max_ctxs = max_cpus;
+-        }
-+    tcg_ctxs = g_new0(TCGContext *, max_cpus);
++        tcg_gen_exit_tb(NULL, 0);
- #endif
+         break;
+     default:
-     tcg_debug_assert(!tcg_regset_test_reg(s->reserved_regs, TCG_AREG0));
+         g_assert_not_reached();
@@ -XXX,XX +XXX,XX @@ static void tcg_reg_alloc_call(TCGContext *s, TCGOp *op)
  static inline
  void tcg_profile_snapshot(TCGProfile *prof, bool counters, bool table)
  {
 -    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 +    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
      unsigned int i;
      for (i = 0; i < n_ctxs; i++) {
@@ -XXX,XX +XXX,XX @@ void tcg_dump_op_count(void)
  int64_t tcg_cpu_exec_time(void)
  {
 -    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 +    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
      unsigned int i;
      int64_t ret = 0;
 --
 .25.1

-[PULL 28/34] tcg: When allocating for !splitwx, begin with PROT_NONE
+[PULL 04/24] target/cris: Drop checks for singlestep_enabled
-There's a change in mprotect() behaviour [1] in the latest macOS
+GDB single-stepping is now handled generically.
 on M1 and it's not yet clear if it's going to be fixed by Apple.
-In this case, instead of changing permissions of N guard pages,
-we change permissions of N rwx regions.  The same number of
-syscalls are required either way.
-[1] https://gist.github.com/hikalium/75ae822466ee4da13cbbe486498a191f
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/region.c | 19 +++++++++----------
+ target/cris/translate.c | 16 ----------------
-file changed, 9 insertions(+), 10 deletions(-)
+file changed, 16 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
+diff --git a/target/cris/translate.c b/target/cris/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
+--- a/target/cris/translate.c
-+++ b/tcg/region.c
++++ b/target/cris/translate.c
-@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
+@@ -XXX,XX +XXX,XX @@ static void cris_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
          error_free_or_abort(errp);
      }
 -    prot = PROT_READ | PROT_WRITE | PROT_EXEC;
 +    /*
 +     * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
 +     * rejects a permission change from RWX -> NONE when reserving the
 +     * guard pages later.  We can go the other way with the same number
 +     * of syscalls, so always begin with PROT_NONE.
 +     */
 +    prot = PROT_NONE;
      flags = MAP_PRIVATE | MAP_ANONYMOUS;
 -#ifdef CONFIG_TCG_INTERPRETER
 -    /* The tcg interpreter does not need execute permission. */
 -    prot = PROT_READ | PROT_WRITE;
 -#elif defined(CONFIG_DARWIN)
 +#ifdef CONFIG_DARWIN
      /* Applicable to both iOS and macOS (Apple Silicon). */
      if (!splitwx) {
          flags |= MAP_JIT;
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
              }
          }
          if (have_prot != 0) {
 -            /*
 -             * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
 -             * rejects a permission change from RWX -> NONE.  Guard pages are
 -             * nice for bug detection but are not essential; ignore any failure.
 -             */
 +            /* Guard pages are nice for bug detection but are not essential. */
              (void)qemu_mprotect_none(end, page_size);
          }
      }
+-    if (unlikely(dc->base.singlestep_enabled)) {
+-        switch (is_jmp) {
+-        case DISAS_TOO_MANY:
+-        case DISAS_UPDATE_NEXT:
+-            tcg_gen_movi_tl(env_pc, npc);
+-            /* fall through */
+-        case DISAS_JUMP:
+-        case DISAS_UPDATE:
+-            t_gen_raise_exception(EXCP_DEBUG);
+-            return;
+-        default:
+-            break;
+-        }
+-        g_assert_not_reached();
+-    }
+-
+     switch (is_jmp) {
+     case DISAS_TOO_MANY:
+         gen_goto_tb(dc, 0, npc);
 --
 .25.1

-[PULL 04/34] tcg: Remove error return from tcg_region_initial_alloc__locked
+[PULL 05/24] target/hexagon: Drop checks for singlestep_enabled
-All callers immediately assert on error, so move the assert
+GDB single-stepping is now handled generically.
 into the function itself.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/tcg.c | 19 ++++++-------------
+ target/hexagon/translate.c | 12 ++----------
-file changed, 6 insertions(+), 13 deletions(-)
+file changed, 2 insertions(+), 10 deletions(-)
-diff --git a/tcg/tcg.c b/tcg/tcg.c
+diff --git a/target/hexagon/translate.c b/target/hexagon/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/tcg.c
+--- a/target/hexagon/translate.c
-+++ b/tcg/tcg.c
++++ b/target/hexagon/translate.c
-@@ -XXX,XX +XXX,XX @@ static bool tcg_region_alloc(TCGContext *s)
+@@ -XXX,XX +XXX,XX @@ static void gen_end_tb(DisasContext *ctx)
   * Perform a context's first region allocation.
   * This function does _not_ increment region.agg_size_full.
   */
 -static inline bool tcg_region_initial_alloc__locked(TCGContext *s)
 +static void tcg_region_initial_alloc__locked(TCGContext *s)
  {
--    return tcg_region_alloc__locked(s);
+     gen_exec_counters(ctx);
-+    bool err = tcg_region_alloc__locked(s);
+     tcg_gen_mov_tl(hex_gpr[HEX_REG_PC], hex_next_PC);
-+    g_assert(!err);
+-    if (ctx->base.singlestep_enabled) {
 -        gen_exception_raw(EXCP_DEBUG);
 -    } else {
 -        tcg_gen_exit_tb(NULL, 0);
 -    }
 +    tcg_gen_exit_tb(NULL, 0);
      ctx->base.is_jmp = DISAS_NORETURN;
  }
- /* Call from a safe-work context */
+@@ -XXX,XX +XXX,XX @@ static void hexagon_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
-@@ -XXX,XX +XXX,XX @@ void tcg_region_reset_all(void)
+     case DISAS_TOO_MANY:
+         gen_exec_counters(ctx);
-     for (i = 0; i < n_ctxs; i++) {
+         tcg_gen_movi_tl(hex_gpr[HEX_REG_PC], ctx->base.pc_next);
-         TCGContext *s = qatomic_read(&tcg_ctxs[i]);
+-        if (ctx->base.singlestep_enabled) {
--        bool err = tcg_region_initial_alloc__locked(s);
+-            gen_exception_raw(EXCP_DEBUG);
--
+-        } else {
--        g_assert(!err);
+-            tcg_gen_exit_tb(NULL, 0);
-+        tcg_region_initial_alloc__locked(s);
+-        }
-     }
++        tcg_gen_exit_tb(NULL, 0);
-     qemu_mutex_unlock(&region.lock);
+         break;
+     case DISAS_NORETURN:
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(void)
+         break;
      /* In user-mode we support only one ctx, so do the initial allocation now */
  #ifdef CONFIG_USER_ONLY
 -    {
 -        bool err = tcg_region_initial_alloc__locked(tcg_ctx);
 -
 -        g_assert(!err);
 -    }
 +    tcg_region_initial_alloc__locked(tcg_ctx);
  #endif
  }
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
      MachineState *ms = MACHINE(qdev_get_machine());
      TCGContext *s = g_malloc(sizeof(*s));
      unsigned int i, n;
 -    bool err;
      *s = tcg_init_ctx;
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
      tcg_ctx = s;
      qemu_mutex_lock(&region.lock);
 -    err = tcg_region_initial_alloc__locked(tcg_ctx);
 -    g_assert(!err);
 +    tcg_region_initial_alloc__locked(s);
      qemu_mutex_unlock(&region.lock);
  }
  #endif /* !CONFIG_USER_ONLY */
 --
 .25.1

-[PULL 05/34] tcg: Split out tcg_region_initial_alloc
+Deleted patch
-This has only one user, and currently needs an ifdef,
-but will make more sense after some code motion.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- tcg/tcg.c | 13 ++++++++++---
-file changed, 10 insertions(+), 3 deletions(-)
-diff --git a/tcg/tcg.c b/tcg/tcg.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/tcg.c
-+++ b/tcg/tcg.c
-@@ -XXX,XX +XXX,XX @@ static void tcg_region_initial_alloc__locked(TCGContext *s)
-     g_assert(!err);
- }
-+#ifndef CONFIG_USER_ONLY
-+static void tcg_region_initial_alloc(TCGContext *s)
-+{
-+    qemu_mutex_lock(&region.lock);
-+    tcg_region_initial_alloc__locked(s);
-+    qemu_mutex_unlock(&region.lock);
-+}
-+#endif
-+
- /* Call from a safe-work context */
- void tcg_region_reset_all(void)
- {
-@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
-     }
-     tcg_ctx = s;
--    qemu_mutex_lock(&region.lock);
--    tcg_region_initial_alloc__locked(s);
--    qemu_mutex_unlock(&region.lock);
-+    tcg_region_initial_alloc(s);
- }
- #endif /* !CONFIG_USER_ONLY */
---
-.25.1

-[PULL 19/34] tcg: Tidy tcg_n_regions
+[PULL 06/24] target/arm: Drop checks for singlestep_enabled
-Compute the value using straight division and bounds,
+GDB single-stepping is now handled generically.
 rather than a loop.  Pass in tb_size rather than reading
 from tcg_init_ctx.code_gen_buffer_size,
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/region.c | 29 ++++++++++++-----------------
+ target/arm/translate-a64.c | 10 ++--------
-file changed, 12 insertions(+), 17 deletions(-)
+ target/arm/translate.c     | 36 ++++++------------------------------
 files changed, 8 insertions(+), 38 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
+diff --git a/target/arm/translate-a64.c b/target/arm/translate-a64.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
+--- a/target/arm/translate-a64.c
-+++ b/tcg/region.c
++++ b/target/arm/translate-a64.c
-@@ -XXX,XX +XXX,XX @@ void tcg_region_reset_all(void)
+@@ -XXX,XX +XXX,XX @@ static inline void gen_goto_tb(DisasContext *s, int n, uint64_t dest)
-     tcg_region_tree_reset_all();
+         gen_a64_set_pc_im(dest);
          if (s->ss_active) {
              gen_step_complete_exception(s);
 -        } else if (s->base.singlestep_enabled) {
 -            gen_exception_internal(EXCP_DEBUG);
          } else {
              tcg_gen_lookup_and_goto_ptr();
              s->base.is_jmp = DISAS_NORETURN;
@@ -XXX,XX +XXX,XX @@ static void aarch64_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
  {
      DisasContext *dc = container_of(dcbase, DisasContext, base);
 -    if (unlikely(dc->base.singlestep_enabled || dc->ss_active)) {
 +    if (unlikely(dc->ss_active)) {
          /* Note that this means single stepping WFI doesn't halt the CPU.
           * For conditional branch insns this is harmless unreachable code as
           * gen_goto_tb() has already handled emitting the debug exception
@@ -XXX,XX +XXX,XX @@ static void aarch64_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
              /* fall through */
          case DISAS_EXIT:
          case DISAS_JUMP:
 -            if (dc->base.singlestep_enabled) {
 -                gen_exception_internal(EXCP_DEBUG);
 -            } else {
 -                gen_step_complete_exception(dc);
 -            }
 +            gen_step_complete_exception(dc);
              break;
          case DISAS_NORETURN:
              break;
 diff --git a/target/arm/translate.c b/target/arm/translate.c
 index XXXXXXX..XXXXXXX 100644
 --- a/target/arm/translate.c
 +++ b/target/arm/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_exception_internal(int excp)
      tcg_temp_free_i32(tcg_excp);
  }
--static size_t tcg_n_regions(unsigned max_cpus)
+-static void gen_step_complete_exception(DisasContext *s)
-+static size_t tcg_n_regions(size_t tb_size, unsigned max_cpus)
++static void gen_singlestep_exception(DisasContext *s)
  {
- #ifdef CONFIG_USER_ONLY
+     /* We just completed step of an insn. Move from Active-not-pending
-     return 1;
+      * to Active-pending, and then also take the swstep exception.
- #else
+@@ -XXX,XX +XXX,XX @@ static void gen_step_complete_exception(DisasContext *s)
-+    size_t n_regions;
+     s->base.is_jmp = DISAS_NORETURN;
-+
+ }
 -static void gen_singlestep_exception(DisasContext *s)
 -{
 -    /* Generate the right kind of exception for singlestep, which is
 -     * either the architectural singlestep or EXCP_DEBUG for QEMU's
 -     * gdb singlestepping.
 -     */
 -    if (s->ss_active) {
 -        gen_step_complete_exception(s);
 -    } else {
 -        gen_exception_internal(EXCP_DEBUG);
 -    }
 -}
 -
 -static inline bool is_singlestepping(DisasContext *s)
 -{
 -    /* Return true if we are singlestepping either because of
 -     * architectural singlestep or QEMU gdbstub singlestep. This does
 -     * not include the command line '-singlestep' mode which is rather
 -     * misnamed as it only means "one instruction per TB" and doesn't
 -     * affect the code we generate.
 -     */
 -    return s->base.singlestep_enabled || s->ss_active;
 -}
 -
  void clear_eci_state(DisasContext *s)
  {
      /*
-      * It is likely that some vCPUs will translate more code than others,
+@@ -XXX,XX +XXX,XX @@ static inline void gen_bx_excret_final_code(DisasContext *s)
-      * so we first try to set more regions than max_cpus, with those regions
+     /* Is the new PC value in the magic range indicating exception return? */
-      * being of reasonable size. If that's not possible we make do by evenly
+     tcg_gen_brcondi_i32(TCG_COND_GEU, cpu_R[15], min_magic, excret_label);
-      * dividing the code_gen_buffer among the vCPUs.
+     /* No: end the TB as we would for a DISAS_JMP */
-      */
+-    if (is_singlestepping(s)) {
--    size_t i;
++    if (s->ss_active) {
--
+         gen_singlestep_exception(s);
-     /* Use a single region if all we have is one vCPU thread */
+     } else {
-     if (max_cpus == 1 || !qemu_tcg_mttcg_enabled()) {
+         tcg_gen_exit_tb(NULL, 0);
-         return 1;
+@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *s, int n, target_ulong dest)
  /* Jump, specifying which TB number to use if we gen_goto_tb() */
  static inline void gen_jmp_tb(DisasContext *s, uint32_t dest, int tbno)
  {
 -    if (unlikely(is_singlestepping(s))) {
 +    if (unlikely(s->ss_active)) {
          /* An indirect jump so that we still trigger the debug exception.  */
          gen_set_pc_im(s, dest);
          s->base.is_jmp = DISAS_JUMP;
@@ -XXX,XX +XXX,XX @@ static void arm_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cs)
      dc->page_start = dc->base.pc_first & TARGET_PAGE_MASK;
      /* If architectural single step active, limit to 1.  */
 -    if (is_singlestepping(dc)) {
 +    if (dc->ss_active) {
          dc->base.max_insns = 1;
      }
--    /* Try to have more regions than max_cpus, with each region being >= 2 MB */
+@@ -XXX,XX +XXX,XX @@ static void arm_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
--    for (i = 8; i > 0; i--) {
+          * insn codepath itself.
--        size_t regions_per_thread = i;
+          */
--        size_t region_size;
+         gen_bx_excret_final_code(dc);
--
+-    } else if (unlikely(is_singlestepping(dc))) {
--        region_size = tcg_init_ctx.code_gen_buffer_size;
++    } else if (unlikely(dc->ss_active)) {
--        region_size /= max_cpus * regions_per_thread;
+         /* Unconditional and "condition passed" instruction codepath. */
--
+         switch (dc->base.is_jmp) {
--        if (region_size >= 2 * 1024u * 1024) {
+         case DISAS_SWI:
--            return max_cpus * regions_per_thread;
+@@ -XXX,XX +XXX,XX @@ static void arm_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
--        }
+         /* "Condition failed" instruction codepath for the branch/trap insn */
-+    /*
+         gen_set_label(dc->condlabel);
-+     * Try to have more regions than max_cpus, with each region being >= 2 MB.
+         gen_set_condexec(dc);
-+     * If we can't, then just allocate one region per vCPU thread.
+-        if (unlikely(is_singlestepping(dc))) {
-+     */
++        if (unlikely(dc->ss_active)) {
-+    n_regions = tb_size / (2 * MiB);
+             gen_set_pc_im(dc, dc->base.pc_next);
-+    if (n_regions <= max_cpus) {
+             gen_singlestep_exception(dc);
-+        return max_cpus;
+         } else {
      }
 -    /* If we can't, then just allocate one region per vCPU thread */
 -    return max_cpus;
 +    return MIN(n_regions, max_cpus * 8);
  #endif
  }
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
      buf = tcg_init_ctx.code_gen_buffer;
      total_size = tcg_init_ctx.code_gen_buffer_size;
      page_size = qemu_real_host_page_size;
 -    n_regions = tcg_n_regions(max_cpus);
 +    n_regions = tcg_n_regions(total_size, max_cpus);
      /* The first region will be 'aligned - buf' bytes larger than the others */
      aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
 --
 .25.1

-[PULL 14/34] accel/tcg: Pass down max_cpus to tcg_init
+[PULL 07/24] target/hppa: Drop checks for singlestep_enabled
-Start removing the include of hw/boards.h from tcg/.
+GDB single-stepping is now handled generically.
 Pass down the max_cpus value from tcg_init_machine,
 where we have the MachineState already.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- include/tcg/tcg.h   |  2 +-
+ target/hppa/translate.c | 17 ++++-------------
- tcg/tcg-internal.h  |  2 +-
+file changed, 4 insertions(+), 13 deletions(-)
  accel/tcg/tcg-all.c | 10 +++++++++-
  tcg/region.c        | 32 +++++++++++---------------------
  tcg/tcg.c           | 10 ++++------
 files changed, 26 insertions(+), 30 deletions(-)
-diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
+diff --git a/target/hppa/translate.c b/target/hppa/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/include/tcg/tcg.h
+--- a/target/hppa/translate.c
-+++ b/include/tcg/tcg.h
++++ b/target/hppa/translate.c
-@@ -XXX,XX +XXX,XX @@ static inline void *tcg_malloc(int size)
+@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int which,
      } else {
          copy_iaoq_entry(cpu_iaoq_f, f, cpu_iaoq_b);
          copy_iaoq_entry(cpu_iaoq_b, b, ctx->iaoq_n_var);
 -        if (ctx->base.singlestep_enabled) {
 -            gen_excp_1(EXCP_DEBUG);
 -        } else {
 -            tcg_gen_lookup_and_goto_ptr();
 -        }
 +        tcg_gen_lookup_and_goto_ptr();
      }
  }
--void tcg_init(size_t tb_size, int splitwx);
+@@ -XXX,XX +XXX,XX @@ static bool do_rfi(DisasContext *ctx, bool rfi_r)
-+void tcg_init(size_t tb_size, int splitwx, unsigned max_cpus);
+         gen_helper_rfi(cpu_env);
  void tcg_register_thread(void);
  void tcg_prologue_init(TCGContext *s);
  void tcg_func_start(TCGContext *s);
 diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
 index XXXXXXX..XXXXXXX 100644
 --- a/tcg/tcg-internal.h
 +++ b/tcg/tcg-internal.h
@@ -XXX,XX +XXX,XX @@
  extern TCGContext **tcg_ctxs;
  extern unsigned int n_tcg_ctxs;
 -void tcg_region_init(size_t tb_size, int splitwx);
 +void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus);
  bool tcg_region_alloc(TCGContext *s);
  void tcg_region_initial_alloc(TCGContext *s);
  void tcg_region_prologue_set(TCGContext *s);
 diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
 index XXXXXXX..XXXXXXX 100644
 --- a/accel/tcg/tcg-all.c
 +++ b/accel/tcg/tcg-all.c
@@ -XXX,XX +XXX,XX @@
  #include "qemu/accel.h"
  #include "qapi/qapi-builtin-visit.h"
  #include "qemu/units.h"
 +#if !defined(CONFIG_USER_ONLY)
 +#include "hw/boards.h"
 +#endif
  #include "internal.h"
  struct TCGState {
@@ -XXX,XX +XXX,XX @@ bool mttcg_enabled;
  static int tcg_init_machine(MachineState *ms)
  {
      TCGState *s = TCG_STATE(current_accel());
 +#ifdef CONFIG_USER_ONLY
 +    unsigned max_cpus = 1;
 +#else
 +    unsigned max_cpus = ms->smp.max_cpus;
 +#endif
      tcg_allowed = true;
      mttcg_enabled = s->mttcg_enabled;
      page_init();
      tb_htable_init();
 -    tcg_init(s->tb_size * MiB, s->splitwx_enabled);
 +    tcg_init(s->tb_size * MiB, s->splitwx_enabled, max_cpus);
  #if defined(CONFIG_SOFTMMU)
      /*
 diff --git a/tcg/region.c b/tcg/region.c
 index XXXXXXX..XXXXXXX 100644
 --- a/tcg/region.c
 +++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@
  #include "qapi/error.h"
  #include "exec/exec-all.h"
  #include "tcg/tcg.h"
 -#if !defined(CONFIG_USER_ONLY)
 -#include "hw/boards.h"
 -#endif
  #include "tcg-internal.h"
@@ -XXX,XX +XXX,XX @@ void tcg_region_reset_all(void)
      tcg_region_tree_reset_all();
  }
 +static size_t tcg_n_regions(unsigned max_cpus)
 +{
  #ifdef CONFIG_USER_ONLY
 -static size_t tcg_n_regions(void)
 -{
      return 1;
 -}
  #else
 -/*
 - * It is likely that some vCPUs will translate more code than others, so we
 - * first try to set more regions than max_cpus, with those regions being of
 - * reasonable size. If that's not possible we make do by evenly dividing
 - * the code_gen_buffer among the vCPUs.
 - */
 -static size_t tcg_n_regions(void)
 -{
 +    /*
 +     * It is likely that some vCPUs will translate more code than others,
 +     * so we first try to set more regions than max_cpus, with those regions
 +     * being of reasonable size. If that's not possible we make do by evenly
 +     * dividing the code_gen_buffer among the vCPUs.
 +     */
      size_t i;
      /* Use a single region if all we have is one vCPU thread */
 -#if !defined(CONFIG_USER_ONLY)
 -    MachineState *ms = MACHINE(qdev_get_machine());
 -    unsigned int max_cpus = ms->smp.max_cpus;
 -#endif
      if (max_cpus == 1 || !qemu_tcg_mttcg_enabled()) {
          return 1;
      }
-@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(void)
+     /* Exit the TB to recognize new interrupts.  */
-     }
+-    if (ctx->base.singlestep_enabled) {
-     /* If we can't, then just allocate one region per vCPU thread */
+-        gen_excp_1(EXCP_DEBUG);
-     return max_cpus;
+-    } else {
--}
+-        tcg_gen_exit_tb(NULL, 0);
- #endif
+-    }
-+}
++    tcg_gen_exit_tb(NULL, 0);
+     ctx->base.is_jmp = DISAS_NORETURN;
- /*
-  * Minimum size of the code gen buffer.  This number is randomly chosen,
+     return nullify_end(ctx);
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
+@@ -XXX,XX +XXX,XX @@ static void hppa_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
-  * in practice. Multi-threaded guests share most if not all of their translated
+         nullify_save(ctx);
-  * code, which makes parallel code generation less appealing than in softmmu.
+         /* FALLTHRU */
-  */
+     case DISAS_IAQ_N_UPDATED:
--void tcg_region_init(size_t tb_size, int splitwx)
+-        if (ctx->base.singlestep_enabled) {
-+void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
+-            gen_excp_1(EXCP_DEBUG);
- {
+-        } else if (is_jmp != DISAS_IAQ_N_STALE_EXIT) {
-     void *buf, *aligned;
++        if (is_jmp != DISAS_IAQ_N_STALE_EXIT) {
-     size_t size;
+             tcg_gen_lookup_and_goto_ptr();
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx)
++            break;
-     buf = tcg_init_ctx.code_gen_buffer;
+         }
-     size = tcg_init_ctx.code_gen_buffer_size;
+         /* FALLTHRU */
-     page_size = qemu_real_host_page_size;
+     case DISAS_EXIT:
 -    n_regions = tcg_n_regions();
 +    n_regions = tcg_n_regions(max_cpus);
      /* The first region will be 'aligned - buf' bytes larger than the others */
      aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
 diff --git a/tcg/tcg.c b/tcg/tcg.c
 index XXXXXXX..XXXXXXX 100644
 --- a/tcg/tcg.c
 +++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ static void process_op_defs(TCGContext *s);
  static TCGTemp *tcg_global_reg_new_internal(TCGContext *s, TCGType type,
                                              TCGReg reg, const char *name);
 -static void tcg_context_init(void)
 +static void tcg_context_init(unsigned max_cpus)
  {
      TCGContext *s = &tcg_init_ctx;
      int op, total_args, n, i;
@@ -XXX,XX +XXX,XX @@ static void tcg_context_init(void)
      tcg_ctxs = &tcg_ctx;
      n_tcg_ctxs = 1;
  #else
 -    MachineState *ms = MACHINE(qdev_get_machine());
 -    unsigned int max_cpus = ms->smp.max_cpus;
      tcg_ctxs = g_new(TCGContext *, max_cpus);
  #endif
@@ -XXX,XX +XXX,XX @@ static void tcg_context_init(void)
      cpu_env = temp_tcgv_ptr(ts);
  }
 -void tcg_init(size_t tb_size, int splitwx)
 +void tcg_init(size_t tb_size, int splitwx, unsigned max_cpus)
  {
 -    tcg_context_init();
 -    tcg_region_init(tb_size, splitwx);
 +    tcg_context_init(max_cpus);
 +    tcg_region_init(tb_size, splitwx, max_cpus);
  }
  /*
 --
 .25.1

-[PULL 34/34] docs/devel: Explain in more detail the TB chaining mechanisms
+[PULL 08/24] target/i386: Check CF_NO_GOTO_TB for dc->jmp_opt
-From: Luis Pires <luis.pires@eldorado.org.br>
+We were using singlestep_enabled as a proxy for whether
 translator_use_goto_tb would always return false.
-Signed-off-by: Luis Pires <luis.pires@eldorado.org.br>
-Message-Id: <20210601125143.191165-1-luis.pires@eldorado.org.br>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- docs/devel/tcg.rst | 101 ++++++++++++++++++++++++++++++++++++++++-----
+ target/i386/tcg/translate.c | 5 +++--
-file changed, 90 insertions(+), 11 deletions(-)
+file changed, 3 insertions(+), 2 deletions(-)
-diff --git a/docs/devel/tcg.rst b/docs/devel/tcg.rst
+diff --git a/target/i386/tcg/translate.c b/target/i386/tcg/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/docs/devel/tcg.rst
+--- a/target/i386/tcg/translate.c
-+++ b/docs/devel/tcg.rst
++++ b/target/i386/tcg/translate.c
-@@ -XXX,XX +XXX,XX @@ performances.
+@@ -XXX,XX +XXX,XX @@ static void i386_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cpu)
- QEMU's dynamic translation backend is called TCG, for "Tiny Code
+     DisasContext *dc = container_of(dcbase, DisasContext, base);
- Generator". For more information, please take a look at ``tcg/README``.
+     CPUX86State *env = cpu->env_ptr;
+     uint32_t flags = dc->base.tb->flags;
--Some notable features of QEMU's dynamic translator are:
++    uint32_t cflags = tb_cflags(dc->base.tb);
-+The following sections outline some notable features and implementation
+     int cpl = (flags >> HF_CPL_SHIFT) & 3;
-+details of QEMU's dynamic translator.
+     int iopl = (flags >> IOPL_SHIFT) & 3;
- CPU state optimisations
+@@ -XXX,XX +XXX,XX @@ static void i386_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cpu)
- -----------------------
+     dc->cpuid_ext3_features = env->features[FEAT_8000_0001_ECX];
+     dc->cpuid_7_0_ebx_features = env->features[FEAT_7_0_EBX];
--The target CPUs have many internal states which change the way it
+     dc->cpuid_xsave_features = env->features[FEAT_XSAVE];
--evaluates instructions. In order to achieve a good speed, the
+-    dc->jmp_opt = !(dc->base.singlestep_enabled ||
-+The target CPUs have many internal states which change the way they
++    dc->jmp_opt = !((cflags & CF_NO_GOTO_TB) ||
-+evaluate instructions. In order to achieve a good speed, the
+                     (flags & (HF_TF_MASK | HF_INHIBIT_IRQ_MASK)));
- translation phase considers that some state information of the virtual
+     /*
- CPU cannot change in it. The state is recorded in the Translation
+      * If jmp_opt, we want to handle each string instruction individually.
- Block (TB). If the state changes (e.g. privilege level), a new TB will
+      * For icount also disable repz optimization so that each iteration
-@@ -XXX,XX +XXX,XX @@ Direct block chaining
+      * is accounted separately.
- ---------------------
+      */
+-    dc->repz_opt = !dc->jmp_opt && !(tb_cflags(dc->base.tb) & CF_USE_ICOUNT);
- After each translated basic block is executed, QEMU uses the simulated
++    dc->repz_opt = !dc->jmp_opt && !(cflags & CF_USE_ICOUNT);
--Program Counter (PC) and other cpu state information (such as the CS
-+Program Counter (PC) and other CPU state information (such as the CS
+     dc->T0 = tcg_temp_new();
- segment base value) to find the next basic block.
+     dc->T1 = tcg_temp_new();
 -In order to accelerate the most common cases where the new simulated PC
 -is known, QEMU can patch a basic block so that it jumps directly to the
 -next one.
 +In its simplest, less optimized form, this is done by exiting from the
 +current TB, going through the TB epilogue, and then back to the
 +main loop. That’s where QEMU looks for the next TB to execute,
 +translating it from the guest architecture if it isn’t already available
 +in memory. Then QEMU proceeds to execute this next TB, starting at the
 +prologue and then moving on to the translated instructions.
 -The most portable code uses an indirect jump. An indirect jump makes
 -it easier to make the jump target modification atomic. On some host
 -architectures (such as x86 or PowerPC), the ``JUMP`` opcode is
 -directly patched so that the block chaining has no overhead.
 +Exiting from the TB this way will cause the ``cpu_exec_interrupt()``
 +callback to be re-evaluated before executing additional instructions.
 +It is mandatory to exit this way after any CPU state changes that may
 +unmask interrupts.
 +
 +In order to accelerate the cases where the TB for the new
 +simulated PC is already available, QEMU has mechanisms that allow
 +multiple TBs to be chained directly, without having to go back to the
 +main loop as described above. These mechanisms are:
 +
 +``lookup_and_goto_ptr``
 +^^^^^^^^^^^^^^^^^^^^^^^
 +
 +Calling ``tcg_gen_lookup_and_goto_ptr()`` will emit a call to
 +``helper_lookup_tb_ptr``. This helper will look for an existing TB that
 +matches the current CPU state. If the destination TB is available its
 +code address is returned, otherwise the address of the JIT epilogue is
 +returned. The call to the helper is always followed by the tcg ``goto_ptr``
 +opcode, which branches to the returned address. In this way, we either
 +branch to the next TB or return to the main loop.
 +
 +``goto_tb + exit_tb``
 +^^^^^^^^^^^^^^^^^^^^^
 +
 +The translation code usually implements branching by performing the
 +following steps:
 +
 +1. Call ``tcg_gen_goto_tb()`` passing a jump slot index (either 0 or 1)
 +   as a parameter.
 +
 +2. Emit TCG instructions to update the CPU state with any information
 +   that has been assumed constant and is required by the main loop to
 +   correctly locate and execute the next TB. For most guests, this is
 +   just the PC of the branch destination, but others may store additional
 +   data. The information updated in this step must be inferable from both
 +   ``cpu_get_tb_cpu_state()`` and ``cpu_restore_state()``.
 +
 +3. Call ``tcg_gen_exit_tb()`` passing the address of the current TB and
 +   the jump slot index again.
 +
 +Step 1, ``tcg_gen_goto_tb()``, will emit a ``goto_tb`` TCG
 +instruction that later on gets translated to a jump to an address
 +associated with the specified jump slot. Initially, this is the address
 +of step 2's instructions, which update the CPU state information. Step 3,
 +``tcg_gen_exit_tb()``, exits from the current TB returning a tagged
 +pointer composed of the last executed TB’s address and the jump slot
 +index.
 +
 +The first time this whole sequence is executed, step 1 simply jumps
 +to step 2. Then the CPU state information gets updated and we exit from
 +the current TB. As a result, the behavior is very similar to the less
 +optimized form described earlier in this section.
 +
 +Next, the main loop looks for the next TB to execute using the
 +current CPU state information (creating the TB if it wasn’t already
 +available) and, before starting to execute the new TB’s instructions,
 +patches the previously executed TB by associating one of its jump
 +slots (the one specified in the call to ``tcg_gen_exit_tb()``) with the
 +address of the new TB.
 +
 +The next time this previous TB is executed and we get to that same
 +``goto_tb`` step, it will already be patched (assuming the destination TB
 +is still in memory) and will jump directly to the first instruction of
 +the destination TB, without going back to the main loop.
 +
 +For the ``goto_tb + exit_tb`` mechanism to be used, the following
 +conditions need to be satisfied:
 +
 +* The change in CPU state must be constant, e.g., a direct branch and
 +  not an indirect branch.
 +
 +* The direct branch cannot cross a page boundary. Memory mappings
 +  may change, causing the code at the destination address to change.
 +
 +Note that, on step 3 (``tcg_gen_exit_tb()``), in addition to the
 +jump slot index, the address of the TB just executed is also returned.
 +This address corresponds to the TB that will be patched; it may be
 +different than the one that was directly executed from the main loop
 +if the latter had already been chained to other TBs.
  Self-modifying code and translated code invalidation
  ----------------------------------------------------
 --
 .25.1

-[PULL 30/34] tcg: Introduce tcg_remove_ops_after
+[PULL 09/24] target/i386: Drop check for singlestep_enabled
-Introduce a function to remove everything emitted
+GDB single-stepping is now handled generically.
 since a given point.
-Reviewed-by: Peter Maydell <peter.maydell@linaro.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- include/tcg/tcg.h | 10 ++++++++++
+ target/i386/helper.h          | 1 -
- tcg/tcg.c         | 13 +++++++++++++
+ target/i386/tcg/misc_helper.c | 8 --------
-files changed, 23 insertions(+)
+ target/i386/tcg/translate.c   | 4 +---
 files changed, 1 insertion(+), 12 deletions(-)
-diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
+diff --git a/target/i386/helper.h b/target/i386/helper.h
 index XXXXXXX..XXXXXXX 100644
---- a/include/tcg/tcg.h
+--- a/target/i386/helper.h
-+++ b/include/tcg/tcg.h
++++ b/target/i386/helper.h
-@@ -XXX,XX +XXX,XX @@ void tcg_op_remove(TCGContext *s, TCGOp *op);
+@@ -XXX,XX +XXX,XX @@ DEF_HELPER_2(syscall, void, env, int)
- TCGOp *tcg_op_insert_before(TCGContext *s, TCGOp *op, TCGOpcode opc);
+ DEF_HELPER_2(sysret, void, env, int)
- TCGOp *tcg_op_insert_after(TCGContext *s, TCGOp *op, TCGOpcode opc);
+ #endif
+ DEF_HELPER_FLAGS_2(pause, TCG_CALL_NO_WG, noreturn, env, int)
-+/**
+-DEF_HELPER_FLAGS_1(debug, TCG_CALL_NO_WG, noreturn, env)
-+ * tcg_remove_ops_after:
+ DEF_HELPER_1(reset_rf, void, env)
-+ * @op: target operation
+ DEF_HELPER_FLAGS_3(raise_interrupt, TCG_CALL_NO_WG, noreturn, env, int, int)
-+ *
+ DEF_HELPER_FLAGS_2(raise_exception, TCG_CALL_NO_WG, noreturn, env, int)
-+ * Discard any opcodes emitted since @op.  Expected usage is to save
+diff --git a/target/i386/tcg/misc_helper.c b/target/i386/tcg/misc_helper.c
 + * a starting point with tcg_last_op(), speculatively emit opcodes,
 + * then decide whether or not to keep those opcodes after the fact.
 + */
 +void tcg_remove_ops_after(TCGOp *op);
 +
  void tcg_optimize(TCGContext *s);
  /* Allocate a new temporary and initialize it with a constant. */
 diff --git a/tcg/tcg.c b/tcg/tcg.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/tcg.c
+--- a/target/i386/tcg/misc_helper.c
-+++ b/tcg/tcg.c
++++ b/target/i386/tcg/misc_helper.c
-@@ -XXX,XX +XXX,XX @@ void tcg_op_remove(TCGContext *s, TCGOp *op)
+@@ -XXX,XX +XXX,XX @@ void QEMU_NORETURN helper_pause(CPUX86State *env, int next_eip_addend)
- #endif
+     do_pause(env);
  }
-+void tcg_remove_ops_after(TCGOp *op)
+-void QEMU_NORETURN helper_debug(CPUX86State *env)
-+{
+-{
-+    TCGContext *s = tcg_ctx;
+-    CPUState *cs = env_cpu(env);
-+
+-
-+    while (true) {
+-    cs->exception_index = EXCP_DEBUG;
-+        TCGOp *last = tcg_last_op();
+-    cpu_loop_exit(cs);
-+        if (last == op) {
+-}
-+            return;
+-
-+        }
+ uint64_t helper_rdpkru(CPUX86State *env, uint32_t ecx)
 +        tcg_op_remove(s, last);
 +    }
 +}
 +
  static TCGOp *tcg_op_alloc(TCGOpcode opc)
  {
-     TCGContext *s = tcg_ctx;
+     if ((env->cr[4] & CR4_PKE_MASK) == 0) {
 diff --git a/target/i386/tcg/translate.c b/target/i386/tcg/translate.c
 index XXXXXXX..XXXXXXX 100644
 --- a/target/i386/tcg/translate.c
 +++ b/target/i386/tcg/translate.c
@@ -XXX,XX +XXX,XX @@ do_gen_eob_worker(DisasContext *s, bool inhibit, bool recheck_tf, bool jr)
      if (s->base.tb->flags & HF_RF_MASK) {
          gen_helper_reset_rf(cpu_env);
      }
 -    if (s->base.singlestep_enabled) {
 -        gen_helper_debug(cpu_env);
 -    } else if (recheck_tf) {
 +    if (recheck_tf) {
          gen_helper_rechecking_single_step(cpu_env);
          tcg_gen_exit_tb(NULL, 0);
      } else if (s->flags & HF_TF_MASK) {
 --
 .25.1

-[PULL 09/34] accel/tcg: Move alloc_code_gen_buffer to tcg/region.c
+[PULL 10/24] target/m68k: Drop checks for singlestep_enabled
-Buffer management is integral to tcg.  Do not leave the allocation
+GDB single-stepping is now handled generically.
 to code outside of tcg/.  This is code movement, with further
 cleanups to follow.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
+Acked-by: Laurent Vivier <laurent@vivier.eu>
 Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- include/tcg/tcg.h         |   2 +-
+ target/m68k/translate.c | 44 +++++++++--------------------------------
- accel/tcg/translate-all.c | 414 +-----------------------------------
+file changed, 9 insertions(+), 35 deletions(-)
  tcg/region.c              | 431 +++++++++++++++++++++++++++++++++++++-
 files changed, 428 insertions(+), 419 deletions(-)
-diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
+diff --git a/target/m68k/translate.c b/target/m68k/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/include/tcg/tcg.h
+--- a/target/m68k/translate.c
-+++ b/include/tcg/tcg.h
++++ b/target/m68k/translate.c
-@@ -XXX,XX +XXX,XX @@ void *tcg_malloc_internal(TCGContext *s, int size);
+@@ -XXX,XX +XXX,XX @@ static void do_writebacks(DisasContext *s)
  void tcg_pool_reset(TCGContext *s);
  TranslationBlock *tcg_tb_alloc(TCGContext *s);
 -void tcg_region_init(void);
 +void tcg_region_init(size_t tb_size, int splitwx);
  void tb_destroy(TranslationBlock *tb);
  void tcg_region_reset_all(void);
 diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
 index XXXXXXX..XXXXXXX 100644
 --- a/accel/tcg/translate-all.c
 +++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@
   */
  #include "qemu/osdep.h"
 -#include "qemu/units.h"
  #include "qemu-common.h"
  #define NO_CPU_IO_DEFS
@@ -XXX,XX +XXX,XX @@
  #include "exec/cputlb.h"
  #include "exec/translate-all.h"
  #include "qemu/bitmap.h"
 -#include "qemu/error-report.h"
  #include "qemu/qemu-print.h"
  #include "qemu/timer.h"
  #include "qemu/main-loop.h"
@@ -XXX,XX +XXX,XX @@ static void page_lock_pair(PageDesc **ret_p1, tb_page_addr_t phys1,
      }
  }
--/* Minimum size of the code gen buffer.  This number is randomly chosen,
+-static bool is_singlestepping(DisasContext *s)
 -   but not so small that we can't have a fair number of TB's live.  */
 -#define MIN_CODE_GEN_BUFFER_SIZE     (1 * MiB)
 -
 -/* Maximum size of the code gen buffer we'd like to use.  Unless otherwise
 -   indicated, this is constrained by the range of direct branches on the
 -   host cpu, as used by the TCG implementation of goto_tb.  */
 -#if defined(__x86_64__)
 -# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
 -#elif defined(__sparc__)
 -# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
 -#elif defined(__powerpc64__)
 -# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
 -#elif defined(__powerpc__)
 -# define MAX_CODE_GEN_BUFFER_SIZE  (32 * MiB)
 -#elif defined(__aarch64__)
 -# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
 -#elif defined(__s390x__)
 -  /* We have a +- 4GB range on the branches; leave some slop.  */
 -# define MAX_CODE_GEN_BUFFER_SIZE  (3 * GiB)
 -#elif defined(__mips__)
 -  /* We have a 256MB branch region, but leave room to make sure the
 -     main executable is also within that region.  */
 -# define MAX_CODE_GEN_BUFFER_SIZE  (128 * MiB)
 -#else
 -# define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
 -#endif
 -
 -#if TCG_TARGET_REG_BITS == 32
 -#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (32 * MiB)
 -#ifdef CONFIG_USER_ONLY
 -/*
 - * For user mode on smaller 32 bit systems we may run into trouble
 - * allocating big chunks of data in the right place. On these systems
 - * we utilise a static code generation buffer directly in the binary.
 - */
 -#define USE_STATIC_CODE_GEN_BUFFER
 -#endif
 -#else /* TCG_TARGET_REG_BITS == 64 */
 -#ifdef CONFIG_USER_ONLY
 -/*
 - * As user-mode emulation typically means running multiple instances
 - * of the translator don't go too nuts with our default code gen
 - * buffer lest we make things too hard for the OS.
 - */
 -#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (128 * MiB)
 -#else
 -/*
 - * We expect most system emulation to run one or two guests per host.
 - * Users running large scale system emulation may want to tweak their
 - * runtime setup via the tb-size control on the command line.
 - */
 -#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (1 * GiB)
 -#endif
 -#endif
 -
 -#define DEFAULT_CODE_GEN_BUFFER_SIZE \
 -  (DEFAULT_CODE_GEN_BUFFER_SIZE_1 < MAX_CODE_GEN_BUFFER_SIZE \
 -   ? DEFAULT_CODE_GEN_BUFFER_SIZE_1 : MAX_CODE_GEN_BUFFER_SIZE)
 -
 -static size_t size_code_gen_buffer(size_t tb_size)
 -{
--    /* Size the buffer.  */
+-    /*
--    if (tb_size == 0) {
+-     * Return true if we are singlestepping either because of
--        size_t phys_mem = qemu_get_host_physmem();
+-     * architectural singlestep or QEMU gdbstub singlestep. This does
--        if (phys_mem == 0) {
+-     * not include the command line '-singlestep' mode which is rather
--            tb_size = DEFAULT_CODE_GEN_BUFFER_SIZE;
+-     * misnamed as it only means "one instruction per TB" and doesn't
--        } else {
+-     * affect the code we generate.
--            tb_size = MIN(DEFAULT_CODE_GEN_BUFFER_SIZE, phys_mem / 8);
+-     */
--        }
+-    return s->base.singlestep_enabled || s->ss_active;
 -    }
 -    if (tb_size < MIN_CODE_GEN_BUFFER_SIZE) {
 -        tb_size = MIN_CODE_GEN_BUFFER_SIZE;
 -    }
 -    if (tb_size > MAX_CODE_GEN_BUFFER_SIZE) {
 -        tb_size = MAX_CODE_GEN_BUFFER_SIZE;
 -    }
 -    return tb_size;
 -}
 -
--#ifdef __mips__
+ /* is_jmp field values */
--/* In order to use J and JAL within the code_gen_buffer, we require
+ #define DISAS_JUMP      DISAS_TARGET_0 /* only pc was modified dynamically */
--   that the buffer not cross a 256MB boundary.  */
+ #define DISAS_EXIT      DISAS_TARGET_1 /* cpu state was modified dynamically */
--static inline bool cross_256mb(void *addr, size_t size)
+@@ -XXX,XX +XXX,XX @@ static void gen_exception(DisasContext *s, uint32_t dest, int nr)
      s->base.is_jmp = DISAS_NORETURN;
  }
 -static void gen_singlestep_exception(DisasContext *s)
 -{
--    return ((uintptr_t)addr ^ ((uintptr_t)addr + size)) & ~0x0ffffffful;
+-    /*
 -     * Generate the right kind of exception for singlestep, which is
 -     * either the architectural singlestep or EXCP_DEBUG for QEMU's
 -     * gdb singlestepping.
 -     */
 -    if (s->ss_active) {
 -        gen_raise_exception(EXCP_TRACE);
 -    } else {
 -        gen_raise_exception(EXCP_DEBUG);
 -    }
 -}
 -
--/* We weren't able to allocate a buffer without crossing that boundary,
+ static inline void gen_addr_fault(DisasContext *s)
 -   so make do with the larger portion of the buffer that doesn't cross.
 -   Returns the new base of the buffer, and adjusts code_gen_buffer_size.  */
 -static inline void *split_cross_256mb(void *buf1, size_t size1)
 -{
 -    void *buf2 = (void *)(((uintptr_t)buf1 + size1) & ~0x0ffffffful);
 -    size_t size2 = buf1 + size1 - buf2;
 -
 -    size1 = buf2 - buf1;
 -    if (size1 < size2) {
 -        size1 = size2;
 -        buf1 = buf2;
 -    }
 -
 -    tcg_ctx->code_gen_buffer_size = size1;
 -    return buf1;
 -}
 -#endif
 -
 -#ifdef USE_STATIC_CODE_GEN_BUFFER
 -static uint8_t static_code_gen_buffer[DEFAULT_CODE_GEN_BUFFER_SIZE]
 -    __attribute__((aligned(CODE_GEN_ALIGN)));
 -
 -static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
 -{
 -    void *buf, *end;
 -    size_t size;
 -
 -    if (splitwx > 0) {
 -        error_setg(errp, "jit split-wx not supported");
 -        return false;
 -    }
 -
 -    /* page-align the beginning and end of the buffer */
 -    buf = static_code_gen_buffer;
 -    end = static_code_gen_buffer + sizeof(static_code_gen_buffer);
 -    buf = QEMU_ALIGN_PTR_UP(buf, qemu_real_host_page_size);
 -    end = QEMU_ALIGN_PTR_DOWN(end, qemu_real_host_page_size);
 -
 -    size = end - buf;
 -
 -    /* Honor a command-line option limiting the size of the buffer.  */
 -    if (size > tb_size) {
 -        size = QEMU_ALIGN_DOWN(tb_size, qemu_real_host_page_size);
 -    }
 -    tcg_ctx->code_gen_buffer_size = size;
 -
 -#ifdef __mips__
 -    if (cross_256mb(buf, size)) {
 -        buf = split_cross_256mb(buf, size);
 -        size = tcg_ctx->code_gen_buffer_size;
 -    }
 -#endif
 -
 -    if (qemu_mprotect_rwx(buf, size)) {
 -        error_setg_errno(errp, errno, "mprotect of jit buffer");
 -        return false;
 -    }
 -    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 -
 -    tcg_ctx->code_gen_buffer = buf;
 -    return true;
 -}
 -#elif defined(_WIN32)
 -static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
 -{
 -    void *buf;
 -
 -    if (splitwx > 0) {
 -        error_setg(errp, "jit split-wx not supported");
 -        return false;
 -    }
 -
 -    buf = VirtualAlloc(NULL, size, MEM_RESERVE | MEM_COMMIT,
 -                             PAGE_EXECUTE_READWRITE);
 -    if (buf == NULL) {
 -        error_setg_win32(errp, GetLastError(),
 -                         "allocate %zu bytes for jit buffer", size);
 -        return false;
 -    }
 -
 -    tcg_ctx->code_gen_buffer = buf;
 -    tcg_ctx->code_gen_buffer_size = size;
 -    return true;
 -}
 -#else
 -static bool alloc_code_gen_buffer_anon(size_t size, int prot,
 -                                       int flags, Error **errp)
 -{
 -    void *buf;
 -
 -    buf = mmap(NULL, size, prot, flags, -1, 0);
 -    if (buf == MAP_FAILED) {
 -        error_setg_errno(errp, errno,
 -                         "allocate %zu bytes for jit buffer", size);
 -        return false;
 -    }
 -    tcg_ctx->code_gen_buffer_size = size;
 -
 -#ifdef __mips__
 -    if (cross_256mb(buf, size)) {
 -        /*
 -         * Try again, with the original still mapped, to avoid re-acquiring
 -         * the same 256mb crossing.
 -         */
 -        size_t size2;
 -        void *buf2 = mmap(NULL, size, prot, flags, -1, 0);
 -        switch ((int)(buf2 != MAP_FAILED)) {
 -        case 1:
 -            if (!cross_256mb(buf2, size)) {
 -                /* Success!  Use the new buffer.  */
 -                munmap(buf, size);
 -                break;
 -            }
 -            /* Failure.  Work with what we had.  */
 -            munmap(buf2, size);
 -            /* fallthru */
 -        default:
 -            /* Split the original buffer.  Free the smaller half.  */
 -            buf2 = split_cross_256mb(buf, size);
 -            size2 = tcg_ctx->code_gen_buffer_size;
 -            if (buf == buf2) {
 -                munmap(buf + size2, size - size2);
 -            } else {
 -                munmap(buf, size - size2);
 -            }
 -            size = size2;
 -            break;
 -        }
 -        buf = buf2;
 -    }
 -#endif
 -
 -    /* Request large pages for the buffer.  */
 -    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 -
 -    tcg_ctx->code_gen_buffer = buf;
 -    return true;
 -}
 -
 -#ifndef CONFIG_TCG_INTERPRETER
 -#ifdef CONFIG_POSIX
 -#include "qemu/memfd.h"
 -
 -static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
 -{
 -    void *buf_rw = NULL, *buf_rx = MAP_FAILED;
 -    int fd = -1;
 -
 -#ifdef __mips__
 -    /* Find space for the RX mapping, vs the 256MiB regions. */
 -    if (!alloc_code_gen_buffer_anon(size, PROT_NONE,
 -                                    MAP_PRIVATE | MAP_ANONYMOUS |
 -                                    MAP_NORESERVE, errp)) {
 -        return false;
 -    }
 -    /* The size of the mapping may have been adjusted. */
 -    size = tcg_ctx->code_gen_buffer_size;
 -    buf_rx = tcg_ctx->code_gen_buffer;
 -#endif
 -
 -    buf_rw = qemu_memfd_alloc("tcg-jit", size, 0, &fd, errp);
 -    if (buf_rw == NULL) {
 -        goto fail;
 -    }
 -
 -#ifdef __mips__
 -    void *tmp = mmap(buf_rx, size, PROT_READ | PROT_EXEC,
 -                     MAP_SHARED | MAP_FIXED, fd, 0);
 -    if (tmp != buf_rx) {
 -        goto fail_rx;
 -    }
 -#else
 -    buf_rx = mmap(NULL, size, PROT_READ | PROT_EXEC, MAP_SHARED, fd, 0);
 -    if (buf_rx == MAP_FAILED) {
 -        goto fail_rx;
 -    }
 -#endif
 -
 -    close(fd);
 -    tcg_ctx->code_gen_buffer = buf_rw;
 -    tcg_ctx->code_gen_buffer_size = size;
 -    tcg_splitwx_diff = buf_rx - buf_rw;
 -
 -    /* Request large pages for the buffer and the splitwx.  */
 -    qemu_madvise(buf_rw, size, QEMU_MADV_HUGEPAGE);
 -    qemu_madvise(buf_rx, size, QEMU_MADV_HUGEPAGE);
 -    return true;
 -
 - fail_rx:
 -    error_setg_errno(errp, errno, "failed to map shared memory for execute");
 - fail:
 -    if (buf_rx != MAP_FAILED) {
 -        munmap(buf_rx, size);
 -    }
 -    if (buf_rw) {
 -        munmap(buf_rw, size);
 -    }
 -    if (fd >= 0) {
 -        close(fd);
 -    }
 -    return false;
 -}
 -#endif /* CONFIG_POSIX */
 -
 -#ifdef CONFIG_DARWIN
 -#include <mach/mach.h>
 -
 -extern kern_return_t mach_vm_remap(vm_map_t target_task,
 -                                   mach_vm_address_t *target_address,
 -                                   mach_vm_size_t size,
 -                                   mach_vm_offset_t mask,
 -                                   int flags,
 -                                   vm_map_t src_task,
 -                                   mach_vm_address_t src_address,
 -                                   boolean_t copy,
 -                                   vm_prot_t *cur_protection,
 -                                   vm_prot_t *max_protection,
 -                                   vm_inherit_t inheritance);
 -
 -static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
 -{
 -    kern_return_t ret;
 -    mach_vm_address_t buf_rw, buf_rx;
 -    vm_prot_t cur_prot, max_prot;
 -
 -    /* Map the read-write portion via normal anon memory. */
 -    if (!alloc_code_gen_buffer_anon(size, PROT_READ | PROT_WRITE,
 -                                    MAP_PRIVATE | MAP_ANONYMOUS, errp)) {
 -        return false;
 -    }
 -
 -    buf_rw = (mach_vm_address_t)tcg_ctx->code_gen_buffer;
 -    buf_rx = 0;
 -    ret = mach_vm_remap(mach_task_self(),
 -                        &buf_rx,
 -                        size,
 -                        0,
 -                        VM_FLAGS_ANYWHERE,
 -                        mach_task_self(),
 -                        buf_rw,
 -                        false,
 -                        &cur_prot,
 -                        &max_prot,
 -                        VM_INHERIT_NONE);
 -    if (ret != KERN_SUCCESS) {
 -        /* TODO: Convert "ret" to a human readable error message. */
 -        error_setg(errp, "vm_remap for jit splitwx failed");
 -        munmap((void *)buf_rw, size);
 -        return false;
 -    }
 -
 -    if (mprotect((void *)buf_rx, size, PROT_READ | PROT_EXEC) != 0) {
 -        error_setg_errno(errp, errno, "mprotect for jit splitwx");
 -        munmap((void *)buf_rx, size);
 -        munmap((void *)buf_rw, size);
 -        return false;
 -    }
 -
 -    tcg_splitwx_diff = buf_rx - buf_rw;
 -    return true;
 -}
 -#endif /* CONFIG_DARWIN */
 -#endif /* CONFIG_TCG_INTERPRETER */
 -
 -static bool alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
 -{
 -#ifndef CONFIG_TCG_INTERPRETER
 -# ifdef CONFIG_DARWIN
 -    return alloc_code_gen_buffer_splitwx_vmremap(size, errp);
 -# endif
 -# ifdef CONFIG_POSIX
 -    return alloc_code_gen_buffer_splitwx_memfd(size, errp);
 -# endif
 -#endif
 -    error_setg(errp, "jit split-wx not supported");
 -    return false;
 -}
 -
 -static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
 -{
 -    ERRP_GUARD();
 -    int prot, flags;
 -
 -    if (splitwx) {
 -        if (alloc_code_gen_buffer_splitwx(size, errp)) {
 -            return true;
 -        }
 -        /*
 -         * If splitwx force-on (1), fail;
 -         * if splitwx default-on (-1), fall through to splitwx off.
 -         */
 -        if (splitwx > 0) {
 -            return false;
 -        }
 -        error_free_or_abort(errp);
 -    }
 -
 -    prot = PROT_READ | PROT_WRITE | PROT_EXEC;
 -    flags = MAP_PRIVATE | MAP_ANONYMOUS;
 -#ifdef CONFIG_TCG_INTERPRETER
 -    /* The tcg interpreter does not need execute permission. */
 -    prot = PROT_READ | PROT_WRITE;
 -#elif defined(CONFIG_DARWIN)
 -    /* Applicable to both iOS and macOS (Apple Silicon). */
 -    if (!splitwx) {
 -        flags |= MAP_JIT;
 -    }
 -#endif
 -
 -    return alloc_code_gen_buffer_anon(size, prot, flags, errp);
 -}
 -#endif /* USE_STATIC_CODE_GEN_BUFFER, WIN32, POSIX */
 -
  static bool tb_cmp(const void *ap, const void *bp)
  {
-     const TranslationBlock *a = ap;
+     gen_exception(s, s->base.pc_next, EXCP_ADDRESS);
-@@ -XXX,XX +XXX,XX @@ static void tb_htable_init(void)
+@@ -XXX,XX +XXX,XX @@ static void gen_exit_tb(DisasContext *s)
-    size. */
+ /* Generate a jump to an immediate address.  */
- void tcg_exec_init(unsigned long tb_size, int splitwx)
+ static void gen_jmp_tb(DisasContext *s, int n, uint32_t dest)
  {
--    bool ok;
+-    if (unlikely(is_singlestepping(s))) {
--
++    if (unlikely(s->ss_active)) {
-     tcg_allowed = true;
+         update_cc_op(s);
-     tcg_context_init(&tcg_init_ctx);
+         tcg_gen_movi_i32(QREG_PC, dest);
-     page_init();
+-        gen_singlestep_exception(s);
-     tb_htable_init();
++        gen_raise_exception(EXCP_TRACE);
--
+     } else if (translator_use_goto_tb(&s->base, dest)) {
--    ok = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
+         tcg_gen_goto_tb(n);
--                               splitwx, &error_fatal);
+         tcg_gen_movi_i32(QREG_PC, dest);
--    assert(ok);
+@@ -XXX,XX +XXX,XX @@ static void m68k_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cpu)
--
--    /* TODO: allocating regions is hand-in-glove with code_gen_buffer. */
+     dc->ss_active = (M68K_SR_TRACE(env->sr) == M68K_SR_TRACE_ANY_INS);
--    tcg_region_init();
+     /* If architectural single step active, limit to 1 */
-+    tcg_region_init(tb_size, splitwx);
+-    if (is_singlestepping(dc)) {
++    if (dc->ss_active) {
- #if defined(CONFIG_SOFTMMU)
+         dc->base.max_insns = 1;
-     /* There's no guest base to take into account, so go ahead and
+     }
 diff --git a/tcg/region.c b/tcg/region.c
 index XXXXXXX..XXXXXXX 100644
 --- a/tcg/region.c
 +++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@
   */
  #include "qemu/osdep.h"
 +#include "qemu/units.h"
 +#include "qapi/error.h"
  #include "exec/exec-all.h"
  #include "tcg/tcg.h"
  #if !defined(CONFIG_USER_ONLY)
@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(void)
  }
- #endif
+@@ -XXX,XX +XXX,XX @@ static void m68k_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
+         break;
-+/*
+     case DISAS_TOO_MANY:
-+ * Minimum size of the code gen buffer.  This number is randomly chosen,
+         update_cc_op(dc);
-+ * but not so small that we can't have a fair number of TB's live.
+-        if (is_singlestepping(dc)) {
-+ */
++        if (dc->ss_active) {
-+#define MIN_CODE_GEN_BUFFER_SIZE     (1 * MiB)
+             tcg_gen_movi_i32(QREG_PC, dc->pc);
-+
+-            gen_singlestep_exception(dc);
-+/*
++            gen_raise_exception(EXCP_TRACE);
-+ * Maximum size of the code gen buffer we'd like to use.  Unless otherwise
+         } else {
-+ * indicated, this is constrained by the range of direct branches on the
+             gen_jmp_tb(dc, 0, dc->pc);
-+ * host cpu, as used by the TCG implementation of goto_tb.
+         }
-+ */
+         break;
-+#if defined(__x86_64__)
+     case DISAS_JUMP:
-+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
+         /* We updated CC_OP and PC in gen_jmp/gen_jmp_im.  */
-+#elif defined(__sparc__)
+-        if (is_singlestepping(dc)) {
-+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
+-            gen_singlestep_exception(dc);
-+#elif defined(__powerpc64__)
++        if (dc->ss_active) {
-+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
++            gen_raise_exception(EXCP_TRACE);
-+#elif defined(__powerpc__)
+         } else {
-+# define MAX_CODE_GEN_BUFFER_SIZE  (32 * MiB)
+             tcg_gen_lookup_and_goto_ptr();
-+#elif defined(__aarch64__)
+         }
-+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
+@@ -XXX,XX +XXX,XX @@ static void m68k_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
-+#elif defined(__s390x__)
+          * We updated CC_OP and PC in gen_exit_tb, but also modified
-+  /* We have a +- 4GB range on the branches; leave some slop.  */
+          * other state that may require returning to the main loop.
-+# define MAX_CODE_GEN_BUFFER_SIZE  (3 * GiB)
+          */
-+#elif defined(__mips__)
+-        if (is_singlestepping(dc)) {
-+  /*
+-            gen_singlestep_exception(dc);
-+   * We have a 256MB branch region, but leave room to make sure the
++        if (dc->ss_active) {
-+   * main executable is also within that region.
++            gen_raise_exception(EXCP_TRACE);
-+   */
+         } else {
-+# define MAX_CODE_GEN_BUFFER_SIZE  (128 * MiB)
+             tcg_gen_exit_tb(NULL, 0);
-+#else
+         }
 +# define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
 +#endif
 +
 +#if TCG_TARGET_REG_BITS == 32
 +#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (32 * MiB)
 +#ifdef CONFIG_USER_ONLY
 +/*
 + * For user mode on smaller 32 bit systems we may run into trouble
 + * allocating big chunks of data in the right place. On these systems
 + * we utilise a static code generation buffer directly in the binary.
 + */
 +#define USE_STATIC_CODE_GEN_BUFFER
 +#endif
 +#else /* TCG_TARGET_REG_BITS == 64 */
 +#ifdef CONFIG_USER_ONLY
 +/*
 + * As user-mode emulation typically means running multiple instances
 + * of the translator don't go too nuts with our default code gen
 + * buffer lest we make things too hard for the OS.
 + */
 +#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (128 * MiB)
 +#else
 +/*
 + * We expect most system emulation to run one or two guests per host.
 + * Users running large scale system emulation may want to tweak their
 + * runtime setup via the tb-size control on the command line.
 + */
 +#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (1 * GiB)
 +#endif
 +#endif
 +
 +#define DEFAULT_CODE_GEN_BUFFER_SIZE \
 +  (DEFAULT_CODE_GEN_BUFFER_SIZE_1 < MAX_CODE_GEN_BUFFER_SIZE \
 +   ? DEFAULT_CODE_GEN_BUFFER_SIZE_1 : MAX_CODE_GEN_BUFFER_SIZE)
 +
 +static size_t size_code_gen_buffer(size_t tb_size)
 +{
 +    /* Size the buffer.  */
 +    if (tb_size == 0) {
 +        size_t phys_mem = qemu_get_host_physmem();
 +        if (phys_mem == 0) {
 +            tb_size = DEFAULT_CODE_GEN_BUFFER_SIZE;
 +        } else {
 +            tb_size = MIN(DEFAULT_CODE_GEN_BUFFER_SIZE, phys_mem / 8);
 +        }
 +    }
 +    if (tb_size < MIN_CODE_GEN_BUFFER_SIZE) {
 +        tb_size = MIN_CODE_GEN_BUFFER_SIZE;
 +    }
 +    if (tb_size > MAX_CODE_GEN_BUFFER_SIZE) {
 +        tb_size = MAX_CODE_GEN_BUFFER_SIZE;
 +    }
 +    return tb_size;
 +}
 +
 +#ifdef __mips__
 +/*
 + * In order to use J and JAL within the code_gen_buffer, we require
 + * that the buffer not cross a 256MB boundary.
 + */
 +static inline bool cross_256mb(void *addr, size_t size)
 +{
 +    return ((uintptr_t)addr ^ ((uintptr_t)addr + size)) & ~0x0ffffffful;
 +}
 +
 +/*
 + * We weren't able to allocate a buffer without crossing that boundary,
 + * so make do with the larger portion of the buffer that doesn't cross.
 + * Returns the new base of the buffer, and adjusts code_gen_buffer_size.
 + */
 +static inline void *split_cross_256mb(void *buf1, size_t size1)
 +{
 +    void *buf2 = (void *)(((uintptr_t)buf1 + size1) & ~0x0ffffffful);
 +    size_t size2 = buf1 + size1 - buf2;
 +
 +    size1 = buf2 - buf1;
 +    if (size1 < size2) {
 +        size1 = size2;
 +        buf1 = buf2;
 +    }
 +
 +    tcg_ctx->code_gen_buffer_size = size1;
 +    return buf1;
 +}
 +#endif
 +
 +#ifdef USE_STATIC_CODE_GEN_BUFFER
 +static uint8_t static_code_gen_buffer[DEFAULT_CODE_GEN_BUFFER_SIZE]
 +    __attribute__((aligned(CODE_GEN_ALIGN)));
 +
 +static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
 +{
 +    void *buf, *end;
 +    size_t size;
 +
 +    if (splitwx > 0) {
 +        error_setg(errp, "jit split-wx not supported");
 +        return false;
 +    }
 +
 +    /* page-align the beginning and end of the buffer */
 +    buf = static_code_gen_buffer;
 +    end = static_code_gen_buffer + sizeof(static_code_gen_buffer);
 +    buf = QEMU_ALIGN_PTR_UP(buf, qemu_real_host_page_size);
 +    end = QEMU_ALIGN_PTR_DOWN(end, qemu_real_host_page_size);
 +
 +    size = end - buf;
 +
 +    /* Honor a command-line option limiting the size of the buffer.  */
 +    if (size > tb_size) {
 +        size = QEMU_ALIGN_DOWN(tb_size, qemu_real_host_page_size);
 +    }
 +    tcg_ctx->code_gen_buffer_size = size;
 +
 +#ifdef __mips__
 +    if (cross_256mb(buf, size)) {
 +        buf = split_cross_256mb(buf, size);
 +        size = tcg_ctx->code_gen_buffer_size;
 +    }
 +#endif
 +
 +    if (qemu_mprotect_rwx(buf, size)) {
 +        error_setg_errno(errp, errno, "mprotect of jit buffer");
 +        return false;
 +    }
 +    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 +
 +    tcg_ctx->code_gen_buffer = buf;
 +    return true;
 +}
 +#elif defined(_WIN32)
 +static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
 +{
 +    void *buf;
 +
 +    if (splitwx > 0) {
 +        error_setg(errp, "jit split-wx not supported");
 +        return false;
 +    }
 +
 +    buf = VirtualAlloc(NULL, size, MEM_RESERVE | MEM_COMMIT,
 +                             PAGE_EXECUTE_READWRITE);
 +    if (buf == NULL) {
 +        error_setg_win32(errp, GetLastError(),
 +                         "allocate %zu bytes for jit buffer", size);
 +        return false;
 +    }
 +
 +    tcg_ctx->code_gen_buffer = buf;
 +    tcg_ctx->code_gen_buffer_size = size;
 +    return true;
 +}
 +#else
 +static bool alloc_code_gen_buffer_anon(size_t size, int prot,
 +                                       int flags, Error **errp)
 +{
 +    void *buf;
 +
 +    buf = mmap(NULL, size, prot, flags, -1, 0);
 +    if (buf == MAP_FAILED) {
 +        error_setg_errno(errp, errno,
 +                         "allocate %zu bytes for jit buffer", size);
 +        return false;
 +    }
 +    tcg_ctx->code_gen_buffer_size = size;
 +
 +#ifdef __mips__
 +    if (cross_256mb(buf, size)) {
 +        /*
 +         * Try again, with the original still mapped, to avoid re-acquiring
 +         * the same 256mb crossing.
 +         */
 +        size_t size2;
 +        void *buf2 = mmap(NULL, size, prot, flags, -1, 0);
 +        switch ((int)(buf2 != MAP_FAILED)) {
 +        case 1:
 +            if (!cross_256mb(buf2, size)) {
 +                /* Success!  Use the new buffer.  */
 +                munmap(buf, size);
 +                break;
 +            }
 +            /* Failure.  Work with what we had.  */
 +            munmap(buf2, size);
 +            /* fallthru */
 +        default:
 +            /* Split the original buffer.  Free the smaller half.  */
 +            buf2 = split_cross_256mb(buf, size);
 +            size2 = tcg_ctx->code_gen_buffer_size;
 +            if (buf == buf2) {
 +                munmap(buf + size2, size - size2);
 +            } else {
 +                munmap(buf, size - size2);
 +            }
 +            size = size2;
 +            break;
 +        }
 +        buf = buf2;
 +    }
 +#endif
 +
 +    /* Request large pages for the buffer.  */
 +    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 +
 +    tcg_ctx->code_gen_buffer = buf;
 +    return true;
 +}
 +
 +#ifndef CONFIG_TCG_INTERPRETER
 +#ifdef CONFIG_POSIX
 +#include "qemu/memfd.h"
 +
 +static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
 +{
 +    void *buf_rw = NULL, *buf_rx = MAP_FAILED;
 +    int fd = -1;
 +
 +#ifdef __mips__
 +    /* Find space for the RX mapping, vs the 256MiB regions. */
 +    if (!alloc_code_gen_buffer_anon(size, PROT_NONE,
 +                                    MAP_PRIVATE | MAP_ANONYMOUS |
 +                                    MAP_NORESERVE, errp)) {
 +        return false;
 +    }
 +    /* The size of the mapping may have been adjusted. */
 +    size = tcg_ctx->code_gen_buffer_size;
 +    buf_rx = tcg_ctx->code_gen_buffer;
 +#endif
 +
 +    buf_rw = qemu_memfd_alloc("tcg-jit", size, 0, &fd, errp);
 +    if (buf_rw == NULL) {
 +        goto fail;
 +    }
 +
 +#ifdef __mips__
 +    void *tmp = mmap(buf_rx, size, PROT_READ | PROT_EXEC,
 +                     MAP_SHARED | MAP_FIXED, fd, 0);
 +    if (tmp != buf_rx) {
 +        goto fail_rx;
 +    }
 +#else
 +    buf_rx = mmap(NULL, size, PROT_READ | PROT_EXEC, MAP_SHARED, fd, 0);
 +    if (buf_rx == MAP_FAILED) {
 +        goto fail_rx;
 +    }
 +#endif
 +
 +    close(fd);
 +    tcg_ctx->code_gen_buffer = buf_rw;
 +    tcg_ctx->code_gen_buffer_size = size;
 +    tcg_splitwx_diff = buf_rx - buf_rw;
 +
 +    /* Request large pages for the buffer and the splitwx.  */
 +    qemu_madvise(buf_rw, size, QEMU_MADV_HUGEPAGE);
 +    qemu_madvise(buf_rx, size, QEMU_MADV_HUGEPAGE);
 +    return true;
 +
 + fail_rx:
 +    error_setg_errno(errp, errno, "failed to map shared memory for execute");
 + fail:
 +    if (buf_rx != MAP_FAILED) {
 +        munmap(buf_rx, size);
 +    }
 +    if (buf_rw) {
 +        munmap(buf_rw, size);
 +    }
 +    if (fd >= 0) {
 +        close(fd);
 +    }
 +    return false;
 +}
 +#endif /* CONFIG_POSIX */
 +
 +#ifdef CONFIG_DARWIN
 +#include <mach/mach.h>
 +
 +extern kern_return_t mach_vm_remap(vm_map_t target_task,
 +                                   mach_vm_address_t *target_address,
 +                                   mach_vm_size_t size,
 +                                   mach_vm_offset_t mask,
 +                                   int flags,
 +                                   vm_map_t src_task,
 +                                   mach_vm_address_t src_address,
 +                                   boolean_t copy,
 +                                   vm_prot_t *cur_protection,
 +                                   vm_prot_t *max_protection,
 +                                   vm_inherit_t inheritance);
 +
 +static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
 +{
 +    kern_return_t ret;
 +    mach_vm_address_t buf_rw, buf_rx;
 +    vm_prot_t cur_prot, max_prot;
 +
 +    /* Map the read-write portion via normal anon memory. */
 +    if (!alloc_code_gen_buffer_anon(size, PROT_READ | PROT_WRITE,
 +                                    MAP_PRIVATE | MAP_ANONYMOUS, errp)) {
 +        return false;
 +    }
 +
 +    buf_rw = (mach_vm_address_t)tcg_ctx->code_gen_buffer;
 +    buf_rx = 0;
 +    ret = mach_vm_remap(mach_task_self(),
 +                        &buf_rx,
 +                        size,
 +                        0,
 +                        VM_FLAGS_ANYWHERE,
 +                        mach_task_self(),
 +                        buf_rw,
 +                        false,
 +                        &cur_prot,
 +                        &max_prot,
 +                        VM_INHERIT_NONE);
 +    if (ret != KERN_SUCCESS) {
 +        /* TODO: Convert "ret" to a human readable error message. */
 +        error_setg(errp, "vm_remap for jit splitwx failed");
 +        munmap((void *)buf_rw, size);
 +        return false;
 +    }
 +
 +    if (mprotect((void *)buf_rx, size, PROT_READ | PROT_EXEC) != 0) {
 +        error_setg_errno(errp, errno, "mprotect for jit splitwx");
 +        munmap((void *)buf_rx, size);
 +        munmap((void *)buf_rw, size);
 +        return false;
 +    }
 +
 +    tcg_splitwx_diff = buf_rx - buf_rw;
 +    return true;
 +}
 +#endif /* CONFIG_DARWIN */
 +#endif /* CONFIG_TCG_INTERPRETER */
 +
 +static bool alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
 +{
 +#ifndef CONFIG_TCG_INTERPRETER
 +# ifdef CONFIG_DARWIN
 +    return alloc_code_gen_buffer_splitwx_vmremap(size, errp);
 +# endif
 +# ifdef CONFIG_POSIX
 +    return alloc_code_gen_buffer_splitwx_memfd(size, errp);
 +# endif
 +#endif
 +    error_setg(errp, "jit split-wx not supported");
 +    return false;
 +}
 +
 +static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
 +{
 +    ERRP_GUARD();
 +    int prot, flags;
 +
 +    if (splitwx) {
 +        if (alloc_code_gen_buffer_splitwx(size, errp)) {
 +            return true;
 +        }
 +        /*
 +         * If splitwx force-on (1), fail;
 +         * if splitwx default-on (-1), fall through to splitwx off.
 +         */
 +        if (splitwx > 0) {
 +            return false;
 +        }
 +        error_free_or_abort(errp);
 +    }
 +
 +    prot = PROT_READ | PROT_WRITE | PROT_EXEC;
 +    flags = MAP_PRIVATE | MAP_ANONYMOUS;
 +#ifdef CONFIG_TCG_INTERPRETER
 +    /* The tcg interpreter does not need execute permission. */
 +    prot = PROT_READ | PROT_WRITE;
 +#elif defined(CONFIG_DARWIN)
 +    /* Applicable to both iOS and macOS (Apple Silicon). */
 +    if (!splitwx) {
 +        flags |= MAP_JIT;
 +    }
 +#endif
 +
 +    return alloc_code_gen_buffer_anon(size, prot, flags, errp);
 +}
 +#endif /* USE_STATIC_CODE_GEN_BUFFER, WIN32, POSIX */
 +
  /*
   * Initializes region partitioning.
   *
@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(void)
   * in practice. Multi-threaded guests share most if not all of their translated
   * code, which makes parallel code generation less appealing than in softmmu.
   */
 -void tcg_region_init(void)
 +void tcg_region_init(size_t tb_size, int splitwx)
  {
 -    void *buf = tcg_init_ctx.code_gen_buffer;
 -    void *aligned;
 -    size_t size = tcg_init_ctx.code_gen_buffer_size;
 -    size_t page_size = qemu_real_host_page_size;
 +    void *buf, *aligned;
 +    size_t size;
 +    size_t page_size;
      size_t region_size;
      size_t n_regions;
      size_t i;
 +    bool ok;
 +    ok = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
 +                               splitwx, &error_fatal);
 +    assert(ok);
 +
 +    buf = tcg_init_ctx.code_gen_buffer;
 +    size = tcg_init_ctx.code_gen_buffer_size;
 +    page_size = qemu_real_host_page_size;
      n_regions = tcg_n_regions();
      /* The first region will be 'aligned - buf' bytes larger than the others */
 --
 .25.1

-[PULL 31/34] tcg: Fix documentation for tcg_constant_* vs tcg_temp_free_*
+[PULL 11/24] target/microblaze: Check CF_NO_GOTO_TB for DISAS_JUMP
-At some point during the development of tcg_constant_*, I changed
+We were using singlestep_enabled as a proxy for whether
-my mind about whether such temps should be able to be passed to
+translator_use_goto_tb would always return false.
 tcg_temp_free_*.  The final version committed allows this, but the
 commentary was not updated to match.
-Fixes: c0522136adf
-Reported-by: Peter Maydell <peter.maydell@linaro.org>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- include/tcg/tcg.h | 3 ++-
+ target/microblaze/translate.c | 4 ++--
-file changed, 2 insertions(+), 1 deletion(-)
+file changed, 2 insertions(+), 2 deletions(-)
-diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
+diff --git a/target/microblaze/translate.c b/target/microblaze/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/include/tcg/tcg.h
+--- a/target/microblaze/translate.c
-+++ b/include/tcg/tcg.h
++++ b/target/microblaze/translate.c
-@@ -XXX,XX +XXX,XX @@ TCGv_vec tcg_const_ones_vec_matching(TCGv_vec);
+@@ -XXX,XX +XXX,XX @@ static void mb_tr_tb_stop(DisasContextBase *dcb, CPUState *cs)
+         break;
- /*
-  * Locate or create a read-only temporary that is a constant.
+     case DISAS_JUMP:
-- * This kind of temporary need not and should not be freed.
+-        if (dc->jmp_dest != -1 && !cs->singlestep_enabled) {
-+ * This kind of temporary need not be freed, but for convenience
++        if (dc->jmp_dest != -1 && !(tb_cflags(dc->base.tb) & CF_NO_GOTO_TB)) {
-+ * will be silently ignored by tcg_temp_free_*.
+             /* Direct jump. */
-  */
+             tcg_gen_discard_i32(cpu_btarget);
- TCGTemp *tcg_constant_internal(TCGType type, int64_t val);
@@ -XXX,XX +XXX,XX @@ static void mb_tr_tb_stop(DisasContextBase *dcb, CPUState *cs)
              return;
          }
 -        /* Indirect jump (or direct jump w/ singlestep) */
 +        /* Indirect jump (or direct jump w/ goto_tb disabled) */
          tcg_gen_mov_i32(cpu_pc, cpu_btarget);
          tcg_gen_discard_i32(cpu_btarget);
 --
 .25.1

-[PULL 32/34] tcg/arm: Fix tcg_out_op function signature
+[PULL 12/24] target/microblaze: Drop checks for singlestep_enabled
-From: "Jose R. Ziviani" <jziviani@suse.de>
+GDB single-stepping is now handled generically.
-Commit 5e8892db93 fixed several function signatures but tcg_out_op for
-arm is missing. This patch fixes it as well.
-Signed-off-by: Jose R. Ziviani <jziviani@suse.de>
-Message-Id: <20210610224450.23425-1-jziviani@suse.de>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/arm/tcg-target.c.inc | 3 ++-
+ target/microblaze/translate.c | 14 ++------------
-file changed, 2 insertions(+), 1 deletion(-)
+file changed, 2 insertions(+), 12 deletions(-)
-diff --git a/tcg/arm/tcg-target.c.inc b/tcg/arm/tcg-target.c.inc
+diff --git a/target/microblaze/translate.c b/target/microblaze/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/arm/tcg-target.c.inc
+--- a/target/microblaze/translate.c
-+++ b/tcg/arm/tcg-target.c.inc
++++ b/target/microblaze/translate.c
-@@ -XXX,XX +XXX,XX @@ static void tcg_out_qemu_st(TCGContext *s, const TCGArg *args, bool is64)
+@@ -XXX,XX +XXX,XX @@ static void gen_raise_hw_excp(DisasContext *dc, uint32_t esr_ec)
- static void tcg_out_epilogue(TCGContext *s);
+ static void gen_goto_tb(DisasContext *dc, int n, target_ulong dest)
  static inline void tcg_out_op(TCGContext *s, TCGOpcode opc,
 -                const TCGArg *args, const int *const_args)
 +                const TCGArg args[TCG_MAX_OP_ARGS],
 +                const int const_args[TCG_MAX_OP_ARGS])
  {
-     TCGArg a0, a1, a2, a3, a4, a5;
+-    if (dc->base.singlestep_enabled) {
-     int c;
+-        TCGv_i32 tmp = tcg_const_i32(EXCP_DEBUG);
 -        tcg_gen_movi_i32(cpu_pc, dest);
 -        gen_helper_raise_exception(cpu_env, tmp);
 -        tcg_temp_free_i32(tmp);
 -    } else if (translator_use_goto_tb(&dc->base, dest)) {
 +    if (translator_use_goto_tb(&dc->base, dest)) {
          tcg_gen_goto_tb(n);
          tcg_gen_movi_i32(cpu_pc, dest);
          tcg_gen_exit_tb(dc->base.tb, n);
@@ -XXX,XX +XXX,XX @@ static void mb_tr_tb_stop(DisasContextBase *dcb, CPUState *cs)
          /* Indirect jump (or direct jump w/ goto_tb disabled) */
          tcg_gen_mov_i32(cpu_pc, cpu_btarget);
          tcg_gen_discard_i32(cpu_btarget);
 -
 -        if (unlikely(cs->singlestep_enabled)) {
 -            gen_raise_exception(dc, EXCP_DEBUG);
 -        } else {
 -            tcg_gen_lookup_and_goto_ptr();
 -        }
 +        tcg_gen_lookup_and_goto_ptr();
          return;
      default:
 --
 .25.1

-[PULL 26/34] tcg: Round the tb_size default from qemu_get_host_physmem
+[PULL 13/24] target/mips: Fix single stepping
-If qemu_get_host_physmem returns an odd number of pages,
+As per an ancient comment in mips_tr_translate_insn about the
-then physmem / 8 will not be a multiple of the page size.
+expectations of gdb, when restarting the insn in a delay slot
 we also re-execute the branch.  Which means that we are
 expected to execute two insns in this case.
-The following was observed on a gitlab runner:
+This has been broken since 8b86d6d2580, where we forced max_insns
 to 1 while single-stepping.  This resulted in an exit from the
 translator loop after the branch but before the delay slot is
 translated.
-ERROR qtest-arm/boot-serial-test - Bail out!
+Increase the max_insns to 2 for this case.  In addition, bypass
-ERROR:../util/osdep.c:80:qemu_mprotect__osdep: \
+the end-of-page check, for when the branch itself ends the page.
   assertion failed: (!(size & ~qemu_real_host_page_mask))
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
+Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/region.c | 47 +++++++++++++++++++++--------------------------
+ target/mips/tcg/translate.c | 25 ++++++++++++++++---------
-file changed, 21 insertions(+), 26 deletions(-)
+file changed, 16 insertions(+), 9 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
+diff --git a/target/mips/tcg/translate.c b/target/mips/tcg/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
+--- a/target/mips/tcg/translate.c
-+++ b/tcg/region.c
++++ b/target/mips/tcg/translate.c
-@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(size_t tb_size, unsigned max_cpus)
+@@ -XXX,XX +XXX,XX @@ static void mips_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cs)
-   (DEFAULT_CODE_GEN_BUFFER_SIZE_1 < MAX_CODE_GEN_BUFFER_SIZE \
+     ctx->default_tcg_memop_mask = (ctx->insn_flags & (ISA_MIPS_R6 |
-    ? DEFAULT_CODE_GEN_BUFFER_SIZE_1 : MAX_CODE_GEN_BUFFER_SIZE)
+                                   INSN_LOONGSON3A)) ? MO_UNALN : MO_ALIGN;
--static size_t size_code_gen_buffer(size_t tb_size)
++    /*
--{
++     * Execute a branch and its delay slot as a single instruction.
--    /* Size the buffer.  */
++     * This is what GDB expects and is consistent with what the
--    if (tb_size == 0) {
++     * hardware does (e.g. if a delay slot instruction faults, the
--        size_t phys_mem = qemu_get_host_physmem();
++     * reported PC is the PC of the branch).
--        if (phys_mem == 0) {
++     */
--            tb_size = DEFAULT_CODE_GEN_BUFFER_SIZE;
++    if (ctx->base.singlestep_enabled && (ctx->hflags & MIPS_HFLAG_BMASK)) {
--        } else {
++        ctx->base.max_insns = 2;
 -            tb_size = MIN(DEFAULT_CODE_GEN_BUFFER_SIZE, phys_mem / 8);
 -        }
 -    }
 -    if (tb_size < MIN_CODE_GEN_BUFFER_SIZE) {
 -        tb_size = MIN_CODE_GEN_BUFFER_SIZE;
 -    }
 -    if (tb_size > MAX_CODE_GEN_BUFFER_SIZE) {
 -        tb_size = MAX_CODE_GEN_BUFFER_SIZE;
 -    }
 -    return tb_size;
 -}
 -
  #ifdef __mips__
  /*
   * In order to use J and JAL within the code_gen_buffer, we require
@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
   */
  void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
  {
 -    size_t page_size;
 +    const size_t page_size = qemu_real_host_page_size;
      size_t region_size;
      size_t i;
      int have_prot;
 -    have_prot = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
 -                                      splitwx, &error_fatal);
 +    /* Size the buffer.  */
 +    if (tb_size == 0) {
 +        size_t phys_mem = qemu_get_host_physmem();
 +        if (phys_mem == 0) {
 +            tb_size = DEFAULT_CODE_GEN_BUFFER_SIZE;
 +        } else {
 +            tb_size = QEMU_ALIGN_DOWN(phys_mem / 8, page_size);
 +            tb_size = MIN(DEFAULT_CODE_GEN_BUFFER_SIZE, tb_size);
 +        }
 +    }
 +    if (tb_size < MIN_CODE_GEN_BUFFER_SIZE) {
 +        tb_size = MIN_CODE_GEN_BUFFER_SIZE;
 +    }
 +    if (tb_size > MAX_CODE_GEN_BUFFER_SIZE) {
 +        tb_size = MAX_CODE_GEN_BUFFER_SIZE;
 +    }
 +
-+    have_prot = alloc_code_gen_buffer(tb_size, splitwx, &error_fatal);
+     LOG_DISAS("\ntb %p idx %d hflags %04x\n", ctx->base.tb, ctx->mem_idx,
-     assert(have_prot >= 0);
+               ctx->hflags);
+ }
-     /* Request large pages for the buffer and the splitwx.  */
+@@ -XXX,XX +XXX,XX @@ static void mips_tr_translate_insn(DisasContextBase *dcbase, CPUState *cs)
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
+     if (ctx->base.is_jmp != DISAS_NEXT) {
-      * As a result of this we might end up with a few extra pages at the end of
+         return;
-      * the buffer; we will assign those to the last region.
+     }
 +
      /*
 -     * Execute a branch and its delay slot as a single instruction.
 -     * This is what GDB expects and is consistent with what the
 -     * hardware does (e.g. if a delay slot instruction faults, the
 -     * reported PC is the PC of the branch).
 +     * End the TB on (most) page crossings.
 +     * See mips_tr_init_disas_context about single-stepping a branch
 +     * together with its delay slot.
       */
--    region.n = tcg_n_regions(region.total_size, max_cpus);
+-    if (ctx->base.singlestep_enabled &&
--    page_size = qemu_real_host_page_size;
+-        (ctx->hflags & MIPS_HFLAG_BMASK) == 0) {
--    region_size = region.total_size / region.n;
+-        ctx->base.is_jmp = DISAS_TOO_MANY;
-+    region.n = tcg_n_regions(tb_size, max_cpus);
+-    }
-+    region_size = tb_size / region.n;
+-    if (ctx->base.pc_next - ctx->page_start >= TARGET_PAGE_SIZE) {
-     region_size = QEMU_ALIGN_DOWN(region_size, page_size);
++    if (ctx->base.pc_next - ctx->page_start >= TARGET_PAGE_SIZE
++        && !ctx->base.singlestep_enabled) {
-     /* A region must have at least 2 pages; one code, one guard */
+         ctx->base.is_jmp = DISAS_TOO_MANY;
      }
  }
 --
 .25.1

-[PULL 11/34] tcg: Create tcg_init
+[PULL 14/24] target/mips: Drop exit checks for singlestep_enabled
-Perform both tcg_context_init and tcg_region_init.
+GDB single-stepping is now handled generically.
 Do not leave this split to the caller.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
+Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- include/tcg/tcg.h         | 3 +--
+ target/mips/tcg/translate.c | 50 +++++++++++++------------------------
- tcg/tcg-internal.h        | 1 +
+file changed, 18 insertions(+), 32 deletions(-)
  accel/tcg/translate-all.c | 3 +--
  tcg/tcg.c                 | 9 ++++++++-
 files changed, 11 insertions(+), 5 deletions(-)
-diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
+diff --git a/target/mips/tcg/translate.c b/target/mips/tcg/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/include/tcg/tcg.h
+--- a/target/mips/tcg/translate.c
-+++ b/include/tcg/tcg.h
++++ b/target/mips/tcg/translate.c
-@@ -XXX,XX +XXX,XX @@ void *tcg_malloc_internal(TCGContext *s, int size);
+@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
- void tcg_pool_reset(TCGContext *s);
+         tcg_gen_exit_tb(ctx->base.tb, n);
- TranslationBlock *tcg_tb_alloc(TCGContext *s);
+     } else {
+         gen_save_pc(dest);
--void tcg_region_init(size_t tb_size, int splitwx);
+-        if (ctx->base.singlestep_enabled) {
- void tb_destroy(TranslationBlock *tb);
+-            save_cpu_state(ctx, 0);
- void tcg_region_reset_all(void);
+-            gen_helper_raise_exception_debug(cpu_env);
+-        } else {
-@@ -XXX,XX +XXX,XX @@ static inline void *tcg_malloc(int size)
+-            tcg_gen_lookup_and_goto_ptr();
 -        }
 +        tcg_gen_lookup_and_goto_ptr();
      }
  }
--void tcg_context_init(TCGContext *s);
+@@ -XXX,XX +XXX,XX @@ static void gen_branch(DisasContext *ctx, int insn_bytes)
-+void tcg_init(size_t tb_size, int splitwx);
+             } else {
- void tcg_register_thread(void);
+                 tcg_gen_mov_tl(cpu_PC, btarget);
- void tcg_prologue_init(TCGContext *s);
+             }
- void tcg_func_start(TCGContext *s);
+-            if (ctx->base.singlestep_enabled) {
-diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
+-                save_cpu_state(ctx, 0);
-index XXXXXXX..XXXXXXX 100644
+-                gen_helper_raise_exception_debug(cpu_env);
---- a/tcg/tcg-internal.h
+-            }
-+++ b/tcg/tcg-internal.h
+             tcg_gen_lookup_and_goto_ptr();
-@@ -XXX,XX +XXX,XX @@
+             break;
- extern TCGContext **tcg_ctxs;
+         default:
- extern unsigned int n_tcg_ctxs;
+@@ -XXX,XX +XXX,XX @@ static void mips_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
 +void tcg_region_init(size_t tb_size, int splitwx);
  bool tcg_region_alloc(TCGContext *s);
  void tcg_region_initial_alloc(TCGContext *s);
  void tcg_region_prologue_set(TCGContext *s);
 diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
 index XXXXXXX..XXXXXXX 100644
 --- a/accel/tcg/translate-all.c
 +++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@ static void tb_htable_init(void)
  void tcg_exec_init(unsigned long tb_size, int splitwx)
  {
-     tcg_allowed = true;
+     DisasContext *ctx = container_of(dcbase, DisasContext, base);
--    tcg_context_init(&tcg_init_ctx);
-     page_init();
+-    if (ctx->base.singlestep_enabled && ctx->base.is_jmp != DISAS_NORETURN) {
-     tb_htable_init();
+-        save_cpu_state(ctx, ctx->base.is_jmp != DISAS_EXIT);
--    tcg_region_init(tb_size, splitwx);
+-        gen_helper_raise_exception_debug(cpu_env);
-+    tcg_init(tb_size, splitwx);
+-    } else {
+-        switch (ctx->base.is_jmp) {
- #if defined(CONFIG_SOFTMMU)
+-        case DISAS_STOP:
-     /* There's no guest base to take into account, so go ahead and
+-            gen_save_pc(ctx->base.pc_next);
-diff --git a/tcg/tcg.c b/tcg/tcg.c
+-            tcg_gen_lookup_and_goto_ptr();
-index XXXXXXX..XXXXXXX 100644
+-            break;
---- a/tcg/tcg.c
+-        case DISAS_NEXT:
-+++ b/tcg/tcg.c
+-        case DISAS_TOO_MANY:
-@@ -XXX,XX +XXX,XX @@ static void process_op_defs(TCGContext *s);
+-            save_cpu_state(ctx, 0);
- static TCGTemp *tcg_global_reg_new_internal(TCGContext *s, TCGType type,
+-            gen_goto_tb(ctx, 0, ctx->base.pc_next);
-                                             TCGReg reg, const char *name);
+-            break;
+-        case DISAS_EXIT:
--void tcg_context_init(TCGContext *s)
+-            tcg_gen_exit_tb(NULL, 0);
-+static void tcg_context_init(void)
+-            break;
- {
+-        case DISAS_NORETURN:
-+    TCGContext *s = &tcg_init_ctx;
+-            break;
-     int op, total_args, n, i;
+-        default:
-     TCGOpDef *def;
+-            g_assert_not_reached();
-     TCGArgConstraint *args_ct;
+-        }
-@@ -XXX,XX +XXX,XX @@ void tcg_context_init(TCGContext *s)
++    switch (ctx->base.is_jmp) {
-     cpu_env = temp_tcgv_ptr(ts);
++    case DISAS_STOP:
 +        gen_save_pc(ctx->base.pc_next);
 +        tcg_gen_lookup_and_goto_ptr();
 +        break;
 +    case DISAS_NEXT:
 +    case DISAS_TOO_MANY:
 +        save_cpu_state(ctx, 0);
 +        gen_goto_tb(ctx, 0, ctx->base.pc_next);
 +        break;
 +    case DISAS_EXIT:
 +        tcg_gen_exit_tb(NULL, 0);
 +        break;
 +    case DISAS_NORETURN:
 +        break;
 +    default:
 +        g_assert_not_reached();
      }
  }
-+void tcg_init(size_t tb_size, int splitwx)
-+{
-+    tcg_context_init();
-+    tcg_region_init(tb_size, splitwx);
-+}
-+
- /*
-  * Allocate TBs right before their corresponding translated code, making
-  * sure that TBs and code are on different cache lines.
 --
 .25.1

-[PULL 13/34] accel/tcg: Use MiB in tcg_init_machine
+[PULL 15/24] target/openrisc: Drop checks for singlestep_enabled
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
+GDB single-stepping is now handled generically.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
 Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- accel/tcg/tcg-all.c | 3 ++-
+ target/openrisc/translate.c | 18 +++---------------
-file changed, 2 insertions(+), 1 deletion(-)
+file changed, 3 insertions(+), 15 deletions(-)
-diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
+diff --git a/target/openrisc/translate.c b/target/openrisc/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/accel/tcg/tcg-all.c
+--- a/target/openrisc/translate.c
-+++ b/accel/tcg/tcg-all.c
++++ b/target/openrisc/translate.c
-@@ -XXX,XX +XXX,XX @@
+@@ -XXX,XX +XXX,XX @@ static void openrisc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
- #include "qemu/error-report.h"
+             /* The jump destination is indirect/computed; use jmp_pc.  */
- #include "qemu/accel.h"
+             tcg_gen_mov_tl(cpu_pc, jmp_pc);
- #include "qapi/qapi-builtin-visit.h"
+             tcg_gen_discard_tl(jmp_pc);
-+#include "qemu/units.h"
+-            if (unlikely(dc->base.singlestep_enabled)) {
- #include "internal.h"
+-                gen_exception(dc, EXCP_DEBUG);
+-            } else {
- struct TCGState {
+-                tcg_gen_lookup_and_goto_ptr();
-@@ -XXX,XX +XXX,XX @@ static int tcg_init_machine(MachineState *ms)
+-            }
++            tcg_gen_lookup_and_goto_ptr();
-     page_init();
+             break;
-     tb_htable_init();
+         }
--    tcg_init(s->tb_size * 1024 * 1024, s->splitwx_enabled);
+         /* The jump destination is direct; use jmp_pc_imm.
-+    tcg_init(s->tb_size * MiB, s->splitwx_enabled);
+@@ -XXX,XX +XXX,XX @@ static void openrisc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
+             break;
- #if defined(CONFIG_SOFTMMU)
+         }
-     /*
+         tcg_gen_movi_tl(cpu_pc, jmp_dest);
 -        if (unlikely(dc->base.singlestep_enabled)) {
 -            gen_exception(dc, EXCP_DEBUG);
 -        } else {
 -            tcg_gen_lookup_and_goto_ptr();
 -        }
 +        tcg_gen_lookup_and_goto_ptr();
          break;
      case DISAS_EXIT:
 -        if (unlikely(dc->base.singlestep_enabled)) {
 -            gen_exception(dc, EXCP_DEBUG);
 -        } else {
 -            tcg_gen_exit_tb(NULL, 0);
 -        }
 +        tcg_gen_exit_tb(NULL, 0);
          break;
      default:
          g_assert_not_reached();
 --
 .25.1

-[PULL 27/34] tcg: Merge buffer protection and guard page protection
+[PULL 16/24] target/ppc: Drop exit checks for singlestep_enabled
-Do not handle protections on a case-by-case basis in the
+GDB single-stepping is now handled generically.
-various alloc_code_gen_buffer instances; do it within a
+Reuse gen_debug_exception to handle architectural debug exceptions.
 single loop in tcg_region_init.
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/region.c | 45 +++++++++++++++++++++++++++++++--------------
+ target/ppc/translate.c | 38 ++++++++------------------------------
-file changed, 31 insertions(+), 14 deletions(-)
+file changed, 8 insertions(+), 30 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
+diff --git a/target/ppc/translate.c b/target/ppc/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
+--- a/target/ppc/translate.c
-+++ b/tcg/region.c
++++ b/target/ppc/translate.c
-@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
+@@ -XXX,XX +XXX,XX @@
  #define CPU_SINGLE_STEP 0x1
  #define CPU_BRANCH_STEP 0x2
 -#define GDBSTUB_SINGLE_STEP 0x4
  /* Include definitions for instructions classes and implementations flags */
  /* #define PPC_DEBUG_DISAS */
@@ -XXX,XX +XXX,XX @@ static uint32_t gen_prep_dbgex(DisasContext *ctx)
  static void gen_debug_exception(DisasContext *ctx)
  {
 -    gen_helper_raise_exception(cpu_env, tcg_constant_i32(EXCP_DEBUG));
 +    gen_helper_raise_exception(cpu_env, tcg_constant_i32(gen_prep_dbgex(ctx)));
      ctx->base.is_jmp = DISAS_NORETURN;
  }
@@ -XXX,XX +XXX,XX @@ static inline bool use_goto_tb(DisasContext *ctx, target_ulong dest)
  static void gen_lookup_and_goto_ptr(DisasContext *ctx)
  {
 -    int sse = ctx->singlestep_enabled;
 -    if (unlikely(sse)) {
 -        if (sse & GDBSTUB_SINGLE_STEP) {
 -            gen_debug_exception(ctx);
 -        } else if (sse & (CPU_SINGLE_STEP | CPU_BRANCH_STEP)) {
 -            gen_helper_raise_exception(cpu_env, tcg_constant_i32(gen_prep_dbgex(ctx)));
 -        } else {
 -            tcg_gen_exit_tb(NULL, 0);
 -        }
 +    if (unlikely(ctx->singlestep_enabled)) {
 +        gen_debug_exception(ctx);
      } else {
          tcg_gen_lookup_and_goto_ptr();
      }
- #endif
+@@ -XXX,XX +XXX,XX @@ static void ppc_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cs)
+     ctx->singlestep_enabled = 0;
--    if (qemu_mprotect_rwx(buf, size)) {
+     if ((hflags >> HFLAGS_SE) & 1) {
--        error_setg_errno(errp, errno, "mprotect of jit buffer");
+         ctx->singlestep_enabled |= CPU_SINGLE_STEP;
--        return false;
++        ctx->base.max_insns = 1;
      }
      if ((hflags >> HFLAGS_BE) & 1) {
          ctx->singlestep_enabled |= CPU_BRANCH_STEP;
      }
 -    if (unlikely(ctx->base.singlestep_enabled)) {
 -        ctx->singlestep_enabled |= GDBSTUB_SINGLE_STEP;
 -    }
 -
-     region.start_aligned = buf;
+-    if (ctx->singlestep_enabled & (CPU_SINGLE_STEP | GDBSTUB_SINGLE_STEP)) {
-     region.total_size = size;
+-        ctx->base.max_insns = 1;
+-    }
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
+ }
- {
-     const size_t page_size = qemu_real_host_page_size;
+ static void ppc_tr_tb_start(DisasContextBase *db, CPUState *cs)
-     size_t region_size;
+@@ -XXX,XX +XXX,XX @@ static void ppc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
--    size_t i;
+     DisasContext *ctx = container_of(dcbase, DisasContext, base);
--    int have_prot;
+     DisasJumpType is_jmp = ctx->base.is_jmp;
-+    int have_prot, need_prot;
+     target_ulong nip = ctx->base.pc_next;
+-    int sse;
-     /* Size the buffer.  */
-     if (tb_size == 0) {
+     if (is_jmp == DISAS_NORETURN) {
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
+         /* We have already exited the TB. */
-      * Set guard pages in the rw buffer, as that's the one into which
+@@ -XXX,XX +XXX,XX @@ static void ppc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
       * buffer overruns could occur.  Do not set guard pages in the rx
       * buffer -- let that one use hugepages throughout.
 +     * Work with the page protections set up with the initial mapping.
       */
 -    for (i = 0; i < region.n; i++) {
 +    need_prot = PAGE_READ | PAGE_WRITE;
 +#ifndef CONFIG_TCG_INTERPRETER
 +    if (tcg_splitwx_diff == 0) {
 +        need_prot |= PAGE_EXEC;
 +    }
 +#endif
 +    for (size_t i = 0, n = region.n; i < n; i++) {
          void *start, *end;
          tcg_region_bounds(i, &start, &end);
 +        if (have_prot != need_prot) {
 +            int rc;
 -        /*
 -         * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
 -         * rejects a permission change from RWX -> NONE.  Guard pages are
 -         * nice for bug detection but are not essential; ignore any failure.
 -         */
 -        (void)qemu_mprotect_none(end, page_size);
 +            if (need_prot == (PAGE_READ | PAGE_WRITE | PAGE_EXEC)) {
 +                rc = qemu_mprotect_rwx(start, end - start);
 +            } else if (need_prot == (PAGE_READ | PAGE_WRITE)) {
 +                rc = qemu_mprotect_rw(start, end - start);
 +            } else {
 +                g_assert_not_reached();
 +            }
 +            if (rc) {
 +                error_setg_errno(&error_fatal, errno,
 +                                 "mprotect of jit buffer");
 +            }
 +        }
 +        if (have_prot != 0) {
 +            /*
 +             * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
 +             * rejects a permission change from RWX -> NONE.  Guard pages are
 +             * nice for bug detection but are not essential; ignore any failure.
 +             */
 +            (void)qemu_mprotect_none(end, page_size);
 +        }
      }
-     tcg_region_trees_init();
+     /* Honor single stepping. */
 -    sse = ctx->singlestep_enabled & (CPU_SINGLE_STEP | GDBSTUB_SINGLE_STEP);
 -    if (unlikely(sse)) {
 +    if (unlikely(ctx->singlestep_enabled & CPU_SINGLE_STEP)
 +        && (nip <= 0x100 || nip > 0xf00)) {
          switch (is_jmp) {
          case DISAS_TOO_MANY:
          case DISAS_EXIT_UPDATE:
@@ -XXX,XX +XXX,XX @@ static void ppc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
              g_assert_not_reached();
          }
 -        if (sse & GDBSTUB_SINGLE_STEP) {
 -            gen_debug_exception(ctx);
 -            return;
 -        }
 -        /* else CPU_SINGLE_STEP... */
 -        if (nip <= 0x100 || nip > 0xf00) {
 -            gen_helper_raise_exception(cpu_env, tcg_constant_i32(gen_prep_dbgex(ctx)));
 -            return;
 -        }
 +        gen_debug_exception(ctx);
 +        return;
      }
      switch (is_jmp) {
 --
 .25.1

-[PULL 20/34] tcg: Tidy split_cross_256mb
+[PULL 17/24] target/riscv: Remove dead code after exception
-Return output buffer and size via output pointer arguments,
+We have already set DISAS_NORETURN in generate_exception,
-rather than returning size via tcg_ctx->code_gen_buffer_size.
+which makes the exit_tb unreachable.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
+Reviewed-by: Alistair Francis <alistair.francis@wdc.com>
 Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/region.c | 19 +++++++++----------
+ target/riscv/insn_trans/trans_privileged.c.inc | 6 +-----
-file changed, 9 insertions(+), 10 deletions(-)
+file changed, 1 insertion(+), 5 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
+diff --git a/target/riscv/insn_trans/trans_privileged.c.inc b/target/riscv/insn_trans/trans_privileged.c.inc
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
+--- a/target/riscv/insn_trans/trans_privileged.c.inc
-+++ b/tcg/region.c
++++ b/target/riscv/insn_trans/trans_privileged.c.inc
-@@ -XXX,XX +XXX,XX @@ static inline bool cross_256mb(void *addr, size_t size)
+@@ -XXX,XX +XXX,XX @@ static bool trans_ecall(DisasContext *ctx, arg_ecall *a)
  /*
   * We weren't able to allocate a buffer without crossing that boundary,
   * so make do with the larger portion of the buffer that doesn't cross.
 - * Returns the new base of the buffer, and adjusts code_gen_buffer_size.
 + * Returns the new base and size of the buffer in *obuf and *osize.
   */
 -static inline void *split_cross_256mb(void *buf1, size_t size1)
 +static inline void split_cross_256mb(void **obuf, size_t *osize,
 +                                     void *buf1, size_t size1)
  {
-     void *buf2 = (void *)(((uintptr_t)buf1 + size1) & ~0x0ffffffful);
+     /* always generates U-level ECALL, fixed in do_interrupt handler */
-     size_t size2 = buf1 + size1 - buf2;
+     generate_exception(ctx, RISCV_EXCP_U_ECALL);
-@@ -XXX,XX +XXX,XX @@ static inline void *split_cross_256mb(void *buf1, size_t size1)
+-    exit_tb(ctx); /* no chaining */
-         buf1 = buf2;
+-    ctx->base.is_jmp = DISAS_NORETURN;
      }
 -    tcg_ctx->code_gen_buffer_size = size1;
 -    return buf1;
 +    *obuf = buf1;
 +    *osize = size1;
  }
  #endif
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
      if (size > tb_size) {
          size = QEMU_ALIGN_DOWN(tb_size, qemu_real_host_page_size);
      }
 -    tcg_ctx->code_gen_buffer_size = size;
  #ifdef __mips__
      if (cross_256mb(buf, size)) {
 -        buf = split_cross_256mb(buf, size);
 -        size = tcg_ctx->code_gen_buffer_size;
 +        split_cross_256mb(&buf, &size, buf, size);
      }
  #endif
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
      qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
      tcg_ctx->code_gen_buffer = buf;
 +    tcg_ctx->code_gen_buffer_size = size;
      return true;
  }
- #elif defined(_WIN32)
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
+@@ -XXX,XX +XXX,XX @@ static bool trans_ebreak(DisasContext *ctx, arg_ebreak *a)
-                          "allocate %zu bytes for jit buffer", size);
+         post   = opcode_at(&ctx->base, post_addr);
          return false;
      }
--    tcg_ctx->code_gen_buffer_size = size;
+-    if  (pre == 0x01f01013 && ebreak == 0x00100073 && post == 0x40705013) {
- #ifdef __mips__
++    if (pre == 0x01f01013 && ebreak == 0x00100073 && post == 0x40705013) {
-     if (cross_256mb(buf, size)) {
+         generate_exception(ctx, RISCV_EXCP_SEMIHOST);
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
+     } else {
-             /* fallthru */
+         generate_exception(ctx, RISCV_EXCP_BREAKPOINT);
-         default:
+     }
-             /* Split the original buffer.  Free the smaller half.  */
+-    exit_tb(ctx); /* no chaining */
--            buf2 = split_cross_256mb(buf, size);
+-    ctx->base.is_jmp = DISAS_NORETURN;
 -            size2 = tcg_ctx->code_gen_buffer_size;
 +            split_cross_256mb(&buf2, &size2, buf, size);
              if (buf == buf2) {
                  munmap(buf + size2, size - size2);
              } else {
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
      qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
      tcg_ctx->code_gen_buffer = buf;
 +    tcg_ctx->code_gen_buffer_size = size;
      return true;
  }
 --
 .25.1

-[PULL 07/34] tcg: Split out region.c
+[PULL 18/24] target/riscv: Remove exit_tb and lookup_and_goto_ptr
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
+GDB single-stepping is now handled generically, which means
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
+we don't need to do anything in the wrappers.
 Reviewed-by: Alistair Francis <alistair.francis@wdc.com>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/tcg-internal.h |  37 +++
+ target/riscv/translate.c                      | 27 +------------------
- tcg/region.c       | 572 +++++++++++++++++++++++++++++++++++++++++++++
+ .../riscv/insn_trans/trans_privileged.c.inc   |  4 +--
- tcg/tcg.c          | 547 +------------------------------------------
+ target/riscv/insn_trans/trans_rvi.c.inc       |  8 +++---
- tcg/meson.build    |   1 +
+ target/riscv/insn_trans/trans_rvv.c.inc       |  2 +-
-files changed, 613 insertions(+), 544 deletions(-)
+files changed, 7 insertions(+), 34 deletions(-)
  create mode 100644 tcg/tcg-internal.h
  create mode 100644 tcg/region.c
-diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
+diff --git a/target/riscv/translate.c b/target/riscv/translate.c
 new file mode 100644
 index XXXXXXX..XXXXXXX
 --- /dev/null
 +++ b/tcg/tcg-internal.h
@@ -XXX,XX +XXX,XX @@
 +/*
 + * Internal declarations for Tiny Code Generator for QEMU
 + *
 + * Copyright (c) 2008 Fabrice Bellard
 + *
 + * Permission is hereby granted, free of charge, to any person obtaining a copy
 + * of this software and associated documentation files (the "Software"), to deal
 + * in the Software without restriction, including without limitation the rights
 + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
 + * copies of the Software, and to permit persons to whom the Software is
 + * furnished to do so, subject to the following conditions:
 + *
 + * The above copyright notice and this permission notice shall be included in
 + * all copies or substantial portions of the Software.
 + *
 + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
 + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
 + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
 + * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
 + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
 + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
 + * THE SOFTWARE.
 + */
 +
 +#ifndef TCG_INTERNAL_H
 +#define TCG_INTERNAL_H 1
 +
 +#define TCG_HIGHWATER 1024
 +
 +extern TCGContext **tcg_ctxs;
 +extern unsigned int n_tcg_ctxs;
 +
 +bool tcg_region_alloc(TCGContext *s);
 +void tcg_region_initial_alloc(TCGContext *s);
 +void tcg_region_prologue_set(TCGContext *s);
 +
 +#endif /* TCG_INTERNAL_H */
 diff --git a/tcg/region.c b/tcg/region.c
 new file mode 100644
 index XXXXXXX..XXXXXXX
 --- /dev/null
 +++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@
 +/*
 + * Memory region management for Tiny Code Generator for QEMU
 + *
 + * Copyright (c) 2008 Fabrice Bellard
 + *
 + * Permission is hereby granted, free of charge, to any person obtaining a copy
 + * of this software and associated documentation files (the "Software"), to deal
 + * in the Software without restriction, including without limitation the rights
 + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
 + * copies of the Software, and to permit persons to whom the Software is
 + * furnished to do so, subject to the following conditions:
 + *
 + * The above copyright notice and this permission notice shall be included in
 + * all copies or substantial portions of the Software.
 + *
 + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
 + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
 + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
 + * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
 + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
 + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
 + * THE SOFTWARE.
 + */
 +
 +#include "qemu/osdep.h"
 +#include "exec/exec-all.h"
 +#include "tcg/tcg.h"
 +#if !defined(CONFIG_USER_ONLY)
 +#include "hw/boards.h"
 +#endif
 +#include "tcg-internal.h"
 +
 +
 +struct tcg_region_tree {
 +    QemuMutex lock;
 +    GTree *tree;
 +    /* padding to avoid false sharing is computed at run-time */
 +};
 +
 +/*
 + * We divide code_gen_buffer into equally-sized "regions" that TCG threads
 + * dynamically allocate from as demand dictates. Given appropriate region
 + * sizing, this minimizes flushes even when some TCG threads generate a lot
 + * more code than others.
 + */
 +struct tcg_region_state {
 +    QemuMutex lock;
 +
 +    /* fields set at init time */
 +    void *start;
 +    void *start_aligned;
 +    void *end;
 +    size_t n;
 +    size_t size; /* size of one region */
 +    size_t stride; /* .size + guard size */
 +
 +    /* fields protected by the lock */
 +    size_t current; /* current region index */
 +    size_t agg_size_full; /* aggregate size of full regions */
 +};
 +
 +static struct tcg_region_state region;
 +
 +/*
 + * This is an array of struct tcg_region_tree's, with padding.
 + * We use void * to simplify the computation of region_trees[i]; each
 + * struct is found every tree_size bytes.
 + */
 +static void *region_trees;
 +static size_t tree_size;
 +
 +/* compare a pointer @ptr and a tb_tc @s */
 +static int ptr_cmp_tb_tc(const void *ptr, const struct tb_tc *s)
 +{
 +    if (ptr >= s->ptr + s->size) {
 +        return 1;
 +    } else if (ptr < s->ptr) {
 +        return -1;
 +    }
 +    return 0;
 +}
 +
 +static gint tb_tc_cmp(gconstpointer ap, gconstpointer bp)
 +{
 +    const struct tb_tc *a = ap;
 +    const struct tb_tc *b = bp;
 +
 +    /*
 +     * When both sizes are set, we know this isn't a lookup.
 +     * This is the most likely case: every TB must be inserted; lookups
 +     * are a lot less frequent.
 +     */
 +    if (likely(a->size && b->size)) {
 +        if (a->ptr > b->ptr) {
 +            return 1;
 +        } else if (a->ptr < b->ptr) {
 +            return -1;
 +        }
 +        /* a->ptr == b->ptr should happen only on deletions */
 +        g_assert(a->size == b->size);
 +        return 0;
 +    }
 +    /*
 +     * All lookups have either .size field set to 0.
 +     * From the glib sources we see that @ap is always the lookup key. However
 +     * the docs provide no guarantee, so we just mark this case as likely.
 +     */
 +    if (likely(a->size == 0)) {
 +        return ptr_cmp_tb_tc(a->ptr, b);
 +    }
 +    return ptr_cmp_tb_tc(b->ptr, a);
 +}
 +
 +static void tcg_region_trees_init(void)
 +{
 +    size_t i;
 +
 +    tree_size = ROUND_UP(sizeof(struct tcg_region_tree), qemu_dcache_linesize);
 +    region_trees = qemu_memalign(qemu_dcache_linesize, region.n * tree_size);
 +    for (i = 0; i < region.n; i++) {
 +        struct tcg_region_tree *rt = region_trees + i * tree_size;
 +
 +        qemu_mutex_init(&rt->lock);
 +        rt->tree = g_tree_new(tb_tc_cmp);
 +    }
 +}
 +
 +static struct tcg_region_tree *tc_ptr_to_region_tree(const void *p)
 +{
 +    size_t region_idx;
 +
 +    /*
 +     * Like tcg_splitwx_to_rw, with no assert.  The pc may come from
 +     * a signal handler over which the caller has no control.
 +     */
 +    if (!in_code_gen_buffer(p)) {
 +        p -= tcg_splitwx_diff;
 +        if (!in_code_gen_buffer(p)) {
 +            return NULL;
 +        }
 +    }
 +
 +    if (p < region.start_aligned) {
 +        region_idx = 0;
 +    } else {
 +        ptrdiff_t offset = p - region.start_aligned;
 +
 +        if (offset > region.stride * (region.n - 1)) {
 +            region_idx = region.n - 1;
 +        } else {
 +            region_idx = offset / region.stride;
 +        }
 +    }
 +    return region_trees + region_idx * tree_size;
 +}
 +
 +void tcg_tb_insert(TranslationBlock *tb)
 +{
 +    struct tcg_region_tree *rt = tc_ptr_to_region_tree(tb->tc.ptr);
 +
 +    g_assert(rt != NULL);
 +    qemu_mutex_lock(&rt->lock);
 +    g_tree_insert(rt->tree, &tb->tc, tb);
 +    qemu_mutex_unlock(&rt->lock);
 +}
 +
 +void tcg_tb_remove(TranslationBlock *tb)
 +{
 +    struct tcg_region_tree *rt = tc_ptr_to_region_tree(tb->tc.ptr);
 +
 +    g_assert(rt != NULL);
 +    qemu_mutex_lock(&rt->lock);
 +    g_tree_remove(rt->tree, &tb->tc);
 +    qemu_mutex_unlock(&rt->lock);
 +}
 +
 +/*
 + * Find the TB 'tb' such that
 + * tb->tc.ptr <= tc_ptr < tb->tc.ptr + tb->tc.size
 + * Return NULL if not found.
 + */
 +TranslationBlock *tcg_tb_lookup(uintptr_t tc_ptr)
 +{
 +    struct tcg_region_tree *rt = tc_ptr_to_region_tree((void *)tc_ptr);
 +    TranslationBlock *tb;
 +    struct tb_tc s = { .ptr = (void *)tc_ptr };
 +
 +    if (rt == NULL) {
 +        return NULL;
 +    }
 +
 +    qemu_mutex_lock(&rt->lock);
 +    tb = g_tree_lookup(rt->tree, &s);
 +    qemu_mutex_unlock(&rt->lock);
 +    return tb;
 +}
 +
 +static void tcg_region_tree_lock_all(void)
 +{
 +    size_t i;
 +
 +    for (i = 0; i < region.n; i++) {
 +        struct tcg_region_tree *rt = region_trees + i * tree_size;
 +
 +        qemu_mutex_lock(&rt->lock);
 +    }
 +}
 +
 +static void tcg_region_tree_unlock_all(void)
 +{
 +    size_t i;
 +
 +    for (i = 0; i < region.n; i++) {
 +        struct tcg_region_tree *rt = region_trees + i * tree_size;
 +
 +        qemu_mutex_unlock(&rt->lock);
 +    }
 +}
 +
 +void tcg_tb_foreach(GTraverseFunc func, gpointer user_data)
 +{
 +    size_t i;
 +
 +    tcg_region_tree_lock_all();
 +    for (i = 0; i < region.n; i++) {
 +        struct tcg_region_tree *rt = region_trees + i * tree_size;
 +
 +        g_tree_foreach(rt->tree, func, user_data);
 +    }
 +    tcg_region_tree_unlock_all();
 +}
 +
 +size_t tcg_nb_tbs(void)
 +{
 +    size_t nb_tbs = 0;
 +    size_t i;
 +
 +    tcg_region_tree_lock_all();
 +    for (i = 0; i < region.n; i++) {
 +        struct tcg_region_tree *rt = region_trees + i * tree_size;
 +
 +        nb_tbs += g_tree_nnodes(rt->tree);
 +    }
 +    tcg_region_tree_unlock_all();
 +    return nb_tbs;
 +}
 +
 +static gboolean tcg_region_tree_traverse(gpointer k, gpointer v, gpointer data)
 +{
 +    TranslationBlock *tb = v;
 +
 +    tb_destroy(tb);
 +    return FALSE;
 +}
 +
 +static void tcg_region_tree_reset_all(void)
 +{
 +    size_t i;
 +
 +    tcg_region_tree_lock_all();
 +    for (i = 0; i < region.n; i++) {
 +        struct tcg_region_tree *rt = region_trees + i * tree_size;
 +
 +        g_tree_foreach(rt->tree, tcg_region_tree_traverse, NULL);
 +        /* Increment the refcount first so that destroy acts as a reset */
 +        g_tree_ref(rt->tree);
 +        g_tree_destroy(rt->tree);
 +    }
 +    tcg_region_tree_unlock_all();
 +}
 +
 +static void tcg_region_bounds(size_t curr_region, void **pstart, void **pend)
 +{
 +    void *start, *end;
 +
 +    start = region.start_aligned + curr_region * region.stride;
 +    end = start + region.size;
 +
 +    if (curr_region == 0) {
 +        start = region.start;
 +    }
 +    if (curr_region == region.n - 1) {
 +        end = region.end;
 +    }
 +
 +    *pstart = start;
 +    *pend = end;
 +}
 +
 +static void tcg_region_assign(TCGContext *s, size_t curr_region)
 +{
 +    void *start, *end;
 +
 +    tcg_region_bounds(curr_region, &start, &end);
 +
 +    s->code_gen_buffer = start;
 +    s->code_gen_ptr = start;
 +    s->code_gen_buffer_size = end - start;
 +    s->code_gen_highwater = end - TCG_HIGHWATER;
 +}
 +
 +static bool tcg_region_alloc__locked(TCGContext *s)
 +{
 +    if (region.current == region.n) {
 +        return true;
 +    }
 +    tcg_region_assign(s, region.current);
 +    region.current++;
 +    return false;
 +}
 +
 +/*
 + * Request a new region once the one in use has filled up.
 + * Returns true on error.
 + */
 +bool tcg_region_alloc(TCGContext *s)
 +{
 +    bool err;
 +    /* read the region size now; alloc__locked will overwrite it on success */
 +    size_t size_full = s->code_gen_buffer_size;
 +
 +    qemu_mutex_lock(&region.lock);
 +    err = tcg_region_alloc__locked(s);
 +    if (!err) {
 +        region.agg_size_full += size_full - TCG_HIGHWATER;
 +    }
 +    qemu_mutex_unlock(&region.lock);
 +    return err;
 +}
 +
 +/*
 + * Perform a context's first region allocation.
 + * This function does _not_ increment region.agg_size_full.
 + */
 +static void tcg_region_initial_alloc__locked(TCGContext *s)
 +{
 +    bool err = tcg_region_alloc__locked(s);
 +    g_assert(!err);
 +}
 +
 +void tcg_region_initial_alloc(TCGContext *s)
 +{
 +    qemu_mutex_lock(&region.lock);
 +    tcg_region_initial_alloc__locked(s);
 +    qemu_mutex_unlock(&region.lock);
 +}
 +
 +/* Call from a safe-work context */
 +void tcg_region_reset_all(void)
 +{
 +    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 +    unsigned int i;
 +
 +    qemu_mutex_lock(&region.lock);
 +    region.current = 0;
 +    region.agg_size_full = 0;
 +
 +    for (i = 0; i < n_ctxs; i++) {
 +        TCGContext *s = qatomic_read(&tcg_ctxs[i]);
 +        tcg_region_initial_alloc__locked(s);
 +    }
 +    qemu_mutex_unlock(&region.lock);
 +
 +    tcg_region_tree_reset_all();
 +}
 +
 +#ifdef CONFIG_USER_ONLY
 +static size_t tcg_n_regions(void)
 +{
 +    return 1;
 +}
 +#else
 +/*
 + * It is likely that some vCPUs will translate more code than others, so we
 + * first try to set more regions than max_cpus, with those regions being of
 + * reasonable size. If that's not possible we make do by evenly dividing
 + * the code_gen_buffer among the vCPUs.
 + */
 +static size_t tcg_n_regions(void)
 +{
 +    size_t i;
 +
 +    /* Use a single region if all we have is one vCPU thread */
 +#if !defined(CONFIG_USER_ONLY)
 +    MachineState *ms = MACHINE(qdev_get_machine());
 +    unsigned int max_cpus = ms->smp.max_cpus;
 +#endif
 +    if (max_cpus == 1 || !qemu_tcg_mttcg_enabled()) {
 +        return 1;
 +    }
 +
 +    /* Try to have more regions than max_cpus, with each region being >= 2 MB */
 +    for (i = 8; i > 0; i--) {
 +        size_t regions_per_thread = i;
 +        size_t region_size;
 +
 +        region_size = tcg_init_ctx.code_gen_buffer_size;
 +        region_size /= max_cpus * regions_per_thread;
 +
 +        if (region_size >= 2 * 1024u * 1024) {
 +            return max_cpus * regions_per_thread;
 +        }
 +    }
 +    /* If we can't, then just allocate one region per vCPU thread */
 +    return max_cpus;
 +}
 +#endif
 +
 +/*
 + * Initializes region partitioning.
 + *
 + * Called at init time from the parent thread (i.e. the one calling
 + * tcg_context_init), after the target's TCG globals have been set.
 + *
 + * Region partitioning works by splitting code_gen_buffer into separate regions,
 + * and then assigning regions to TCG threads so that the threads can translate
 + * code in parallel without synchronization.
 + *
 + * In softmmu the number of TCG threads is bounded by max_cpus, so we use at
 + * least max_cpus regions in MTTCG. In !MTTCG we use a single region.
 + * Note that the TCG options from the command-line (i.e. -accel accel=tcg,[...])
 + * must have been parsed before calling this function, since it calls
 + * qemu_tcg_mttcg_enabled().
 + *
 + * In user-mode we use a single region.  Having multiple regions in user-mode
 + * is not supported, because the number of vCPU threads (recall that each thread
 + * spawned by the guest corresponds to a vCPU thread) is only bounded by the
 + * OS, and usually this number is huge (tens of thousands is not uncommon).
 + * Thus, given this large bound on the number of vCPU threads and the fact
 + * that code_gen_buffer is allocated at compile-time, we cannot guarantee
 + * that the availability of at least one region per vCPU thread.
 + *
 + * However, this user-mode limitation is unlikely to be a significant problem
 + * in practice. Multi-threaded guests share most if not all of their translated
 + * code, which makes parallel code generation less appealing than in softmmu.
 + */
 +void tcg_region_init(void)
 +{
 +    void *buf = tcg_init_ctx.code_gen_buffer;
 +    void *aligned;
 +    size_t size = tcg_init_ctx.code_gen_buffer_size;
 +    size_t page_size = qemu_real_host_page_size;
 +    size_t region_size;
 +    size_t n_regions;
 +    size_t i;
 +
 +    n_regions = tcg_n_regions();
 +
 +    /* The first region will be 'aligned - buf' bytes larger than the others */
 +    aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
 +    g_assert(aligned < tcg_init_ctx.code_gen_buffer + size);
 +    /*
 +     * Make region_size a multiple of page_size, using aligned as the start.
 +     * As a result of this we might end up with a few extra pages at the end of
 +     * the buffer; we will assign those to the last region.
 +     */
 +    region_size = (size - (aligned - buf)) / n_regions;
 +    region_size = QEMU_ALIGN_DOWN(region_size, page_size);
 +
 +    /* A region must have at least 2 pages; one code, one guard */
 +    g_assert(region_size >= 2 * page_size);
 +
 +    /* init the region struct */
 +    qemu_mutex_init(&region.lock);
 +    region.n = n_regions;
 +    region.size = region_size - page_size;
 +    region.stride = region_size;
 +    region.start = buf;
 +    region.start_aligned = aligned;
 +    /* page-align the end, since its last page will be a guard page */
 +    region.end = QEMU_ALIGN_PTR_DOWN(buf + size, page_size);
 +    /* account for that last guard page */
 +    region.end -= page_size;
 +
 +    /*
 +     * Set guard pages in the rw buffer, as that's the one into which
 +     * buffer overruns could occur.  Do not set guard pages in the rx
 +     * buffer -- let that one use hugepages throughout.
 +     */
 +    for (i = 0; i < region.n; i++) {
 +        void *start, *end;
 +
 +        tcg_region_bounds(i, &start, &end);
 +
 +        /*
 +         * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
 +         * rejects a permission change from RWX -> NONE.  Guard pages are
 +         * nice for bug detection but are not essential; ignore any failure.
 +         */
 +        (void)qemu_mprotect_none(end, page_size);
 +    }
 +
 +    tcg_region_trees_init();
 +
 +    /*
 +     * Leave the initial context initialized to the first region.
 +     * This will be the context into which we generate the prologue.
 +     * It is also the only context for CONFIG_USER_ONLY.
 +     */
 +    tcg_region_initial_alloc__locked(&tcg_init_ctx);
 +}
 +
 +void tcg_region_prologue_set(TCGContext *s)
 +{
 +    /* Deduct the prologue from the first region.  */
 +    g_assert(region.start == s->code_gen_buffer);
 +    region.start = s->code_ptr;
 +
 +    /* Recompute boundaries of the first region. */
 +    tcg_region_assign(s, 0);
 +
 +    /* Register the balance of the buffer with gdb. */
 +    tcg_register_jit(tcg_splitwx_to_rx(region.start),
 +                     region.end - region.start);
 +}
 +
 +/*
 + * Returns the size (in bytes) of all translated code (i.e. from all regions)
 + * currently in the cache.
 + * See also: tcg_code_capacity()
 + * Do not confuse with tcg_current_code_size(); that one applies to a single
 + * TCG context.
 + */
 +size_t tcg_code_size(void)
 +{
 +    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 +    unsigned int i;
 +    size_t total;
 +
 +    qemu_mutex_lock(&region.lock);
 +    total = region.agg_size_full;
 +    for (i = 0; i < n_ctxs; i++) {
 +        const TCGContext *s = qatomic_read(&tcg_ctxs[i]);
 +        size_t size;
 +
 +        size = qatomic_read(&s->code_gen_ptr) - s->code_gen_buffer;
 +        g_assert(size <= s->code_gen_buffer_size);
 +        total += size;
 +    }
 +    qemu_mutex_unlock(&region.lock);
 +    return total;
 +}
 +
 +/*
 + * Returns the code capacity (in bytes) of the entire cache, i.e. including all
 + * regions.
 + * See also: tcg_code_size()
 + */
 +size_t tcg_code_capacity(void)
 +{
 +    size_t guard_size, capacity;
 +
 +    /* no need for synchronization; these variables are set at init time */
 +    guard_size = region.stride - region.size;
 +    capacity = region.end + guard_size - region.start;
 +    capacity -= region.n * (guard_size + TCG_HIGHWATER);
 +    return capacity;
 +}
 +
 +size_t tcg_tb_phys_invalidate_count(void)
 +{
 +    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 +    unsigned int i;
 +    size_t total = 0;
 +
 +    for (i = 0; i < n_ctxs; i++) {
 +        const TCGContext *s = qatomic_read(&tcg_ctxs[i]);
 +
 +        total += qatomic_read(&s->tb_phys_invalidate_count);
 +    }
 +    return total;
 +}
 diff --git a/tcg/tcg.c b/tcg/tcg.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/tcg.c
+--- a/target/riscv/translate.c
-+++ b/tcg/tcg.c
++++ b/target/riscv/translate.c
-@@ -XXX,XX +XXX,XX @@
+@@ -XXX,XX +XXX,XX @@ static void generate_exception_mtval(DisasContext *ctx, int excp)
+     ctx->base.is_jmp = DISAS_NORETURN;
- #include "elf.h"
+ }
- #include "exec/log.h"
-+#include "tcg-internal.h"
+-static void gen_exception_debug(void)
  /* Forward declarations for functions declared in tcg-target.c.inc and
     used here. */
@@ -XXX,XX +XXX,XX @@ static bool tcg_target_const_match(int64_t val, TCGType type, int ct);
  static int tcg_out_ldst_finalize(TCGContext *s);
  #endif
 -#define TCG_HIGHWATER 1024
 -
 -static TCGContext **tcg_ctxs;
 -static unsigned int n_tcg_ctxs;
 +TCGContext **tcg_ctxs;
 +unsigned int n_tcg_ctxs;
  TCGv_env cpu_env = 0;
  const void *tcg_code_gen_epilogue;
  uintptr_t tcg_splitwx_diff;
@@ -XXX,XX +XXX,XX @@ uintptr_t tcg_splitwx_diff;
  tcg_prologue_fn *tcg_qemu_tb_exec;
  #endif
 -struct tcg_region_tree {
 -    QemuMutex lock;
 -    GTree *tree;
 -    /* padding to avoid false sharing is computed at run-time */
 -};
 -
 -/*
 - * We divide code_gen_buffer into equally-sized "regions" that TCG threads
 - * dynamically allocate from as demand dictates. Given appropriate region
 - * sizing, this minimizes flushes even when some TCG threads generate a lot
 - * more code than others.
 - */
 -struct tcg_region_state {
 -    QemuMutex lock;
 -
 -    /* fields set at init time */
 -    void *start;
 -    void *start_aligned;
 -    void *end;
 -    size_t n;
 -    size_t size; /* size of one region */
 -    size_t stride; /* .size + guard size */
 -
 -    /* fields protected by the lock */
 -    size_t current; /* current region index */
 -    size_t agg_size_full; /* aggregate size of full regions */
 -};
 -
 -static struct tcg_region_state region;
 -/*
 - * This is an array of struct tcg_region_tree's, with padding.
 - * We use void * to simplify the computation of region_trees[i]; each
 - * struct is found every tree_size bytes.
 - */
 -static void *region_trees;
 -static size_t tree_size;
  static TCGRegSet tcg_target_available_regs[TCG_TYPE_COUNT];
  static TCGRegSet tcg_target_call_clobber_regs;
@@ -XXX,XX +XXX,XX @@ static const TCGTargetOpDef constraint_sets[] = {
  #include "tcg-target.c.inc"
 -/* compare a pointer @ptr and a tb_tc @s */
 -static int ptr_cmp_tb_tc(const void *ptr, const struct tb_tc *s)
 -{
--    if (ptr >= s->ptr + s->size) {
+-    gen_helper_raise_exception(cpu_env, tcg_constant_i32(EXCP_DEBUG));
 -        return 1;
 -    } else if (ptr < s->ptr) {
 -        return -1;
 -    }
 -    return 0;
 -}
 -
--static gint tb_tc_cmp(gconstpointer ap, gconstpointer bp)
+-/* Wrapper around tcg_gen_exit_tb that handles single stepping */
 -static void exit_tb(DisasContext *ctx)
 -{
--    const struct tb_tc *a = ap;
+-    if (ctx->base.singlestep_enabled) {
--    const struct tb_tc *b = bp;
+-        gen_exception_debug();
--
+-    } else {
--    /*
+-        tcg_gen_exit_tb(NULL, 0);
 -     * When both sizes are set, we know this isn't a lookup.
 -     * This is the most likely case: every TB must be inserted; lookups
 -     * are a lot less frequent.
 -     */
 -    if (likely(a->size && b->size)) {
 -        if (a->ptr > b->ptr) {
 -            return 1;
 -        } else if (a->ptr < b->ptr) {
 -            return -1;
 -        }
 -        /* a->ptr == b->ptr should happen only on deletions */
 -        g_assert(a->size == b->size);
 -        return 0;
 -    }
 -    /*
 -     * All lookups have either .size field set to 0.
 -     * From the glib sources we see that @ap is always the lookup key. However
 -     * the docs provide no guarantee, so we just mark this case as likely.
 -     */
 -    if (likely(a->size == 0)) {
 -        return ptr_cmp_tb_tc(a->ptr, b);
 -    }
 -    return ptr_cmp_tb_tc(b->ptr, a);
 -}
 -
 -static void tcg_region_trees_init(void)
 -{
 -    size_t i;
 -
 -    tree_size = ROUND_UP(sizeof(struct tcg_region_tree), qemu_dcache_linesize);
 -    region_trees = qemu_memalign(qemu_dcache_linesize, region.n * tree_size);
 -    for (i = 0; i < region.n; i++) {
 -        struct tcg_region_tree *rt = region_trees + i * tree_size;
 -
 -        qemu_mutex_init(&rt->lock);
 -        rt->tree = g_tree_new(tb_tc_cmp);
 -    }
 -}
 -
--static struct tcg_region_tree *tc_ptr_to_region_tree(const void *p)
+-/* Wrapper around tcg_gen_lookup_and_goto_ptr that handles single stepping */
 -static void lookup_and_goto_ptr(DisasContext *ctx)
 -{
--    size_t region_idx;
+-    if (ctx->base.singlestep_enabled) {
--
+-        gen_exception_debug();
 -    /*
 -     * Like tcg_splitwx_to_rw, with no assert.  The pc may come from
 -     * a signal handler over which the caller has no control.
 -     */
 -    if (!in_code_gen_buffer(p)) {
 -        p -= tcg_splitwx_diff;
 -        if (!in_code_gen_buffer(p)) {
 -            return NULL;
 -        }
 -    }
 -
 -    if (p < region.start_aligned) {
 -        region_idx = 0;
 -    } else {
--        ptrdiff_t offset = p - region.start_aligned;
+-        tcg_gen_lookup_and_goto_ptr();
 -
 -        if (offset > region.stride * (region.n - 1)) {
 -            region_idx = region.n - 1;
 -        } else {
 -            region_idx = offset / region.stride;
 -        }
 -    }
 -    return region_trees + region_idx * tree_size;
 -}
 -
 -void tcg_tb_insert(TranslationBlock *tb)
 -{
 -    struct tcg_region_tree *rt = tc_ptr_to_region_tree(tb->tc.ptr);
 -
 -    g_assert(rt != NULL);
 -    qemu_mutex_lock(&rt->lock);
 -    g_tree_insert(rt->tree, &tb->tc, tb);
 -    qemu_mutex_unlock(&rt->lock);
 -}
 -
 -void tcg_tb_remove(TranslationBlock *tb)
 -{
 -    struct tcg_region_tree *rt = tc_ptr_to_region_tree(tb->tc.ptr);
 -
 -    g_assert(rt != NULL);
 -    qemu_mutex_lock(&rt->lock);
 -    g_tree_remove(rt->tree, &tb->tc);
 -    qemu_mutex_unlock(&rt->lock);
 -}
 -
 -/*
 - * Find the TB 'tb' such that
 - * tb->tc.ptr <= tc_ptr < tb->tc.ptr + tb->tc.size
 - * Return NULL if not found.
 - */
 -TranslationBlock *tcg_tb_lookup(uintptr_t tc_ptr)
 -{
 -    struct tcg_region_tree *rt = tc_ptr_to_region_tree((void *)tc_ptr);
 -    TranslationBlock *tb;
 -    struct tb_tc s = { .ptr = (void *)tc_ptr };
 -
 -    if (rt == NULL) {
 -        return NULL;
 -    }
 -
 -    qemu_mutex_lock(&rt->lock);
 -    tb = g_tree_lookup(rt->tree, &s);
 -    qemu_mutex_unlock(&rt->lock);
 -    return tb;
 -}
 -
 -static void tcg_region_tree_lock_all(void)
 -{
 -    size_t i;
 -
 -    for (i = 0; i < region.n; i++) {
 -        struct tcg_region_tree *rt = region_trees + i * tree_size;
 -
 -        qemu_mutex_lock(&rt->lock);
 -    }
 -}
 -
--static void tcg_region_tree_unlock_all(void)
+ static void gen_exception_illegal(DisasContext *ctx)
--{
+ {
--    size_t i;
+     generate_exception(ctx, RISCV_EXCP_ILLEGAL_INST);
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
          tcg_gen_exit_tb(ctx->base.tb, n);
      } else {
          tcg_gen_movi_tl(cpu_pc, dest);
 -        lookup_and_goto_ptr(ctx);
 +        tcg_gen_lookup_and_goto_ptr();
      }
  }
 diff --git a/target/riscv/insn_trans/trans_privileged.c.inc b/target/riscv/insn_trans/trans_privileged.c.inc
 index XXXXXXX..XXXXXXX 100644
 --- a/target/riscv/insn_trans/trans_privileged.c.inc
 +++ b/target/riscv/insn_trans/trans_privileged.c.inc
@@ -XXX,XX +XXX,XX @@ static bool trans_sret(DisasContext *ctx, arg_sret *a)
      if (has_ext(ctx, RVS)) {
          gen_helper_sret(cpu_pc, cpu_env, cpu_pc);
 -        exit_tb(ctx); /* no chaining */
 +        tcg_gen_exit_tb(NULL, 0); /* no chaining */
          ctx->base.is_jmp = DISAS_NORETURN;
      } else {
          return false;
@@ -XXX,XX +XXX,XX @@ static bool trans_mret(DisasContext *ctx, arg_mret *a)
  #ifndef CONFIG_USER_ONLY
      tcg_gen_movi_tl(cpu_pc, ctx->base.pc_next);
      gen_helper_mret(cpu_pc, cpu_env, cpu_pc);
 -    exit_tb(ctx); /* no chaining */
 +    tcg_gen_exit_tb(NULL, 0); /* no chaining */
      ctx->base.is_jmp = DISAS_NORETURN;
      return true;
  #else
 diff --git a/target/riscv/insn_trans/trans_rvi.c.inc b/target/riscv/insn_trans/trans_rvi.c.inc
 index XXXXXXX..XXXXXXX 100644
 --- a/target/riscv/insn_trans/trans_rvi.c.inc
 +++ b/target/riscv/insn_trans/trans_rvi.c.inc
@@ -XXX,XX +XXX,XX @@ static bool trans_jalr(DisasContext *ctx, arg_jalr *a)
      if (a->rd != 0) {
          tcg_gen_movi_tl(cpu_gpr[a->rd], ctx->pc_succ_insn);
      }
 -
--    for (i = 0; i < region.n; i++) {
+-    /* No chaining with JALR. */
--        struct tcg_region_tree *rt = region_trees + i * tree_size;
+-    lookup_and_goto_ptr(ctx);
--
++    tcg_gen_lookup_and_goto_ptr();
--        qemu_mutex_unlock(&rt->lock);
--    }
+     if (misaligned) {
--}
+         gen_set_label(misaligned);
--
+@@ -XXX,XX +XXX,XX @@ static bool trans_fence_i(DisasContext *ctx, arg_fence_i *a)
--void tcg_tb_foreach(GTraverseFunc func, gpointer user_data)
+      * however we need to end the translation block
--{
+      */
--    size_t i;
+     tcg_gen_movi_tl(cpu_pc, ctx->pc_succ_insn);
--
+-    exit_tb(ctx);
--    tcg_region_tree_lock_all();
++    tcg_gen_exit_tb(NULL, 0);
--    for (i = 0; i < region.n; i++) {
+     ctx->base.is_jmp = DISAS_NORETURN;
--        struct tcg_region_tree *rt = region_trees + i * tree_size;
+     return true;
--
+ }
--        g_tree_foreach(rt->tree, func, user_data);
+@@ -XXX,XX +XXX,XX @@ static bool do_csr_post(DisasContext *ctx)
 -    }
 -    tcg_region_tree_unlock_all();
 -}
 -
 -size_t tcg_nb_tbs(void)
 -{
 -    size_t nb_tbs = 0;
 -    size_t i;
 -
 -    tcg_region_tree_lock_all();
 -    for (i = 0; i < region.n; i++) {
 -        struct tcg_region_tree *rt = region_trees + i * tree_size;
 -
 -        nb_tbs += g_tree_nnodes(rt->tree);
 -    }
 -    tcg_region_tree_unlock_all();
 -    return nb_tbs;
 -}
 -
 -static gboolean tcg_region_tree_traverse(gpointer k, gpointer v, gpointer data)
 -{
 -    TranslationBlock *tb = v;
 -
 -    tb_destroy(tb);
 -    return FALSE;
 -}
 -
 -static void tcg_region_tree_reset_all(void)
 -{
 -    size_t i;
 -
 -    tcg_region_tree_lock_all();
 -    for (i = 0; i < region.n; i++) {
 -        struct tcg_region_tree *rt = region_trees + i * tree_size;
 -
 -        g_tree_foreach(rt->tree, tcg_region_tree_traverse, NULL);
 -        /* Increment the refcount first so that destroy acts as a reset */
 -        g_tree_ref(rt->tree);
 -        g_tree_destroy(rt->tree);
 -    }
 -    tcg_region_tree_unlock_all();
 -}
 -
 -static void tcg_region_bounds(size_t curr_region, void **pstart, void **pend)
 -{
 -    void *start, *end;
 -
 -    start = region.start_aligned + curr_region * region.stride;
 -    end = start + region.size;
 -
 -    if (curr_region == 0) {
 -        start = region.start;
 -    }
 -    if (curr_region == region.n - 1) {
 -        end = region.end;
 -    }
 -
 -    *pstart = start;
 -    *pend = end;
 -}
 -
 -static void tcg_region_assign(TCGContext *s, size_t curr_region)
 -{
 -    void *start, *end;
 -
 -    tcg_region_bounds(curr_region, &start, &end);
 -
 -    s->code_gen_buffer = start;
 -    s->code_gen_ptr = start;
 -    s->code_gen_buffer_size = end - start;
 -    s->code_gen_highwater = end - TCG_HIGHWATER;
 -}
 -
 -static bool tcg_region_alloc__locked(TCGContext *s)
 -{
 -    if (region.current == region.n) {
 -        return true;
 -    }
 -    tcg_region_assign(s, region.current);
 -    region.current++;
 -    return false;
 -}
 -
 -/*
 - * Request a new region once the one in use has filled up.
 - * Returns true on error.
 - */
 -static bool tcg_region_alloc(TCGContext *s)
 -{
 -    bool err;
 -    /* read the region size now; alloc__locked will overwrite it on success */
 -    size_t size_full = s->code_gen_buffer_size;
 -
 -    qemu_mutex_lock(&region.lock);
 -    err = tcg_region_alloc__locked(s);
 -    if (!err) {
 -        region.agg_size_full += size_full - TCG_HIGHWATER;
 -    }
 -    qemu_mutex_unlock(&region.lock);
 -    return err;
 -}
 -
 -/*
 - * Perform a context's first region allocation.
 - * This function does _not_ increment region.agg_size_full.
 - */
 -static void tcg_region_initial_alloc__locked(TCGContext *s)
 -{
 -    bool err = tcg_region_alloc__locked(s);
 -    g_assert(!err);
 -}
 -
 -#ifndef CONFIG_USER_ONLY
 -static void tcg_region_initial_alloc(TCGContext *s)
 -{
 -    qemu_mutex_lock(&region.lock);
 -    tcg_region_initial_alloc__locked(s);
 -    qemu_mutex_unlock(&region.lock);
 -}
 -#endif
 -
 -/* Call from a safe-work context */
 -void tcg_region_reset_all(void)
 -{
 -    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 -    unsigned int i;
 -
 -    qemu_mutex_lock(&region.lock);
 -    region.current = 0;
 -    region.agg_size_full = 0;
 -
 -    for (i = 0; i < n_ctxs; i++) {
 -        TCGContext *s = qatomic_read(&tcg_ctxs[i]);
 -        tcg_region_initial_alloc__locked(s);
 -    }
 -    qemu_mutex_unlock(&region.lock);
 -
 -    tcg_region_tree_reset_all();
 -}
 -
 -#ifdef CONFIG_USER_ONLY
 -static size_t tcg_n_regions(void)
 -{
 -    return 1;
 -}
 -#else
 -/*
 - * It is likely that some vCPUs will translate more code than others, so we
 - * first try to set more regions than max_cpus, with those regions being of
 - * reasonable size. If that's not possible we make do by evenly dividing
 - * the code_gen_buffer among the vCPUs.
 - */
 -static size_t tcg_n_regions(void)
 -{
 -    size_t i;
 -
 -    /* Use a single region if all we have is one vCPU thread */
 -#if !defined(CONFIG_USER_ONLY)
 -    MachineState *ms = MACHINE(qdev_get_machine());
 -    unsigned int max_cpus = ms->smp.max_cpus;
 -#endif
 -    if (max_cpus == 1 || !qemu_tcg_mttcg_enabled()) {
 -        return 1;
 -    }
 -
 -    /* Try to have more regions than max_cpus, with each region being >= 2 MB */
 -    for (i = 8; i > 0; i--) {
 -        size_t regions_per_thread = i;
 -        size_t region_size;
 -
 -        region_size = tcg_init_ctx.code_gen_buffer_size;
 -        region_size /= max_cpus * regions_per_thread;
 -
 -        if (region_size >= 2 * 1024u * 1024) {
 -            return max_cpus * regions_per_thread;
 -        }
 -    }
 -    /* If we can't, then just allocate one region per vCPU thread */
 -    return max_cpus;
 -}
 -#endif
 -
 -/*
 - * Initializes region partitioning.
 - *
 - * Called at init time from the parent thread (i.e. the one calling
 - * tcg_context_init), after the target's TCG globals have been set.
 - *
 - * Region partitioning works by splitting code_gen_buffer into separate regions,
 - * and then assigning regions to TCG threads so that the threads can translate
 - * code in parallel without synchronization.
 - *
 - * In softmmu the number of TCG threads is bounded by max_cpus, so we use at
 - * least max_cpus regions in MTTCG. In !MTTCG we use a single region.
 - * Note that the TCG options from the command-line (i.e. -accel accel=tcg,[...])
 - * must have been parsed before calling this function, since it calls
 - * qemu_tcg_mttcg_enabled().
 - *
 - * In user-mode we use a single region.  Having multiple regions in user-mode
 - * is not supported, because the number of vCPU threads (recall that each thread
 - * spawned by the guest corresponds to a vCPU thread) is only bounded by the
 - * OS, and usually this number is huge (tens of thousands is not uncommon).
 - * Thus, given this large bound on the number of vCPU threads and the fact
 - * that code_gen_buffer is allocated at compile-time, we cannot guarantee
 - * that the availability of at least one region per vCPU thread.
 - *
 - * However, this user-mode limitation is unlikely to be a significant problem
 - * in practice. Multi-threaded guests share most if not all of their translated
 - * code, which makes parallel code generation less appealing than in softmmu.
 - */
 -void tcg_region_init(void)
 -{
 -    void *buf = tcg_init_ctx.code_gen_buffer;
 -    void *aligned;
 -    size_t size = tcg_init_ctx.code_gen_buffer_size;
 -    size_t page_size = qemu_real_host_page_size;
 -    size_t region_size;
 -    size_t n_regions;
 -    size_t i;
 -
 -    n_regions = tcg_n_regions();
 -
 -    /* The first region will be 'aligned - buf' bytes larger than the others */
 -    aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
 -    g_assert(aligned < tcg_init_ctx.code_gen_buffer + size);
 -    /*
 -     * Make region_size a multiple of page_size, using aligned as the start.
 -     * As a result of this we might end up with a few extra pages at the end of
 -     * the buffer; we will assign those to the last region.
 -     */
 -    region_size = (size - (aligned - buf)) / n_regions;
 -    region_size = QEMU_ALIGN_DOWN(region_size, page_size);
 -
 -    /* A region must have at least 2 pages; one code, one guard */
 -    g_assert(region_size >= 2 * page_size);
 -
 -    /* init the region struct */
 -    qemu_mutex_init(&region.lock);
 -    region.n = n_regions;
 -    region.size = region_size - page_size;
 -    region.stride = region_size;
 -    region.start = buf;
 -    region.start_aligned = aligned;
 -    /* page-align the end, since its last page will be a guard page */
 -    region.end = QEMU_ALIGN_PTR_DOWN(buf + size, page_size);
 -    /* account for that last guard page */
 -    region.end -= page_size;
 -
 -    /*
 -     * Set guard pages in the rw buffer, as that's the one into which
 -     * buffer overruns could occur.  Do not set guard pages in the rx
 -     * buffer -- let that one use hugepages throughout.
 -     */
 -    for (i = 0; i < region.n; i++) {
 -        void *start, *end;
 -
 -        tcg_region_bounds(i, &start, &end);
 -
 -        /*
 -         * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
 -         * rejects a permission change from RWX -> NONE.  Guard pages are
 -         * nice for bug detection but are not essential; ignore any failure.
 -         */
 -        (void)qemu_mprotect_none(end, page_size);
 -    }
 -
 -    tcg_region_trees_init();
 -
 -    /*
 -     * Leave the initial context initialized to the first region.
 -     * This will be the context into which we generate the prologue.
 -     * It is also the only context for CONFIG_USER_ONLY.
 -     */
 -    tcg_region_initial_alloc__locked(&tcg_init_ctx);
 -}
 -
 -static void tcg_region_prologue_set(TCGContext *s)
 -{
 -    /* Deduct the prologue from the first region.  */
 -    g_assert(region.start == s->code_gen_buffer);
 -    region.start = s->code_ptr;
 -
 -    /* Recompute boundaries of the first region. */
 -    tcg_region_assign(s, 0);
 -
 -    /* Register the balance of the buffer with gdb. */
 -    tcg_register_jit(tcg_splitwx_to_rx(region.start),
 -                     region.end - region.start);
 -}
 -
  #ifdef CONFIG_DEBUG_TCG
  const void *tcg_splitwx_to_rx(void *rw)
  {
-@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
+     /* We may have changed important cpu state -- exit to main loop. */
      tcg_gen_movi_tl(cpu_pc, ctx->pc_succ_insn);
 -    exit_tb(ctx);
 +    tcg_gen_exit_tb(NULL, 0);
      ctx->base.is_jmp = DISAS_NORETURN;
      return true;
  }
- #endif /* !CONFIG_USER_ONLY */
+diff --git a/target/riscv/insn_trans/trans_rvv.c.inc b/target/riscv/insn_trans/trans_rvv.c.inc
 -/*
 - * Returns the size (in bytes) of all translated code (i.e. from all regions)
 - * currently in the cache.
 - * See also: tcg_code_capacity()
 - * Do not confuse with tcg_current_code_size(); that one applies to a single
 - * TCG context.
 - */
 -size_t tcg_code_size(void)
 -{
 -    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 -    unsigned int i;
 -    size_t total;
 -
 -    qemu_mutex_lock(&region.lock);
 -    total = region.agg_size_full;
 -    for (i = 0; i < n_ctxs; i++) {
 -        const TCGContext *s = qatomic_read(&tcg_ctxs[i]);
 -        size_t size;
 -
 -        size = qatomic_read(&s->code_gen_ptr) - s->code_gen_buffer;
 -        g_assert(size <= s->code_gen_buffer_size);
 -        total += size;
 -    }
 -    qemu_mutex_unlock(&region.lock);
 -    return total;
 -}
 -
 -/*
 - * Returns the code capacity (in bytes) of the entire cache, i.e. including all
 - * regions.
 - * See also: tcg_code_size()
 - */
 -size_t tcg_code_capacity(void)
 -{
 -    size_t guard_size, capacity;
 -
 -    /* no need for synchronization; these variables are set at init time */
 -    guard_size = region.stride - region.size;
 -    capacity = region.end + guard_size - region.start;
 -    capacity -= region.n * (guard_size + TCG_HIGHWATER);
 -    return capacity;
 -}
 -
 -size_t tcg_tb_phys_invalidate_count(void)
 -{
 -    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
 -    unsigned int i;
 -    size_t total = 0;
 -
 -    for (i = 0; i < n_ctxs; i++) {
 -        const TCGContext *s = qatomic_read(&tcg_ctxs[i]);
 -
 -        total += qatomic_read(&s->tb_phys_invalidate_count);
 -    }
 -    return total;
 -}
 -
  /* pool based memory allocation */
  void *tcg_malloc_internal(TCGContext *s, int size)
  {
 diff --git a/tcg/meson.build b/tcg/meson.build
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/meson.build
+--- a/target/riscv/insn_trans/trans_rvv.c.inc
-+++ b/tcg/meson.build
++++ b/target/riscv/insn_trans/trans_rvv.c.inc
-@@ -XXX,XX +XXX,XX @@ tcg_ss = ss.source_set()
+@@ -XXX,XX +XXX,XX @@ static bool trans_vsetvl(DisasContext *ctx, arg_vsetvl *a)
+     gen_set_gpr(ctx, a->rd, dst);
- tcg_ss.add(files(
-   'optimize.c',
+     tcg_gen_movi_tl(cpu_pc, ctx->pc_succ_insn);
-+  'region.c',
+-    lookup_and_goto_ptr(ctx);
-   'tcg.c',
++    tcg_gen_lookup_and_goto_ptr();
-   'tcg-common.c',
+     ctx->base.is_jmp = DISAS_NORETURN;
-   'tcg-op.c',
+     return true;
  }
 --
 .25.1

-[PULL 25/34] util/osdep: Add qemu_mprotect_rw
+[PULL 19/24] target/rx: Drop checks for singlestep_enabled
-For --enable-tcg-interpreter on Windows, we will need this.
+GDB single-stepping is now handled generically.
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
 Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- include/qemu/osdep.h | 1 +
+ target/rx/helper.h    |  1 -
- util/osdep.c         | 9 +++++++++
+ target/rx/op_helper.c |  8 --------
-files changed, 10 insertions(+)
+ target/rx/translate.c | 12 ++----------
 files changed, 2 insertions(+), 19 deletions(-)
-diff --git a/include/qemu/osdep.h b/include/qemu/osdep.h
+diff --git a/target/rx/helper.h b/target/rx/helper.h
 index XXXXXXX..XXXXXXX 100644
---- a/include/qemu/osdep.h
+--- a/target/rx/helper.h
-+++ b/include/qemu/osdep.h
++++ b/target/rx/helper.h
-@@ -XXX,XX +XXX,XX @@ void sigaction_invoke(struct sigaction *action,
+@@ -XXX,XX +XXX,XX @@ DEF_HELPER_1(raise_illegal_instruction, noreturn, env)
- #endif
+ DEF_HELPER_1(raise_access_fault, noreturn, env)
+ DEF_HELPER_1(raise_privilege_violation, noreturn, env)
- int qemu_madvise(void *addr, size_t len, int advice);
+ DEF_HELPER_1(wait, noreturn, env)
-+int qemu_mprotect_rw(void *addr, size_t size);
+-DEF_HELPER_1(debug, noreturn, env)
- int qemu_mprotect_rwx(void *addr, size_t size);
+ DEF_HELPER_2(rxint, noreturn, env, i32)
- int qemu_mprotect_none(void *addr, size_t size);
+ DEF_HELPER_1(rxbrk, noreturn, env)
+ DEF_HELPER_FLAGS_3(fadd, TCG_CALL_NO_WG, f32, env, f32, f32)
-diff --git a/util/osdep.c b/util/osdep.c
+diff --git a/target/rx/op_helper.c b/target/rx/op_helper.c
 index XXXXXXX..XXXXXXX 100644
---- a/util/osdep.c
+--- a/target/rx/op_helper.c
-+++ b/util/osdep.c
++++ b/target/rx/op_helper.c
-@@ -XXX,XX +XXX,XX @@ static int qemu_mprotect__osdep(void *addr, size_t size, int prot)
+@@ -XXX,XX +XXX,XX @@ void QEMU_NORETURN helper_wait(CPURXState *env)
- #endif
+     raise_exception(env, EXCP_HLT, 0);
  }
-+int qemu_mprotect_rw(void *addr, size_t size)
+-void QEMU_NORETURN helper_debug(CPURXState *env)
-+{
+-{
-+#ifdef _WIN32
+-    CPUState *cs = env_cpu(env);
-+    return qemu_mprotect__osdep(addr, size, PAGE_READWRITE);
+-
-+#else
+-    cs->exception_index = EXCP_DEBUG;
-+    return qemu_mprotect__osdep(addr, size, PROT_READ | PROT_WRITE);
+-    cpu_loop_exit(cs);
-+#endif
+-}
-+}
+-
-+
+ void QEMU_NORETURN helper_rxint(CPURXState *env, uint32_t vec)
  int qemu_mprotect_rwx(void *addr, size_t size)
  {
- #ifdef _WIN32
+     raise_exception(env, 0x100 + vec, 0);
 diff --git a/target/rx/translate.c b/target/rx/translate.c
 index XXXXXXX..XXXXXXX 100644
 --- a/target/rx/translate.c
 +++ b/target/rx/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *dc, int n, target_ulong dest)
          tcg_gen_exit_tb(dc->base.tb, n);
      } else {
          tcg_gen_movi_i32(cpu_pc, dest);
 -        if (dc->base.singlestep_enabled) {
 -            gen_helper_debug(cpu_env);
 -        } else {
 -            tcg_gen_lookup_and_goto_ptr();
 -        }
 +        tcg_gen_lookup_and_goto_ptr();
      }
      dc->base.is_jmp = DISAS_NORETURN;
  }
@@ -XXX,XX +XXX,XX @@ static void rx_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
          gen_goto_tb(ctx, 0, dcbase->pc_next);
          break;
      case DISAS_JUMP:
 -        if (ctx->base.singlestep_enabled) {
 -            gen_helper_debug(cpu_env);
 -        } else {
 -            tcg_gen_lookup_and_goto_ptr();
 -        }
 +        tcg_gen_lookup_and_goto_ptr();
          break;
      case DISAS_UPDATE:
          tcg_gen_movi_i32(cpu_pc, ctx->base.pc_next);
 --
 .25.1

-[PULL 29/34] tcg: Move tcg_init_ctx and tcg_ctx from accel/tcg/
+[PULL 20/24] target/s390x: Drop check for singlestep_enabled
-These variables belong to the jit side, not the user side.
+GDB single-stepping is now handled generically.
-Since tcg_init_ctx is no longer used outside of tcg/, move
-the declaration to tcg-internal.h.
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Suggested-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- include/tcg/tcg.h         | 1 -
+ target/s390x/tcg/translate.c | 8 ++------
- tcg/tcg-internal.h        | 1 +
+file changed, 2 insertions(+), 6 deletions(-)
  accel/tcg/translate-all.c | 3 ---
  tcg/tcg.c                 | 3 +++
 files changed, 4 insertions(+), 4 deletions(-)
-diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
+diff --git a/target/s390x/tcg/translate.c b/target/s390x/tcg/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/include/tcg/tcg.h
+--- a/target/s390x/tcg/translate.c
-+++ b/include/tcg/tcg.h
++++ b/target/s390x/tcg/translate.c
-@@ -XXX,XX +XXX,XX @@ static inline bool temp_readonly(TCGTemp *ts)
+@@ -XXX,XX +XXX,XX @@ struct DisasContext {
-     return ts->kind >= TEMP_FIXED;
+     uint64_t pc_tmp;
      uint32_t ilen;
      enum cc_op cc_op;
 -    bool do_debug;
  };
  /* Information carried about a condition to be evaluated.  */
@@ -XXX,XX +XXX,XX @@ static void s390x_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cs)
      dc->cc_op = CC_OP_DYNAMIC;
      dc->ex_value = dc->base.tb->cs_base;
 -    dc->do_debug = dc->base.singlestep_enabled;
  }
--extern TCGContext tcg_init_ctx;
+ static void s390x_tr_tb_start(DisasContextBase *db, CPUState *cs)
- extern __thread TCGContext *tcg_ctx;
+@@ -XXX,XX +XXX,XX @@ static void s390x_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
- extern const void *tcg_code_gen_epilogue;
+         /* FALLTHRU */
- extern uintptr_t tcg_splitwx_diff;
+     case DISAS_PC_CC_UPDATED:
-diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
+         /* Exit the TB, either by raising a debug exception or by return.  */
-index XXXXXXX..XXXXXXX 100644
+-        if (dc->do_debug) {
---- a/tcg/tcg-internal.h
+-            gen_exception(EXCP_DEBUG);
-+++ b/tcg/tcg-internal.h
+-        } else if ((dc->base.tb->flags & FLAG_MASK_PER) ||
-@@ -XXX,XX +XXX,XX @@
+-                   dc->base.is_jmp == DISAS_PC_STALE_NOCHAIN) {
++        if ((dc->base.tb->flags & FLAG_MASK_PER) ||
- #define TCG_HIGHWATER 1024
++             dc->base.is_jmp == DISAS_PC_STALE_NOCHAIN) {
+             tcg_gen_exit_tb(NULL, 0);
-+extern TCGContext tcg_init_ctx;
+         } else {
- extern TCGContext **tcg_ctxs;
+             tcg_gen_lookup_and_goto_ptr();
  extern unsigned int tcg_cur_ctxs;
  extern unsigned int tcg_max_ctxs;
 diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
 index XXXXXXX..XXXXXXX 100644
 --- a/accel/tcg/translate-all.c
 +++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@ static int v_l2_levels;
  static void *l1_map[V_L1_MAX_SIZE];
 -/* code generation context */
 -TCGContext tcg_init_ctx;
 -__thread TCGContext *tcg_ctx;
  TBContext tb_ctx;
  static void page_table_config_init(void)
 diff --git a/tcg/tcg.c b/tcg/tcg.c
 index XXXXXXX..XXXXXXX 100644
 --- a/tcg/tcg.c
 +++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ static bool tcg_target_const_match(int64_t val, TCGType type, int ct);
  static int tcg_out_ldst_finalize(TCGContext *s);
  #endif
 +TCGContext tcg_init_ctx;
 +__thread TCGContext *tcg_ctx;
 +
  TCGContext **tcg_ctxs;
  unsigned int tcg_cur_ctxs;
  unsigned int tcg_max_ctxs;
 --
 .25.1

-[PULL 12/34] accel/tcg: Merge tcg_exec_init into tcg_init_machine
+[PULL 21/24] target/sh4: Drop check for singlestep_enabled
-There is only one caller, and shortly we will need access
+GDB single-stepping is now handled generically.
 to the MachineState, which tcg_init_machine already has.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
+Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- accel/tcg/internal.h      |  2 ++
+ target/sh4/helper.h    |  1 -
- include/sysemu/tcg.h      |  2 --
+ target/sh4/op_helper.c |  5 -----
- accel/tcg/tcg-all.c       | 16 +++++++++++++++-
+ target/sh4/translate.c | 14 +++-----------
- accel/tcg/translate-all.c | 21 ++-------------------
+files changed, 3 insertions(+), 17 deletions(-)
  bsd-user/main.c           |  2 +-
 files changed, 20 insertions(+), 23 deletions(-)
-diff --git a/accel/tcg/internal.h b/accel/tcg/internal.h
+diff --git a/target/sh4/helper.h b/target/sh4/helper.h
 index XXXXXXX..XXXXXXX 100644
---- a/accel/tcg/internal.h
+--- a/target/sh4/helper.h
-+++ b/accel/tcg/internal.h
++++ b/target/sh4/helper.h
-@@ -XXX,XX +XXX,XX @@ TranslationBlock *tb_gen_code(CPUState *cpu, target_ulong pc,
+@@ -XXX,XX +XXX,XX @@ DEF_HELPER_1(raise_illegal_instruction, noreturn, env)
-                               int cflags);
+ DEF_HELPER_1(raise_slot_illegal_instruction, noreturn, env)
+ DEF_HELPER_1(raise_fpu_disable, noreturn, env)
- void QEMU_NORETURN cpu_io_recompile(CPUState *cpu, uintptr_t retaddr);
+ DEF_HELPER_1(raise_slot_fpu_disable, noreturn, env)
-+void page_init(void);
+-DEF_HELPER_1(debug, noreturn, env)
-+void tb_htable_init(void);
+ DEF_HELPER_1(sleep, noreturn, env)
+ DEF_HELPER_2(trapa, noreturn, env, i32)
- #endif /* ACCEL_TCG_INTERNAL_H */
+ DEF_HELPER_1(exclusive, noreturn, env)
-diff --git a/include/sysemu/tcg.h b/include/sysemu/tcg.h
+diff --git a/target/sh4/op_helper.c b/target/sh4/op_helper.c
 index XXXXXXX..XXXXXXX 100644
---- a/include/sysemu/tcg.h
+--- a/target/sh4/op_helper.c
-+++ b/include/sysemu/tcg.h
++++ b/target/sh4/op_helper.c
-@@ -XXX,XX +XXX,XX @@
+@@ -XXX,XX +XXX,XX @@ void helper_raise_slot_fpu_disable(CPUSH4State *env)
- #ifndef SYSEMU_TCG_H
+     raise_exception(env, 0x820, 0);
  #define SYSEMU_TCG_H
 -void tcg_exec_init(unsigned long tb_size, int splitwx);
 -
  #ifdef CONFIG_TCG
  extern bool tcg_allowed;
  #define tcg_enabled() (tcg_allowed)
 diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
 index XXXXXXX..XXXXXXX 100644
 --- a/accel/tcg/tcg-all.c
 +++ b/accel/tcg/tcg-all.c
@@ -XXX,XX +XXX,XX @@
  #include "qemu/error-report.h"
  #include "qemu/accel.h"
  #include "qapi/qapi-builtin-visit.h"
 +#include "internal.h"
  struct TCGState {
      AccelState parent_obj;
@@ -XXX,XX +XXX,XX @@ static int tcg_init_machine(MachineState *ms)
  {
      TCGState *s = TCG_STATE(current_accel());
 -    tcg_exec_init(s->tb_size * 1024 * 1024, s->splitwx_enabled);
 +    tcg_allowed = true;
      mttcg_enabled = s->mttcg_enabled;
 +
 +    page_init();
 +    tb_htable_init();
 +    tcg_init(s->tb_size * 1024 * 1024, s->splitwx_enabled);
 +
 +#if defined(CONFIG_SOFTMMU)
 +    /*
 +     * There's no guest base to take into account, so go ahead and
 +     * initialize the prologue now.
 +     */
 +    tcg_prologue_init(tcg_ctx);
 +#endif
 +
      return 0;
  }
-diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
+-void helper_debug(CPUSH4State *env)
 index XXXXXXX..XXXXXXX 100644
 --- a/accel/tcg/translate-all.c
 +++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@ bool cpu_restore_state(CPUState *cpu, uintptr_t host_pc, bool will_exit)
      return false;
  }
 -static void page_init(void)
 +void page_init(void)
  {
      page_size_init();
      page_table_config_init();
@@ -XXX,XX +XXX,XX @@ static bool tb_cmp(const void *ap, const void *bp)
          a->page_addr[1] == b->page_addr[1];
  }
 -static void tb_htable_init(void)
 +void tb_htable_init(void)
  {
      unsigned int mode = QHT_MODE_AUTO_RESIZE;
      qht_init(&tb_ctx.htable, tb_cmp, CODE_GEN_HTABLE_SIZE, mode);
  }
 -/* Must be called before using the QEMU cpus. 'tb_size' is the size
 -   (in bytes) allocated to the translation buffer. Zero means default
 -   size. */
 -void tcg_exec_init(unsigned long tb_size, int splitwx)
 -{
--    tcg_allowed = true;
+-    raise_exception(env, EXCP_DEBUG, 0);
 -    page_init();
 -    tb_htable_init();
 -    tcg_init(tb_size, splitwx);
 -
 -#if defined(CONFIG_SOFTMMU)
 -    /* There's no guest base to take into account, so go ahead and
 -       initialize the prologue now.  */
 -    tcg_prologue_init(tcg_ctx);
 -#endif
 -}
 -
- /* call with @p->lock held */
+ void helper_sleep(CPUSH4State *env)
  static inline void invalidate_page_bitmap(PageDesc *p)
  {
-diff --git a/bsd-user/main.c b/bsd-user/main.c
+     CPUState *cs = env_cpu(env);
 diff --git a/target/sh4/translate.c b/target/sh4/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/bsd-user/main.c
+--- a/target/sh4/translate.c
-+++ b/bsd-user/main.c
++++ b/target/sh4/translate.c
-@@ -XXX,XX +XXX,XX @@ int main(int argc, char **argv)
+@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
-     envlist_free(envlist);
+         tcg_gen_exit_tb(ctx->base.tb, n);
+     } else {
-     /*
+         tcg_gen_movi_i32(cpu_pc, dest);
--     * Now that page sizes are configured in tcg_exec_init() we can do
+-        if (ctx->base.singlestep_enabled) {
-+     * Now that page sizes are configured we can do
+-            gen_helper_debug(cpu_env);
-      * proper page alignment for guest_base.
+-        } else if (use_exit_tb(ctx)) {
-      */
++        if (use_exit_tb(ctx)) {
-     guest_base = HOST_PAGE_ALIGN(guest_base);
+             tcg_gen_exit_tb(NULL, 0);
          } else {
              tcg_gen_lookup_and_goto_ptr();
@@ -XXX,XX +XXX,XX @@ static void gen_jump(DisasContext * ctx)
         delayed jump as immediate jump are conditinal jumps */
      tcg_gen_mov_i32(cpu_pc, cpu_delayed_pc);
          tcg_gen_discard_i32(cpu_delayed_pc);
 -        if (ctx->base.singlestep_enabled) {
 -            gen_helper_debug(cpu_env);
 -        } else if (use_exit_tb(ctx)) {
 +        if (use_exit_tb(ctx)) {
              tcg_gen_exit_tb(NULL, 0);
          } else {
              tcg_gen_lookup_and_goto_ptr();
@@ -XXX,XX +XXX,XX @@ static void sh4_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
      switch (ctx->base.is_jmp) {
      case DISAS_STOP:
          gen_save_cpu_state(ctx, true);
 -        if (ctx->base.singlestep_enabled) {
 -            gen_helper_debug(cpu_env);
 -        } else {
 -            tcg_gen_exit_tb(NULL, 0);
 -        }
 +        tcg_gen_exit_tb(NULL, 0);
          break;
      case DISAS_NEXT:
      case DISAS_TOO_MANY:
 --
 .25.1

-[PULL 08/34] accel/tcg: Inline cpu_gen_init
+[PULL 22/24] target/tricore: Drop check for singlestep_enabled
-It consists of one function call and has only one caller.
+GDB single-stepping is now handled generically.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
 Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- accel/tcg/translate-all.c | 7 +------
+ target/tricore/helper.h    |  1 -
-file changed, 1 insertion(+), 6 deletions(-)
+ target/tricore/op_helper.c |  7 -------
  target/tricore/translate.c | 14 +-------------
 files changed, 1 insertion(+), 21 deletions(-)
-diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
+diff --git a/target/tricore/helper.h b/target/tricore/helper.h
 index XXXXXXX..XXXXXXX 100644
---- a/accel/tcg/translate-all.c
+--- a/target/tricore/helper.h
-+++ b/accel/tcg/translate-all.c
++++ b/target/tricore/helper.h
-@@ -XXX,XX +XXX,XX @@ static void page_table_config_init(void)
+@@ -XXX,XX +XXX,XX @@ DEF_HELPER_2(psw_write, void, env, i32)
-     assert(v_l2_levels >= 0);
+ DEF_HELPER_1(psw_read, i32, env)
  /* Exceptions */
  DEF_HELPER_3(raise_exception_sync, noreturn, env, i32, i32)
 -DEF_HELPER_2(qemu_excp, noreturn, env, i32)
 diff --git a/target/tricore/op_helper.c b/target/tricore/op_helper.c
 index XXXXXXX..XXXXXXX 100644
 --- a/target/tricore/op_helper.c
 +++ b/target/tricore/op_helper.c
@@ -XXX,XX +XXX,XX @@ static void raise_exception_sync_helper(CPUTriCoreState *env, uint32_t class,
      raise_exception_sync_internal(env, class, tin, pc, 0);
  }
--static void cpu_gen_init(void)
+-void helper_qemu_excp(CPUTriCoreState *env, uint32_t excp)
 -{
--    tcg_context_init(&tcg_init_ctx);
+-    CPUState *cs = env_cpu(env);
 -    cs->exception_index = excp;
 -    cpu_loop_exit(cs);
 -}
 -
- /* Encode VAL as a signed leb128 sequence at P.
+ /* Addressing mode helper */
-    Return P incremented past the encoded value.  */
- static uint8_t *encode_sleb128(uint8_t *p, target_long val)
+ static uint16_t reverse16(uint16_t val)
-@@ -XXX,XX +XXX,XX @@ void tcg_exec_init(unsigned long tb_size, int splitwx)
+diff --git a/target/tricore/translate.c b/target/tricore/translate.c
-     bool ok;
+index XXXXXXX..XXXXXXX 100644
+--- a/target/tricore/translate.c
-     tcg_allowed = true;
++++ b/target/tricore/translate.c
--    cpu_gen_init();
+@@ -XXX,XX +XXX,XX @@ static inline void gen_save_pc(target_ulong pc)
-+    tcg_context_init(&tcg_init_ctx);
+     tcg_gen_movi_tl(cpu_PC, pc);
-     page_init();
+ }
-     tb_htable_init();
 -static void generate_qemu_excp(DisasContext *ctx, int excp)
 -{
 -    TCGv_i32 tmp = tcg_const_i32(excp);
 -    gen_helper_qemu_excp(cpu_env, tmp);
 -    ctx->base.is_jmp = DISAS_NORETURN;
 -    tcg_temp_free(tmp);
 -}
 -
  static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
  {
      if (translator_use_goto_tb(&ctx->base, dest)) {
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
          tcg_gen_exit_tb(ctx->base.tb, n);
      } else {
          gen_save_pc(dest);
 -        if (ctx->base.singlestep_enabled) {
 -            generate_qemu_excp(ctx, EXCP_DEBUG);
 -        } else {
 -            tcg_gen_lookup_and_goto_ptr();
 -        }
 +        tcg_gen_lookup_and_goto_ptr();
      }
  }
 --
 .25.1

-[PULL 24/34] tcg: Sink qemu_madvise call to common code
+[PULL 23/24] target/xtensa: Drop check for singlestep_enabled
-Move the call out of the N versions of alloc_code_gen_buffer
+GDB single-stepping is now handled generically.
 and into tcg_region_init.
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/region.c | 14 +++++++-------
+ target/xtensa/translate.c | 25 ++++++++-----------------
-file changed, 7 insertions(+), 7 deletions(-)
+file changed, 8 insertions(+), 17 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
+diff --git a/target/xtensa/translate.c b/target/xtensa/translate.c
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
+--- a/target/xtensa/translate.c
-+++ b/tcg/region.c
++++ b/target/xtensa/translate.c
-@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
+@@ -XXX,XX +XXX,XX @@ static void gen_jump_slot(DisasContext *dc, TCGv dest, int slot)
-         error_setg_errno(errp, errno, "mprotect of jit buffer");
+     if (dc->icount) {
-         return false;
+         tcg_gen_mov_i32(cpu_SR[ICOUNT], dc->next_icount);
      }
--    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
+-    if (dc->base.singlestep_enabled) {
+-        gen_exception(dc, EXCP_DEBUG);
-     region.start_aligned = buf;
++    if (dc->op_flags & XTENSA_OP_POSTPROCESS) {
-     region.total_size = size;
++        slot = gen_postprocess(dc, slot);
-@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer_anon(size_t size, int prot,
++    }
 +    if (slot >= 0) {
 +        tcg_gen_goto_tb(slot);
 +        tcg_gen_exit_tb(dc->base.tb, slot);
      } else {
 -        if (dc->op_flags & XTENSA_OP_POSTPROCESS) {
 -            slot = gen_postprocess(dc, slot);
 -        }
 -        if (slot >= 0) {
 -            tcg_gen_goto_tb(slot);
 -            tcg_gen_exit_tb(dc->base.tb, slot);
 -        } else {
 -            tcg_gen_exit_tb(NULL, 0);
 -        }
 +        tcg_gen_exit_tb(NULL, 0);
      }
- #endif
+     dc->base.is_jmp = DISAS_NORETURN;
+ }
--    /* Request large pages for the buffer.  */
+@@ -XXX,XX +XXX,XX @@ static void xtensa_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
--    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
+     case DISAS_NORETURN:
--
+         break;
-     region.start_aligned = buf;
+     case DISAS_TOO_MANY:
-     region.total_size = size;
+-        if (dc->base.singlestep_enabled) {
-     return prot;
+-            tcg_gen_movi_i32(cpu_pc, dc->pc);
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
+-            gen_exception(dc, EXCP_DEBUG);
-     region.total_size = size;
+-        } else {
-     tcg_splitwx_diff = buf_rx - buf_rw;
+-            gen_jumpi(dc, dc->pc, 0);
+-        }
--    /* Request large pages for the buffer and the splitwx.  */
++        gen_jumpi(dc, dc->pc, 0);
--    qemu_madvise(buf_rw, size, QEMU_MADV_HUGEPAGE);
+         break;
--    qemu_madvise(buf_rx, size, QEMU_MADV_HUGEPAGE);
+     default:
-     return PROT_READ | PROT_WRITE;
+         g_assert_not_reached();
   fail_rx:
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
                                        splitwx, &error_fatal);
      assert(have_prot >= 0);
 +    /* Request large pages for the buffer and the splitwx.  */
 +    qemu_madvise(region.start_aligned, region.total_size, QEMU_MADV_HUGEPAGE);
 +    if (tcg_splitwx_diff) {
 +        qemu_madvise(region.start_aligned + tcg_splitwx_diff,
 +                     region.total_size, QEMU_MADV_HUGEPAGE);
 +    }
 +
      /*
       * Make region_size a multiple of page_size, using aligned as the start.
       * As a result of this we might end up with a few extra pages at the end of
 --
 .25.1

-[PULL 06/34] tcg: Split out tcg_region_prologue_set
+[PULL 24/24] Revert "cpu: Move cpu_common_props to hw/core/cpu.c"
-This has only one user, but will make more sense after some
+This reverts commit 1b36e4f5a5de585210ea95f2257839c2312be28f.
 code motion.
-Always leave the tcg_init_ctx initialized to the first region,
+Despite a comment saying why cpu_common_props cannot be placed in
-in preparation for tcg_prologue_init().  This also requires
+a file that is compiled once, it was moved anyway.  Revert that.
 that we don't re-allocate the region for the first cpu, lest
 we hit the assertion for total number of regions allocated .
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
+Since then, Property is not defined in hw/core/cpu.h, so it is now
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
+easier to declare a function to install the properties rather than
 the Property array itself.
 Cc: Eduardo Habkost <ehabkost@redhat.com>
 Suggested-by: Peter Maydell <peter.maydell@linaro.org>
 Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
 ---
- tcg/tcg.c | 37 ++++++++++++++++++++++---------------
+ include/hw/core/cpu.h |  1 +
-file changed, 22 insertions(+), 15 deletions(-)
+ cpu.c                 | 21 +++++++++++++++++++++
  hw/core/cpu-common.c  | 17 +----------------
 files changed, 23 insertions(+), 16 deletions(-)
-diff --git a/tcg/tcg.c b/tcg/tcg.c
+diff --git a/include/hw/core/cpu.h b/include/hw/core/cpu.h
 index XXXXXXX..XXXXXXX 100644
---- a/tcg/tcg.c
+--- a/include/hw/core/cpu.h
-+++ b/tcg/tcg.c
++++ b/include/hw/core/cpu.h
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(void)
+@@ -XXX,XX +XXX,XX @@ void QEMU_NORETURN cpu_abort(CPUState *cpu, const char *fmt, ...)
+     GCC_FMT_ATTR(2, 3);
-     tcg_region_trees_init();
+ /* $(top_srcdir)/cpu.c */
--    /* In user-mode we support only one ctx, so do the initial allocation now */
++void cpu_class_init_props(DeviceClass *dc);
--#ifdef CONFIG_USER_ONLY
+ void cpu_exec_initfn(CPUState *cpu);
--    tcg_region_initial_alloc__locked(tcg_ctx);
+ void cpu_exec_realizefn(CPUState *cpu, Error **errp);
--#endif
+ void cpu_exec_unrealizefn(CPUState *cpu);
 diff --git a/cpu.c b/cpu.c
 index XXXXXXX..XXXXXXX 100644
 --- a/cpu.c
 +++ b/cpu.c
@@ -XXX,XX +XXX,XX @@ void cpu_exec_unrealizefn(CPUState *cpu)
      cpu_list_remove(cpu);
  }
 +static Property cpu_common_props[] = {
 +#ifndef CONFIG_USER_ONLY
 +    /*
-+     * Leave the initial context initialized to the first region.
++     * Create a memory property for softmmu CPU object,
-+     * This will be the context into which we generate the prologue.
++     * so users can wire up its memory. (This can't go in hw/core/cpu.c
-+     * It is also the only context for CONFIG_USER_ONLY.
++     * because that file is compiled only once for both user-mode
 +     * and system builds.) The default if no link is set up is to use
 +     * the system address space.
 +     */
-+    tcg_region_initial_alloc__locked(&tcg_init_ctx);
++    DEFINE_PROP_LINK("memory", CPUState, memory, TYPE_MEMORY_REGION,
 +                     MemoryRegion *),
 +#endif
 +    DEFINE_PROP_BOOL("start-powered-off", CPUState, start_powered_off, false),
 +    DEFINE_PROP_END_OF_LIST(),
 +};
 +
 +void cpu_class_init_props(DeviceClass *dc)
 +{
 +    device_class_set_props(dc, cpu_common_props);
 +}
 +
-+static void tcg_region_prologue_set(TCGContext *s)
+ void cpu_exec_initfn(CPUState *cpu)
-+{
+ {
-+    /* Deduct the prologue from the first region.  */
+     cpu->as = NULL;
-+    g_assert(region.start == s->code_gen_buffer);
+diff --git a/hw/core/cpu-common.c b/hw/core/cpu-common.c
-+    region.start = s->code_ptr;
+index XXXXXXX..XXXXXXX 100644
-+
+--- a/hw/core/cpu-common.c
-+    /* Recompute boundaries of the first region. */
++++ b/hw/core/cpu-common.c
-+    tcg_region_assign(s, 0);
+@@ -XXX,XX +XXX,XX @@ static int64_t cpu_common_get_arch_id(CPUState *cpu)
-+
+     return cpu->cpu_index;
 +    /* Register the balance of the buffer with gdb. */
 +    tcg_register_jit(tcg_splitwx_to_rx(region.start),
 +                     region.end - region.start);
  }
- #ifdef CONFIG_DEBUG_TCG
+-static Property cpu_common_props[] = {
-@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
+-#ifndef CONFIG_USER_ONLY
+-    /* Create a memory property for softmmu CPU object,
-     if (n > 0) {
+-     * so users can wire up its memory. (This can't go in hw/core/cpu.c
-         alloc_tcg_plugin_context(s);
+-     * because that file is compiled only once for both user-mode
-+        tcg_region_initial_alloc(s);
+-     * and system builds.) The default if no link is set up is to use
-     }
+-     * the system address space.
+-     */
-     tcg_ctx = s;
+-    DEFINE_PROP_LINK("memory", CPUState, memory, TYPE_MEMORY_REGION,
--    tcg_region_initial_alloc(s);
+-                     MemoryRegion *),
- }
+-#endif
- #endif /* !CONFIG_USER_ONLY */
+-    DEFINE_PROP_BOOL("start-powered-off", CPUState, start_powered_off, false),
+-    DEFINE_PROP_END_OF_LIST(),
-@@ -XXX,XX +XXX,XX @@ void tcg_prologue_init(TCGContext *s)
+-};
 -
  static void cpu_class_init(ObjectClass *klass, void *data)
  {
-     size_t prologue_size;
+     DeviceClass *dc = DEVICE_CLASS(klass);
+@@ -XXX,XX +XXX,XX @@ static void cpu_class_init(ObjectClass *klass, void *data)
--    /* Put the prologue at the beginning of code_gen_buffer.  */
+     dc->realize = cpu_common_realizefn;
--    tcg_region_assign(s, 0);
+     dc->unrealize = cpu_common_unrealizefn;
-     s->code_ptr = s->code_gen_ptr;
+     dc->reset = cpu_common_reset;
-     s->code_buf = s->code_gen_ptr;
+-    device_class_set_props(dc, cpu_common_props);
-     s->data_gen_ptr = NULL;
++    cpu_class_init_props(dc);
-@@ -XXX,XX +XXX,XX @@ void tcg_prologue_init(TCGContext *s)
+     /*
-                         (uintptr_t)s->code_buf, prologue_size);
+      * Reason: CPUs still need special care by board code: wiring up
- #endif
+      * IRQs, adding reset handlers, halting non-first CPUs, ...
 -    /* Deduct the prologue from the first region.  */
 -    region.start = s->code_ptr;
 -
 -    /* Recompute boundaries of the first region. */
 -    tcg_region_assign(s, 0);
 -
 -    tcg_register_jit(tcg_splitwx_to_rx(region.start),
 -                     region.end - region.start);
 +    tcg_region_prologue_set(s);
  #ifdef DEBUG_DISAS
      if (qemu_loglevel_mask(CPU_LOG_TB_OUT_ASM)) {
 --
 .25.1

-[PULL 10/34] accel/tcg: Rename tcg_init to tcg_init_machine
+Deleted patch
-We shortly want to use tcg_init for something else.
-Since the hook is called init_machine, match that.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- accel/tcg/tcg-all.c | 4 ++--
-file changed, 2 insertions(+), 2 deletions(-)
-diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
-index XXXXXXX..XXXXXXX 100644
---- a/accel/tcg/tcg-all.c
-+++ b/accel/tcg/tcg-all.c
-@@ -XXX,XX +XXX,XX @@ static void tcg_accel_instance_init(Object *obj)
- bool mttcg_enabled;
--static int tcg_init(MachineState *ms)
-+static int tcg_init_machine(MachineState *ms)
- {
-     TCGState *s = TCG_STATE(current_accel());
-@@ -XXX,XX +XXX,XX @@ static void tcg_accel_class_init(ObjectClass *oc, void *data)
- {
-     AccelClass *ac = ACCEL_CLASS(oc);
-     ac->name = "tcg";
--    ac->init_machine = tcg_init;
-+    ac->init_machine = tcg_init_machine;
-     ac->allowed = &tcg_allowed;
-     object_class_property_add_str(oc, "thread",
---
-.25.1

-[PULL 16/34] tcg: Move MAX_CODE_GEN_BUFFER_SIZE to tcg-target.h
+Deleted patch
-Remove the ifdef ladder and move each define into the
-appropriate header file.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- tcg/aarch64/tcg-target.h |  1 +
- tcg/arm/tcg-target.h     |  1 +
- tcg/i386/tcg-target.h    |  2 ++
- tcg/mips/tcg-target.h    |  6 ++++++
- tcg/ppc/tcg-target.h     |  2 ++
- tcg/riscv/tcg-target.h   |  1 +
- tcg/s390/tcg-target.h    |  3 +++
- tcg/sparc/tcg-target.h   |  1 +
- tcg/tci/tcg-target.h     |  1 +
- tcg/region.c             | 33 +++++----------------------------
-files changed, 23 insertions(+), 28 deletions(-)
-diff --git a/tcg/aarch64/tcg-target.h b/tcg/aarch64/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/aarch64/tcg-target.h
-+++ b/tcg/aarch64/tcg-target.h
-@@ -XXX,XX +XXX,XX @@
- #define TCG_TARGET_INSN_UNIT_SIZE  4
- #define TCG_TARGET_TLB_DISPLACEMENT_BITS 24
-+#define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
- #undef TCG_TARGET_STACK_GROWSUP
- typedef enum {
-diff --git a/tcg/arm/tcg-target.h b/tcg/arm/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/arm/tcg-target.h
-+++ b/tcg/arm/tcg-target.h
-@@ -XXX,XX +XXX,XX @@ extern int arm_arch;
- #undef TCG_TARGET_STACK_GROWSUP
- #define TCG_TARGET_INSN_UNIT_SIZE 4
- #define TCG_TARGET_TLB_DISPLACEMENT_BITS 16
-+#define MAX_CODE_GEN_BUFFER_SIZE  UINT32_MAX
- typedef enum {
-     TCG_REG_R0 = 0,
-diff --git a/tcg/i386/tcg-target.h b/tcg/i386/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/i386/tcg-target.h
-+++ b/tcg/i386/tcg-target.h
-@@ -XXX,XX +XXX,XX @@
- #ifdef __x86_64__
- # define TCG_TARGET_REG_BITS  64
- # define TCG_TARGET_NB_REGS   32
-+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
- #else
- # define TCG_TARGET_REG_BITS  32
- # define TCG_TARGET_NB_REGS   24
-+# define MAX_CODE_GEN_BUFFER_SIZE  UINT32_MAX
- #endif
- typedef enum {
-diff --git a/tcg/mips/tcg-target.h b/tcg/mips/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/mips/tcg-target.h
-+++ b/tcg/mips/tcg-target.h
-@@ -XXX,XX +XXX,XX @@
- #define TCG_TARGET_TLB_DISPLACEMENT_BITS 16
- #define TCG_TARGET_NB_REGS 32
-+/*
-+ * We have a 256MB branch region, but leave room to make sure the
-+ * main executable is also within that region.
-+ */
-+#define MAX_CODE_GEN_BUFFER_SIZE  (128 * MiB)
-+
- typedef enum {
-     TCG_REG_ZERO = 0,
-     TCG_REG_AT,
-diff --git a/tcg/ppc/tcg-target.h b/tcg/ppc/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/ppc/tcg-target.h
-+++ b/tcg/ppc/tcg-target.h
-@@ -XXX,XX +XXX,XX @@
- #ifdef _ARCH_PPC64
- # define TCG_TARGET_REG_BITS  64
-+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
- #else
- # define TCG_TARGET_REG_BITS  32
-+# define MAX_CODE_GEN_BUFFER_SIZE  (32 * MiB)
- #endif
- #define TCG_TARGET_NB_REGS 64
-diff --git a/tcg/riscv/tcg-target.h b/tcg/riscv/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/riscv/tcg-target.h
-+++ b/tcg/riscv/tcg-target.h
-@@ -XXX,XX +XXX,XX @@
- #define TCG_TARGET_INSN_UNIT_SIZE 4
- #define TCG_TARGET_TLB_DISPLACEMENT_BITS 20
- #define TCG_TARGET_NB_REGS 32
-+#define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
- typedef enum {
-     TCG_REG_ZERO,
-diff --git a/tcg/s390/tcg-target.h b/tcg/s390/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/s390/tcg-target.h
-+++ b/tcg/s390/tcg-target.h
-@@ -XXX,XX +XXX,XX @@
- #define TCG_TARGET_INSN_UNIT_SIZE 2
- #define TCG_TARGET_TLB_DISPLACEMENT_BITS 19
-+/* We have a +- 4GB range on the branches; leave some slop.  */
-+#define MAX_CODE_GEN_BUFFER_SIZE  (3 * GiB)
-+
- typedef enum TCGReg {
-     TCG_REG_R0 = 0,
-     TCG_REG_R1,
-diff --git a/tcg/sparc/tcg-target.h b/tcg/sparc/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/sparc/tcg-target.h
-+++ b/tcg/sparc/tcg-target.h
-@@ -XXX,XX +XXX,XX @@
- #define TCG_TARGET_INSN_UNIT_SIZE 4
- #define TCG_TARGET_TLB_DISPLACEMENT_BITS 32
- #define TCG_TARGET_NB_REGS 32
-+#define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
- typedef enum {
-     TCG_REG_G0 = 0,
-diff --git a/tcg/tci/tcg-target.h b/tcg/tci/tcg-target.h
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/tci/tcg-target.h
-+++ b/tcg/tci/tcg-target.h
-@@ -XXX,XX +XXX,XX @@
- #define TCG_TARGET_INTERPRETER 1
- #define TCG_TARGET_INSN_UNIT_SIZE 1
- #define TCG_TARGET_TLB_DISPLACEMENT_BITS 32
-+#define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
- #if UINTPTR_MAX == UINT32_MAX
- # define TCG_TARGET_REG_BITS 32
-diff --git a/tcg/region.c b/tcg/region.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
-+++ b/tcg/region.c
-@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(unsigned max_cpus)
- /*
-  * Minimum size of the code gen buffer.  This number is randomly chosen,
-  * but not so small that we can't have a fair number of TB's live.
-+ *
-+ * Maximum size, MAX_CODE_GEN_BUFFER_SIZE, is defined in tcg-target.h.
-+ * Unless otherwise indicated, this is constrained by the range of
-+ * direct branches on the host cpu, as used by the TCG implementation
-+ * of goto_tb.
-  */
- #define MIN_CODE_GEN_BUFFER_SIZE     (1 * MiB)
--/*
-- * Maximum size of the code gen buffer we'd like to use.  Unless otherwise
-- * indicated, this is constrained by the range of direct branches on the
-- * host cpu, as used by the TCG implementation of goto_tb.
-- */
--#if defined(__x86_64__)
--# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
--#elif defined(__sparc__)
--# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
--#elif defined(__powerpc64__)
--# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
--#elif defined(__powerpc__)
--# define MAX_CODE_GEN_BUFFER_SIZE  (32 * MiB)
--#elif defined(__aarch64__)
--# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
--#elif defined(__s390x__)
--  /* We have a +- 4GB range on the branches; leave some slop.  */
--# define MAX_CODE_GEN_BUFFER_SIZE  (3 * GiB)
--#elif defined(__mips__)
--  /*
--   * We have a 256MB branch region, but leave room to make sure the
--   * main executable is also within that region.
--   */
--# define MAX_CODE_GEN_BUFFER_SIZE  (128 * MiB)
--#else
--# define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
--#endif
--
- #if TCG_TARGET_REG_BITS == 32
- #define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (32 * MiB)
- #ifdef CONFIG_USER_ONLY
---
-.25.1

-[PULL 17/34] tcg: Replace region.end with region.total_size
+Deleted patch
-A size is easier to work with than an end point,
-particularly during initial buffer allocation.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- tcg/region.c | 30 ++++++++++++++++++------------
-file changed, 18 insertions(+), 12 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
-+++ b/tcg/region.c
-@@ -XXX,XX +XXX,XX @@ struct tcg_region_state {
-     /* fields set at init time */
-     void *start;
-     void *start_aligned;
--    void *end;
-     size_t n;
-     size_t size; /* size of one region */
-     size_t stride; /* .size + guard size */
-+    size_t total_size; /* size of entire buffer, >= n * stride */
-     /* fields protected by the lock */
-     size_t current; /* current region index */
-@@ -XXX,XX +XXX,XX @@ static void tcg_region_bounds(size_t curr_region, void **pstart, void **pend)
-     if (curr_region == 0) {
-         start = region.start;
-     }
-+    /* The final region may have a few extra pages due to earlier rounding. */
-     if (curr_region == region.n - 1) {
--        end = region.end;
-+        end = region.start_aligned + region.total_size;
-     }
-     *pstart = start;
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
-  */
- void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
- {
--    void *buf, *aligned;
--    size_t size;
-+    void *buf, *aligned, *end;
-+    size_t total_size;
-     size_t page_size;
-     size_t region_size;
-     size_t n_regions;
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
-     assert(ok);
-     buf = tcg_init_ctx.code_gen_buffer;
--    size = tcg_init_ctx.code_gen_buffer_size;
-+    total_size = tcg_init_ctx.code_gen_buffer_size;
-     page_size = qemu_real_host_page_size;
-     n_regions = tcg_n_regions(max_cpus);
-     /* The first region will be 'aligned - buf' bytes larger than the others */
-     aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
--    g_assert(aligned < tcg_init_ctx.code_gen_buffer + size);
-+    g_assert(aligned < tcg_init_ctx.code_gen_buffer + total_size);
-+
-     /*
-      * Make region_size a multiple of page_size, using aligned as the start.
-      * As a result of this we might end up with a few extra pages at the end of
-      * the buffer; we will assign those to the last region.
-      */
--    region_size = (size - (aligned - buf)) / n_regions;
-+    region_size = (total_size - (aligned - buf)) / n_regions;
-     region_size = QEMU_ALIGN_DOWN(region_size, page_size);
-     /* A region must have at least 2 pages; one code, one guard */
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
-     region.start = buf;
-     region.start_aligned = aligned;
-     /* page-align the end, since its last page will be a guard page */
--    region.end = QEMU_ALIGN_PTR_DOWN(buf + size, page_size);
-+    end = QEMU_ALIGN_PTR_DOWN(buf + total_size, page_size);
-     /* account for that last guard page */
--    region.end -= page_size;
-+    end -= page_size;
-+    total_size = end - aligned;
-+    region.total_size = total_size;
-     /*
-      * Set guard pages in the rw buffer, as that's the one into which
-@@ -XXX,XX +XXX,XX @@ void tcg_region_prologue_set(TCGContext *s)
-     /* Register the balance of the buffer with gdb. */
-     tcg_register_jit(tcg_splitwx_to_rx(region.start),
--                     region.end - region.start);
-+                     region.start_aligned + region.total_size - region.start);
- }
- /*
-@@ -XXX,XX +XXX,XX @@ size_t tcg_code_capacity(void)
-     /* no need for synchronization; these variables are set at init time */
-     guard_size = region.stride - region.size;
--    capacity = region.end + guard_size - region.start;
--    capacity -= region.n * (guard_size + TCG_HIGHWATER);
-+    capacity = region.total_size;
-+    capacity -= (region.n - 1) * guard_size;
-+    capacity -= region.n * TCG_HIGHWATER;
-+
-     return capacity;
- }
---
-.25.1

-[PULL 18/34] tcg: Rename region.start to region.after_prologue
+Deleted patch
-Give the field a name reflecting its actual meaning.
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- tcg/region.c | 15 ++++++++-------
-file changed, 8 insertions(+), 7 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
-+++ b/tcg/region.c
-@@ -XXX,XX +XXX,XX @@ struct tcg_region_state {
-     QemuMutex lock;
-     /* fields set at init time */
--    void *start;
-     void *start_aligned;
-+    void *after_prologue;
-     size_t n;
-     size_t size; /* size of one region */
-     size_t stride; /* .size + guard size */
-@@ -XXX,XX +XXX,XX @@ static void tcg_region_bounds(size_t curr_region, void **pstart, void **pend)
-     end = start + region.size;
-     if (curr_region == 0) {
--        start = region.start;
-+        start = region.after_prologue;
-     }
-     /* The final region may have a few extra pages due to earlier rounding. */
-     if (curr_region == region.n - 1) {
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
-     region.n = n_regions;
-     region.size = region_size - page_size;
-     region.stride = region_size;
--    region.start = buf;
-+    region.after_prologue = buf;
-     region.start_aligned = aligned;
-     /* page-align the end, since its last page will be a guard page */
-     end = QEMU_ALIGN_PTR_DOWN(buf + total_size, page_size);
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
- void tcg_region_prologue_set(TCGContext *s)
- {
-     /* Deduct the prologue from the first region.  */
--    g_assert(region.start == s->code_gen_buffer);
--    region.start = s->code_ptr;
-+    g_assert(region.start_aligned == s->code_gen_buffer);
-+    region.after_prologue = s->code_ptr;
-     /* Recompute boundaries of the first region. */
-     tcg_region_assign(s, 0);
-     /* Register the balance of the buffer with gdb. */
--    tcg_register_jit(tcg_splitwx_to_rx(region.start),
--                     region.start_aligned + region.total_size - region.start);
-+    tcg_register_jit(tcg_splitwx_to_rx(region.after_prologue),
-+                     region.start_aligned + region.total_size -
-+                     region.after_prologue);
- }
- /*
---
-.25.1

-[PULL 21/34] tcg: Move in_code_gen_buffer and tests to region.c
+Deleted patch
-Shortly, the full code_gen_buffer will only be visible
-to region.c, so move in_code_gen_buffer out-of-line.
-Move the debugging versions of tcg_splitwx_to_{rx,rw}
-to region.c as well, so that the compiler gets to see
-the implementation of in_code_gen_buffer.
-This leaves exactly one use of in_code_gen_buffer outside
-of region.c, in cpu_restore_state.  Which, being on the
-exception path, is not performance critical.
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- include/tcg/tcg.h | 11 +----------
- tcg/region.c      | 34 ++++++++++++++++++++++++++++++++++
- tcg/tcg.c         | 23 -----------------------
-files changed, 35 insertions(+), 33 deletions(-)
-diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
-index XXXXXXX..XXXXXXX 100644
---- a/include/tcg/tcg.h
-+++ b/include/tcg/tcg.h
-@@ -XXX,XX +XXX,XX @@ extern const void *tcg_code_gen_epilogue;
- extern uintptr_t tcg_splitwx_diff;
- extern TCGv_env cpu_env;
--static inline bool in_code_gen_buffer(const void *p)
--{
--    const TCGContext *s = &tcg_init_ctx;
--    /*
--     * Much like it is valid to have a pointer to the byte past the
--     * end of an array (so long as you don't dereference it), allow
--     * a pointer to the byte past the end of the code gen buffer.
--     */
--    return (size_t)(p - s->code_gen_buffer) <= s->code_gen_buffer_size;
--}
-+bool in_code_gen_buffer(const void *p);
- #ifdef CONFIG_DEBUG_TCG
- const void *tcg_splitwx_to_rx(void *rw);
-diff --git a/tcg/region.c b/tcg/region.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
-+++ b/tcg/region.c
-@@ -XXX,XX +XXX,XX @@ static struct tcg_region_state region;
- static void *region_trees;
- static size_t tree_size;
-+bool in_code_gen_buffer(const void *p)
-+{
-+    const TCGContext *s = &tcg_init_ctx;
-+    /*
-+     * Much like it is valid to have a pointer to the byte past the
-+     * end of an array (so long as you don't dereference it), allow
-+     * a pointer to the byte past the end of the code gen buffer.
-+     */
-+    return (size_t)(p - s->code_gen_buffer) <= s->code_gen_buffer_size;
-+}
-+
-+#ifdef CONFIG_DEBUG_TCG
-+const void *tcg_splitwx_to_rx(void *rw)
-+{
-+    /* Pass NULL pointers unchanged. */
-+    if (rw) {
-+        g_assert(in_code_gen_buffer(rw));
-+        rw += tcg_splitwx_diff;
-+    }
-+    return rw;
-+}
-+
-+void *tcg_splitwx_to_rw(const void *rx)
-+{
-+    /* Pass NULL pointers unchanged. */
-+    if (rx) {
-+        rx -= tcg_splitwx_diff;
-+        /* Assert that we end with a pointer in the rw region. */
-+        g_assert(in_code_gen_buffer(rx));
-+    }
-+    return (void *)rx;
-+}
-+#endif /* CONFIG_DEBUG_TCG */
-+
- /* compare a pointer @ptr and a tb_tc @s */
- static int ptr_cmp_tb_tc(const void *ptr, const struct tb_tc *s)
- {
-diff --git a/tcg/tcg.c b/tcg/tcg.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/tcg.c
-+++ b/tcg/tcg.c
-@@ -XXX,XX +XXX,XX @@ static const TCGTargetOpDef constraint_sets[] = {
- #include "tcg-target.c.inc"
--#ifdef CONFIG_DEBUG_TCG
--const void *tcg_splitwx_to_rx(void *rw)
--{
--    /* Pass NULL pointers unchanged. */
--    if (rw) {
--        g_assert(in_code_gen_buffer(rw));
--        rw += tcg_splitwx_diff;
--    }
--    return rw;
--}
--
--void *tcg_splitwx_to_rw(const void *rx)
--{
--    /* Pass NULL pointers unchanged. */
--    if (rx) {
--        rx -= tcg_splitwx_diff;
--        /* Assert that we end with a pointer in the rw region. */
--        g_assert(in_code_gen_buffer(rx));
--    }
--    return (void *)rx;
--}
--#endif /* CONFIG_DEBUG_TCG */
--
- static void alloc_tcg_plugin_context(TCGContext *s)
- {
- #ifdef CONFIG_PLUGIN
---
-.25.1

-[PULL 22/34] tcg: Allocate code_gen_buffer into struct tcg_region_state
+Deleted patch
-Do not mess around with setting values within tcg_init_ctx.
-Put the values into 'region' directly, which is where they
-will live for the lifetime of the program.
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- tcg/region.c | 64 ++++++++++++++++++++++------------------------------
-file changed, 27 insertions(+), 37 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
-+++ b/tcg/region.c
-@@ -XXX,XX +XXX,XX @@ static size_t tree_size;
- bool in_code_gen_buffer(const void *p)
- {
--    const TCGContext *s = &tcg_init_ctx;
-     /*
-      * Much like it is valid to have a pointer to the byte past the
-      * end of an array (so long as you don't dereference it), allow
-      * a pointer to the byte past the end of the code gen buffer.
-      */
--    return (size_t)(p - s->code_gen_buffer) <= s->code_gen_buffer_size;
-+    return (size_t)(p - region.start_aligned) <= region.total_size;
- }
- #ifdef CONFIG_DEBUG_TCG
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
-     }
-     qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
--    tcg_ctx->code_gen_buffer = buf;
--    tcg_ctx->code_gen_buffer_size = size;
-+    region.start_aligned = buf;
-+    region.total_size = size;
-     return true;
- }
- #elif defined(_WIN32)
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
-         return false;
-     }
--    tcg_ctx->code_gen_buffer = buf;
--    tcg_ctx->code_gen_buffer_size = size;
-+    region.start_aligned = buf;
-+    region.total_size = size;
-     return true;
- }
- #else
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
-     /* Request large pages for the buffer.  */
-     qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
--    tcg_ctx->code_gen_buffer = buf;
--    tcg_ctx->code_gen_buffer_size = size;
-+    region.start_aligned = buf;
-+    region.total_size = size;
-     return true;
- }
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
-         return false;
-     }
-     /* The size of the mapping may have been adjusted. */
--    size = tcg_ctx->code_gen_buffer_size;
--    buf_rx = tcg_ctx->code_gen_buffer;
-+    buf_rx = region.start_aligned;
-+    size = region.total_size;
- #endif
-     buf_rw = qemu_memfd_alloc("tcg-jit", size, 0, &fd, errp);
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
- #endif
-     close(fd);
--    tcg_ctx->code_gen_buffer = buf_rw;
--    tcg_ctx->code_gen_buffer_size = size;
-+    region.start_aligned = buf_rw;
-+    region.total_size = size;
-     tcg_splitwx_diff = buf_rx - buf_rw;
-     /* Request large pages for the buffer and the splitwx.  */
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
-         return false;
-     }
--    buf_rw = (mach_vm_address_t)tcg_ctx->code_gen_buffer;
-+    buf_rw = region.start_aligned;
-     buf_rx = 0;
-     ret = mach_vm_remap(mach_task_self(),
-                         &buf_rx,
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
-  */
- void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
- {
--    void *buf, *aligned, *end;
--    size_t total_size;
-     size_t page_size;
-     size_t region_size;
--    size_t n_regions;
-     size_t i;
-     bool ok;
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
-                                splitwx, &error_fatal);
-     assert(ok);
--    buf = tcg_init_ctx.code_gen_buffer;
--    total_size = tcg_init_ctx.code_gen_buffer_size;
--    page_size = qemu_real_host_page_size;
--    n_regions = tcg_n_regions(total_size, max_cpus);
--
--    /* The first region will be 'aligned - buf' bytes larger than the others */
--    aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
--    g_assert(aligned < tcg_init_ctx.code_gen_buffer + total_size);
--
-     /*
-      * Make region_size a multiple of page_size, using aligned as the start.
-      * As a result of this we might end up with a few extra pages at the end of
-      * the buffer; we will assign those to the last region.
-      */
--    region_size = (total_size - (aligned - buf)) / n_regions;
-+    region.n = tcg_n_regions(region.total_size, max_cpus);
-+    page_size = qemu_real_host_page_size;
-+    region_size = region.total_size / region.n;
-     region_size = QEMU_ALIGN_DOWN(region_size, page_size);
-     /* A region must have at least 2 pages; one code, one guard */
-     g_assert(region_size >= 2 * page_size);
-+    region.stride = region_size;
-+
-+    /* Reserve space for guard pages. */
-+    region.size = region_size - page_size;
-+    region.total_size -= page_size;
-+
-+    /*
-+     * The first region will be smaller than the others, via the prologue,
-+     * which has yet to be allocated.  For now, the first region begins at
-+     * the page boundary.
-+     */
-+    region.after_prologue = region.start_aligned;
-     /* init the region struct */
-     qemu_mutex_init(&region.lock);
--    region.n = n_regions;
--    region.size = region_size - page_size;
--    region.stride = region_size;
--    region.after_prologue = buf;
--    region.start_aligned = aligned;
--    /* page-align the end, since its last page will be a guard page */
--    end = QEMU_ALIGN_PTR_DOWN(buf + total_size, page_size);
--    /* account for that last guard page */
--    end -= page_size;
--    total_size = end - aligned;
--    region.total_size = total_size;
-     /*
-      * Set guard pages in the rw buffer, as that's the one into which
---
-.25.1

-[PULL 23/34] tcg: Return the map protection from alloc_code_gen_buffer
+Deleted patch
-Change the interface from a boolean error indication to a
-negative error vs a non-negative protection.  For the moment
-this is only interface change, not making use of the new data.
-Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
-Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
-Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
----
- tcg/region.c | 63 +++++++++++++++++++++++++++-------------------------
-file changed, 33 insertions(+), 30 deletions(-)
-diff --git a/tcg/region.c b/tcg/region.c
-index XXXXXXX..XXXXXXX 100644
---- a/tcg/region.c
-+++ b/tcg/region.c
-@@ -XXX,XX +XXX,XX @@ static inline void split_cross_256mb(void **obuf, size_t *osize,
- static uint8_t static_code_gen_buffer[DEFAULT_CODE_GEN_BUFFER_SIZE]
-     __attribute__((aligned(CODE_GEN_ALIGN)));
--static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
-+static int alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
- {
-     void *buf, *end;
-     size_t size;
-     if (splitwx > 0) {
-         error_setg(errp, "jit split-wx not supported");
--        return false;
-+        return -1;
-     }
-     /* page-align the beginning and end of the buffer */
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
-     region.start_aligned = buf;
-     region.total_size = size;
--    return true;
-+
-+    return PROT_READ | PROT_WRITE;
- }
- #elif defined(_WIN32)
--static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
-+static int alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
- {
-     void *buf;
-     if (splitwx > 0) {
-         error_setg(errp, "jit split-wx not supported");
--        return false;
-+        return -1;
-     }
-     buf = VirtualAlloc(NULL, size, MEM_RESERVE | MEM_COMMIT,
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
-     region.start_aligned = buf;
-     region.total_size = size;
--    return true;
-+
-+    return PAGE_READ | PAGE_WRITE | PAGE_EXEC;
- }
- #else
--static bool alloc_code_gen_buffer_anon(size_t size, int prot,
--                                       int flags, Error **errp)
-+static int alloc_code_gen_buffer_anon(size_t size, int prot,
-+                                      int flags, Error **errp)
- {
-     void *buf;
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
-     if (buf == MAP_FAILED) {
-         error_setg_errno(errp, errno,
-                          "allocate %zu bytes for jit buffer", size);
--        return false;
-+        return -1;
-     }
- #ifdef __mips__
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
-     region.start_aligned = buf;
-     region.total_size = size;
--    return true;
-+    return prot;
- }
- #ifndef CONFIG_TCG_INTERPRETER
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
- #ifdef __mips__
-     /* Find space for the RX mapping, vs the 256MiB regions. */
--    if (!alloc_code_gen_buffer_anon(size, PROT_NONE,
--                                    MAP_PRIVATE | MAP_ANONYMOUS |
--                                    MAP_NORESERVE, errp)) {
-+    if (alloc_code_gen_buffer_anon(size, PROT_NONE,
-+                                   MAP_PRIVATE | MAP_ANONYMOUS |
-+                                   MAP_NORESERVE, errp) < 0) {
-         return false;
-     }
-     /* The size of the mapping may have been adjusted. */
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
-     /* Request large pages for the buffer and the splitwx.  */
-     qemu_madvise(buf_rw, size, QEMU_MADV_HUGEPAGE);
-     qemu_madvise(buf_rx, size, QEMU_MADV_HUGEPAGE);
--    return true;
-+    return PROT_READ | PROT_WRITE;
-  fail_rx:
-     error_setg_errno(errp, errno, "failed to map shared memory for execute");
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
-     if (fd >= 0) {
-         close(fd);
-     }
--    return false;
-+    return -1;
- }
- #endif /* CONFIG_POSIX */
-@@ -XXX,XX +XXX,XX @@ extern kern_return_t mach_vm_remap(vm_map_t target_task,
-                                    vm_prot_t *max_protection,
-                                    vm_inherit_t inheritance);
--static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
-+static int alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
- {
-     kern_return_t ret;
-     mach_vm_address_t buf_rw, buf_rx;
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
-     /* Map the read-write portion via normal anon memory. */
-     if (!alloc_code_gen_buffer_anon(size, PROT_READ | PROT_WRITE,
-                                     MAP_PRIVATE | MAP_ANONYMOUS, errp)) {
--        return false;
-+        return -1;
-     }
-     buf_rw = region.start_aligned;
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
-         /* TODO: Convert "ret" to a human readable error message. */
-         error_setg(errp, "vm_remap for jit splitwx failed");
-         munmap((void *)buf_rw, size);
--        return false;
-+        return -1;
-     }
-     if (mprotect((void *)buf_rx, size, PROT_READ | PROT_EXEC) != 0) {
-         error_setg_errno(errp, errno, "mprotect for jit splitwx");
-         munmap((void *)buf_rx, size);
-         munmap((void *)buf_rw, size);
--        return false;
-+        return -1;
-     }
-     tcg_splitwx_diff = buf_rx - buf_rw;
--    return true;
-+    return PROT_READ | PROT_WRITE;
- }
- #endif /* CONFIG_DARWIN */
- #endif /* CONFIG_TCG_INTERPRETER */
--static bool alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
-+static int alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
- {
- #ifndef CONFIG_TCG_INTERPRETER
- # ifdef CONFIG_DARWIN
-@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
- # endif
- #endif
-     error_setg(errp, "jit split-wx not supported");
--    return false;
-+    return -1;
- }
--static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
-+static int alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
- {
-     ERRP_GUARD();
-     int prot, flags;
-     if (splitwx) {
--        if (alloc_code_gen_buffer_splitwx(size, errp)) {
--            return true;
-+        prot = alloc_code_gen_buffer_splitwx(size, errp);
-+        if (prot >= 0) {
-+            return prot;
-         }
-         /*
-          * If splitwx force-on (1), fail;
-          * if splitwx default-on (-1), fall through to splitwx off.
-          */
-         if (splitwx > 0) {
--            return false;
-+            return -1;
-         }
-         error_free_or_abort(errp);
-     }
-@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
-     size_t page_size;
-     size_t region_size;
-     size_t i;
--    bool ok;
-+    int have_prot;
--    ok = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
--                               splitwx, &error_fatal);
--    assert(ok);
-+    have_prot = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
-+                                      splitwx, &error_fatal);
-+    assert(have_prot >= 0);
-     /*
-      * Make region_size a multiple of page_size, using aligned as the start.
---
-.25.1

This is mostly my code_gen_buffer cleanup, plus a few other random
changes thrown in.  Including a fix for a recent float32_exp2 bug.

The following changes since commit 894fc4fd670aaf04a67dc7507739f914ff4bacf2:

Merge remote-tracking branch 'remotes/jasowang/tags/net-pull-request' into staging (2021-06-11 09:21:48 +0100)

are available in the Git repository at:

https://gitlab.com/rth7680/qemu.git tags/pull-tcg-20210611

for you to fetch changes up to 60afaddc208d34f6dc86dd974f6e02724fba6eb6:

docs/devel: Explain in more detail the TB chaining mechanisms (2021-06-11 09:41:25 -0700)

----------------------------------------------------------------
Clean up code_gen_buffer allocation.
Add tcg_remove_ops_after.
Fix tcg_constant_* documentation.
Improve TB chaining documentation.
Fix float32_exp2.

----------------------------------------------------------------
Jose R. Ziviani (1):
      tcg/arm: Fix tcg_out_op function signature

Luis Pires (1):
      docs/devel: Explain in more detail the TB chaining mechanisms

Richard Henderson (32):
      meson: Split out tcg/meson.build
      meson: Split out fpu/meson.build
      tcg: Re-order tcg_region_init vs tcg_prologue_init
      tcg: Remove error return from tcg_region_initial_alloc__locked
      tcg: Split out tcg_region_initial_alloc
      tcg: Split out tcg_region_prologue_set
      tcg: Split out region.c
      accel/tcg: Inline cpu_gen_init
      accel/tcg: Move alloc_code_gen_buffer to tcg/region.c
      accel/tcg: Rename tcg_init to tcg_init_machine
      tcg: Create tcg_init
      accel/tcg: Merge tcg_exec_init into tcg_init_machine
      accel/tcg: Use MiB in tcg_init_machine
      accel/tcg: Pass down max_cpus to tcg_init
      tcg: Introduce tcg_max_ctxs
      tcg: Move MAX_CODE_GEN_BUFFER_SIZE to tcg-target.h
      tcg: Replace region.end with region.total_size
      tcg: Rename region.start to region.after_prologue
      tcg: Tidy tcg_n_regions
      tcg: Tidy split_cross_256mb
      tcg: Move in_code_gen_buffer and tests to region.c
      tcg: Allocate code_gen_buffer into struct tcg_region_state
      tcg: Return the map protection from alloc_code_gen_buffer
      tcg: Sink qemu_madvise call to common code
      util/osdep: Add qemu_mprotect_rw
      tcg: Round the tb_size default from qemu_get_host_physmem
      tcg: Merge buffer protection and guard page protection
      tcg: When allocating for !splitwx, begin with PROT_NONE
      tcg: Move tcg_init_ctx and tcg_ctx from accel/tcg/
      tcg: Introduce tcg_remove_ops_after
      tcg: Fix documentation for tcg_constant_* vs tcg_temp_free_*
      softfloat: Fix tp init in float32_exp2

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 meson.build     |  8 +-------
 tcg/meson.build | 13 +++++++++++++
 2 files changed, 14 insertions(+), 7 deletions(-)
 create mode 100644 tcg/meson.build

diff --git a/meson.build b/meson.build
index XXXXXXX..XXXXXXX 100644
--- a/meson.build
+++ b/meson.build
@@ -XXX,XX +XXX,XX @@ common_ss.add(capstone)
 specific_ss.add(files('cpu.c', 'disas.c', 'gdbstub.c'), capstone)
 specific_ss.add(when: 'CONFIG_TCG', if_true: files(
   'fpu/softfloat.c',
-  'tcg/optimize.c',
-  'tcg/tcg-common.c',
-  'tcg/tcg-op-gvec.c',
-  'tcg/tcg-op-vec.c',
-  'tcg/tcg-op.c',
-  'tcg/tcg.c',
 ))
-specific_ss.add(when: 'CONFIG_TCG_INTERPRETER', if_true: files('tcg/tci.c'))
 
 # Work around a gcc bug/misfeature wherein constant propagation looks
 # through an alias:
@@ -XXX,XX +XXX,XX @@ subdir('net')
 subdir('replay')
 subdir('semihosting')
 subdir('hw')
+subdir('tcg')
 subdir('accel')
 subdir('plugins')
 subdir('bsd-user')
diff --git a/tcg/meson.build b/tcg/meson.build
new file mode 100644
index XXXXXXX..XXXXXXX
--- /dev/null
+++ b/tcg/meson.build
@@ -XXX,XX +XXX,XX @@
+tcg_ss = ss.source_set()
+
+tcg_ss.add(files(
+  'optimize.c',
+  'tcg.c',
+  'tcg-common.c',
+  'tcg-op.c',
+  'tcg-op-gvec.c',
+  'tcg-op-vec.c',
+))
+tcg_ss.add(when: 'CONFIG_TCG_INTERPRETER', if_true: files('tci.c'))
+
+specific_ss.add_all(when: 'CONFIG_TCG', if_true: tcg_ss)
-- 
2.25.1

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 meson.build     | 4 +---
 fpu/meson.build | 1 +
 2 files changed, 2 insertions(+), 3 deletions(-)
 create mode 100644 fpu/meson.build

diff --git a/meson.build b/meson.build
index XXXXXXX..XXXXXXX 100644
--- a/meson.build
+++ b/meson.build
@@ -XXX,XX +XXX,XX @@ subdir('softmmu')
 
 common_ss.add(capstone)
 specific_ss.add(files('cpu.c', 'disas.c', 'gdbstub.c'), capstone)
-specific_ss.add(when: 'CONFIG_TCG', if_true: files(
-  'fpu/softfloat.c',
-))
 
 # Work around a gcc bug/misfeature wherein constant propagation looks
 # through an alias:
@@ -XXX,XX +XXX,XX @@ subdir('replay')
 subdir('semihosting')
 subdir('hw')
 subdir('tcg')
+subdir('fpu')
 subdir('accel')
 subdir('plugins')
 subdir('bsd-user')
diff --git a/fpu/meson.build b/fpu/meson.build
new file mode 100644
index XXXXXXX..XXXXXXX
--- /dev/null
+++ b/fpu/meson.build
@@ -0,0 +1 @@
+specific_ss.add(when: 'CONFIG_TCG', if_true: files('softfloat.c'))
-- 
2.25.1

Instead of delaying tcg_region_init until after tcg_prologue_init
is complete, do tcg_region_init first and let tcg_prologue_init
shrink the first region by the size of the generated prologue.

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 accel/tcg/tcg-all.c       | 11 ---------
 accel/tcg/translate-all.c |  3 +++
 bsd-user/main.c           |  1 -
 linux-user/main.c         |  1 -
 tcg/tcg.c                 | 52 ++++++++++++++-------------------------
 5 files changed, 22 insertions(+), 46 deletions(-)

diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/tcg-all.c
+++ b/accel/tcg/tcg-all.c
@@ -XXX,XX +XXX,XX @@ static int tcg_init(MachineState *ms)
 
     tcg_exec_init(s->tb_size * 1024 * 1024, s->splitwx_enabled);
     mttcg_enabled = s->mttcg_enabled;
-
-    /*
-     * Initialize TCG regions only for softmmu.
-     *
-     * This needs to be done later for user mode, because the prologue
-     * generation needs to be delayed so that GUEST_BASE is already set.
-     */
-#ifndef CONFIG_USER_ONLY
-    tcg_region_init();
-#endif /* !CONFIG_USER_ONLY */
-
     return 0;
 }
 
diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/translate-all.c
+++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@ void tcg_exec_init(unsigned long tb_size, int splitwx)
                                splitwx, &error_fatal);
     assert(ok);
 
+    /* TODO: allocating regions is hand-in-glove with code_gen_buffer. */
+    tcg_region_init();
+
 #if defined(CONFIG_SOFTMMU)
     /* There's no guest base to take into account, so go ahead and
        initialize the prologue now.  */
diff --git a/bsd-user/main.c b/bsd-user/main.c
index XXXXXXX..XXXXXXX 100644
--- a/bsd-user/main.c
+++ b/bsd-user/main.c
@@ -XXX,XX +XXX,XX @@ int main(int argc, char **argv)
      * the real value of GUEST_BASE into account.
      */
     tcg_prologue_init(tcg_ctx);
-    tcg_region_init();
 
     /* build Task State */
     memset(ts, 0, sizeof(TaskState));
diff --git a/linux-user/main.c b/linux-user/main.c
index XXXXXXX..XXXXXXX 100644
--- a/linux-user/main.c
+++ b/linux-user/main.c
@@ -XXX,XX +XXX,XX @@ int main(int argc, char **argv, char **envp)
        generating the prologue until now so that the prologue can take
        the real value of GUEST_BASE into account.  */
     tcg_prologue_init(tcg_ctx);
-    tcg_region_init();
 
     target_cpu_copy_regs(env, regs);
 
diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ TranslationBlock *tcg_tb_alloc(TCGContext *s)
 
 void tcg_prologue_init(TCGContext *s)
 {
-    size_t prologue_size, total_size;
-    void *buf0, *buf1;
+    size_t prologue_size;
 
     /* Put the prologue at the beginning of code_gen_buffer.  */
-    buf0 = s->code_gen_buffer;
-    total_size = s->code_gen_buffer_size;
-    s->code_ptr = buf0;
-    s->code_buf = buf0;
+    tcg_region_assign(s, 0);
+    s->code_ptr = s->code_gen_ptr;
+    s->code_buf = s->code_gen_ptr;
     s->data_gen_ptr = NULL;
 
-    /*
-     * The region trees are not yet configured, but tcg_splitwx_to_rx
-     * needs the bounds for an assert.
-     */
-    region.start = buf0;
-    region.end = buf0 + total_size;
-
 #ifndef CONFIG_TCG_INTERPRETER
-    tcg_qemu_tb_exec = (tcg_prologue_fn *)tcg_splitwx_to_rx(buf0);
+    tcg_qemu_tb_exec = (tcg_prologue_fn *)tcg_splitwx_to_rx(s->code_ptr);
 #endif
 
-    /* Compute a high-water mark, at which we voluntarily flush the buffer
-       and start over.  The size here is arbitrary, significantly larger
-       than we expect the code generation for any one opcode to require.  */
-    s->code_gen_highwater = s->code_gen_buffer + (total_size - TCG_HIGHWATER);
-
 #ifdef TCG_TARGET_NEED_POOL_LABELS
     s->pool_labels = NULL;
 #endif
@@ -XXX,XX +XXX,XX @@ void tcg_prologue_init(TCGContext *s)
     }
 #endif
 
-    buf1 = s->code_ptr;
+    prologue_size = tcg_current_code_size(s);
+
 #ifndef CONFIG_TCG_INTERPRETER
-    flush_idcache_range((uintptr_t)tcg_splitwx_to_rx(buf0), (uintptr_t)buf0,
-                        tcg_ptr_byte_diff(buf1, buf0));
+    flush_idcache_range((uintptr_t)tcg_splitwx_to_rx(s->code_buf),
+                        (uintptr_t)s->code_buf, prologue_size);
 #endif
 
-    /* Deduct the prologue from the buffer.  */
-    prologue_size = tcg_current_code_size(s);
-    s->code_gen_ptr = buf1;
-    s->code_gen_buffer = buf1;
-    s->code_buf = buf1;
-    total_size -= prologue_size;
-    s->code_gen_buffer_size = total_size;
+    /* Deduct the prologue from the first region.  */
+    region.start = s->code_ptr;
 
-    tcg_register_jit(tcg_splitwx_to_rx(s->code_gen_buffer), total_size);
+    /* Recompute boundaries of the first region. */
+    tcg_region_assign(s, 0);
+
+    tcg_register_jit(tcg_splitwx_to_rx(region.start),
+                     region.end - region.start);
 
 #ifdef DEBUG_DISAS
     if (qemu_loglevel_mask(CPU_LOG_TB_OUT_ASM)) {
         FILE *logfile = qemu_log_lock();
         qemu_log("PROLOGUE: [size=%zu]\n", prologue_size);
         if (s->data_gen_ptr) {
-            size_t code_size = s->data_gen_ptr - buf0;
+            size_t code_size = s->data_gen_ptr - s->code_gen_ptr;
             size_t data_size = prologue_size - code_size;
             size_t i;
 
-            log_disas(buf0, code_size);
+            log_disas(s->code_gen_ptr, code_size);
 
             for (i = 0; i < data_size; i += sizeof(tcg_target_ulong)) {
                 if (sizeof(tcg_target_ulong) == 8) {
@@ -XXX,XX +XXX,XX @@ void tcg_prologue_init(TCGContext *s)
                 }
             }
         } else {
-            log_disas(buf0, prologue_size);
+            log_disas(s->code_gen_ptr, prologue_size);
         }
         qemu_log("\n");
         qemu_log_flush();
-- 
2.25.1

All callers immediately assert on error, so move the assert
into the function itself.

diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ static bool tcg_region_alloc(TCGContext *s)
  * Perform a context's first region allocation.
  * This function does _not_ increment region.agg_size_full.
  */
-static inline bool tcg_region_initial_alloc__locked(TCGContext *s)
+static void tcg_region_initial_alloc__locked(TCGContext *s)
 {
-    return tcg_region_alloc__locked(s);
+    bool err = tcg_region_alloc__locked(s);
+    g_assert(!err);
 }
 
 /* Call from a safe-work context */
@@ -XXX,XX +XXX,XX @@ void tcg_region_reset_all(void)
 
     for (i = 0; i < n_ctxs; i++) {
         TCGContext *s = qatomic_read(&tcg_ctxs[i]);
-        bool err = tcg_region_initial_alloc__locked(s);
-
-        g_assert(!err);
+        tcg_region_initial_alloc__locked(s);
     }
     qemu_mutex_unlock(&region.lock);
 
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(void)
 
     /* In user-mode we support only one ctx, so do the initial allocation now */
 #ifdef CONFIG_USER_ONLY
-    {
-        bool err = tcg_region_initial_alloc__locked(tcg_ctx);
-
-        g_assert(!err);
-    }
+    tcg_region_initial_alloc__locked(tcg_ctx);
 #endif
 }
 
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
     MachineState *ms = MACHINE(qdev_get_machine());
     TCGContext *s = g_malloc(sizeof(*s));
     unsigned int i, n;
-    bool err;
 
     *s = tcg_init_ctx;
 
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
 
     tcg_ctx = s;
     qemu_mutex_lock(&region.lock);
-    err = tcg_region_initial_alloc__locked(tcg_ctx);
-    g_assert(!err);
+    tcg_region_initial_alloc__locked(s);
     qemu_mutex_unlock(&region.lock);
 }
 #endif /* !CONFIG_USER_ONLY */
-- 
2.25.1

This has only one user, and currently needs an ifdef,
but will make more sense after some code motion.

diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ static void tcg_region_initial_alloc__locked(TCGContext *s)
     g_assert(!err);
 }
 
+#ifndef CONFIG_USER_ONLY
+static void tcg_region_initial_alloc(TCGContext *s)
+{
+    qemu_mutex_lock(&region.lock);
+    tcg_region_initial_alloc__locked(s);
+    qemu_mutex_unlock(&region.lock);
+}
+#endif
+
 /* Call from a safe-work context */
 void tcg_region_reset_all(void)
 {
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
     }
 
     tcg_ctx = s;
-    qemu_mutex_lock(&region.lock);
-    tcg_region_initial_alloc__locked(s);
-    qemu_mutex_unlock(&region.lock);
+    tcg_region_initial_alloc(s);
 }
 #endif /* !CONFIG_USER_ONLY */
 
-- 
2.25.1

This has only one user, but will make more sense after some
code motion.

Always leave the tcg_init_ctx initialized to the first region,
in preparation for tcg_prologue_init().  This also requires
that we don't re-allocate the region for the first cpu, lest
we hit the assertion for total number of regions allocated .

diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(void)
 
     tcg_region_trees_init();
 
-    /* In user-mode we support only one ctx, so do the initial allocation now */
-#ifdef CONFIG_USER_ONLY
-    tcg_region_initial_alloc__locked(tcg_ctx);
-#endif
+    /*
+     * Leave the initial context initialized to the first region.
+     * This will be the context into which we generate the prologue.
+     * It is also the only context for CONFIG_USER_ONLY.
+     */
+    tcg_region_initial_alloc__locked(&tcg_init_ctx);
+}
+
+static void tcg_region_prologue_set(TCGContext *s)
+{
+    /* Deduct the prologue from the first region.  */
+    g_assert(region.start == s->code_gen_buffer);
+    region.start = s->code_ptr;
+
+    /* Recompute boundaries of the first region. */
+    tcg_region_assign(s, 0);
+
+    /* Register the balance of the buffer with gdb. */
+    tcg_register_jit(tcg_splitwx_to_rx(region.start),
+                     region.end - region.start);
 }
 
 #ifdef CONFIG_DEBUG_TCG
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
 
     if (n > 0) {
         alloc_tcg_plugin_context(s);
+        tcg_region_initial_alloc(s);
     }
 
     tcg_ctx = s;
-    tcg_region_initial_alloc(s);
 }
 #endif /* !CONFIG_USER_ONLY */
 
@@ -XXX,XX +XXX,XX @@ void tcg_prologue_init(TCGContext *s)
 {
     size_t prologue_size;
 
-    /* Put the prologue at the beginning of code_gen_buffer.  */
-    tcg_region_assign(s, 0);
     s->code_ptr = s->code_gen_ptr;
     s->code_buf = s->code_gen_ptr;
     s->data_gen_ptr = NULL;
@@ -XXX,XX +XXX,XX @@ void tcg_prologue_init(TCGContext *s)
                         (uintptr_t)s->code_buf, prologue_size);
 #endif
 
-    /* Deduct the prologue from the first region.  */
-    region.start = s->code_ptr;
-
-    /* Recompute boundaries of the first region. */
-    tcg_region_assign(s, 0);
-
-    tcg_register_jit(tcg_splitwx_to_rx(region.start),
-                     region.end - region.start);
+    tcg_region_prologue_set(s);
 
 #ifdef DEBUG_DISAS
     if (qemu_loglevel_mask(CPU_LOG_TB_OUT_ASM)) {
-- 
2.25.1

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/tcg-internal.h |  37 +++
 tcg/region.c       | 572 +++++++++++++++++++++++++++++++++++++++++++++
 tcg/tcg.c          | 547 +------------------------------------------
 tcg/meson.build    |   1 +
 4 files changed, 613 insertions(+), 544 deletions(-)
 create mode 100644 tcg/tcg-internal.h
 create mode 100644 tcg/region.c

diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
new file mode 100644
index XXXXXXX..XXXXXXX
--- /dev/null
+++ b/tcg/tcg-internal.h
@@ -XXX,XX +XXX,XX @@
+/*
+ * Internal declarations for Tiny Code Generator for QEMU
+ *
+ * Copyright (c) 2008 Fabrice Bellard
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining a copy
+ * of this software and associated documentation files (the "Software"), to deal
+ * in the Software without restriction, including without limitation the rights
+ * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+ * copies of the Software, and to permit persons to whom the Software is
+ * furnished to do so, subject to the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be included in
+ * all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+ * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+ * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
+ * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+ * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+ * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
+ * THE SOFTWARE.
+ */
+
+#ifndef TCG_INTERNAL_H
+#define TCG_INTERNAL_H 1
+
+#define TCG_HIGHWATER 1024
+
+extern TCGContext **tcg_ctxs;
+extern unsigned int n_tcg_ctxs;
+
+bool tcg_region_alloc(TCGContext *s);
+void tcg_region_initial_alloc(TCGContext *s);
+void tcg_region_prologue_set(TCGContext *s);
+
+#endif /* TCG_INTERNAL_H */
diff --git a/tcg/region.c b/tcg/region.c
new file mode 100644
index XXXXXXX..XXXXXXX
--- /dev/null
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@
+/*
+ * Memory region management for Tiny Code Generator for QEMU
+ *
+ * Copyright (c) 2008 Fabrice Bellard
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining a copy
+ * of this software and associated documentation files (the "Software"), to deal
+ * in the Software without restriction, including without limitation the rights
+ * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+ * copies of the Software, and to permit persons to whom the Software is
+ * furnished to do so, subject to the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be included in
+ * all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+ * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+ * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
+ * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+ * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+ * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
+ * THE SOFTWARE.
+ */
+
+#include "qemu/osdep.h"
+#include "exec/exec-all.h"
+#include "tcg/tcg.h"
+#if !defined(CONFIG_USER_ONLY)
+#include "hw/boards.h"
+#endif
+#include "tcg-internal.h"
+
+
+struct tcg_region_tree {
+    QemuMutex lock;
+    GTree *tree;
+    /* padding to avoid false sharing is computed at run-time */
+};
+
+/*
+ * We divide code_gen_buffer into equally-sized "regions" that TCG threads
+ * dynamically allocate from as demand dictates. Given appropriate region
+ * sizing, this minimizes flushes even when some TCG threads generate a lot
+ * more code than others.
+ */
+struct tcg_region_state {
+    QemuMutex lock;
+
+    /* fields set at init time */
+    void *start;
+    void *start_aligned;
+    void *end;
+    size_t n;
+    size_t size; /* size of one region */
+    size_t stride; /* .size + guard size */
+
+    /* fields protected by the lock */
+    size_t current; /* current region index */
+    size_t agg_size_full; /* aggregate size of full regions */
+};
+
+static struct tcg_region_state region;
+
+/*
+ * This is an array of struct tcg_region_tree's, with padding.
+ * We use void * to simplify the computation of region_trees[i]; each
+ * struct is found every tree_size bytes.
+ */
+static void *region_trees;
+static size_t tree_size;
+
+/* compare a pointer @ptr and a tb_tc @s */
+static int ptr_cmp_tb_tc(const void *ptr, const struct tb_tc *s)
+{
+    if (ptr >= s->ptr + s->size) {
+        return 1;
+    } else if (ptr < s->ptr) {
+        return -1;
+    }
+    return 0;
+}
+
+static gint tb_tc_cmp(gconstpointer ap, gconstpointer bp)
+{
+    const struct tb_tc *a = ap;
+    const struct tb_tc *b = bp;
+
+    /*
+     * When both sizes are set, we know this isn't a lookup.
+     * This is the most likely case: every TB must be inserted; lookups
+     * are a lot less frequent.
+     */
+    if (likely(a->size && b->size)) {
+        if (a->ptr > b->ptr) {
+            return 1;
+        } else if (a->ptr < b->ptr) {
+            return -1;
+        }
+        /* a->ptr == b->ptr should happen only on deletions */
+        g_assert(a->size == b->size);
+        return 0;
+    }
+    /*
+     * All lookups have either .size field set to 0.
+     * From the glib sources we see that @ap is always the lookup key. However
+     * the docs provide no guarantee, so we just mark this case as likely.
+     */
+    if (likely(a->size == 0)) {
+        return ptr_cmp_tb_tc(a->ptr, b);
+    }
+    return ptr_cmp_tb_tc(b->ptr, a);
+}
+
+static void tcg_region_trees_init(void)
+{
+    size_t i;
+
+    tree_size = ROUND_UP(sizeof(struct tcg_region_tree), qemu_dcache_linesize);
+    region_trees = qemu_memalign(qemu_dcache_linesize, region.n * tree_size);
+    for (i = 0; i < region.n; i++) {
+        struct tcg_region_tree *rt = region_trees + i * tree_size;
+
+        qemu_mutex_init(&rt->lock);
+        rt->tree = g_tree_new(tb_tc_cmp);
+    }
+}
+
+static struct tcg_region_tree *tc_ptr_to_region_tree(const void *p)
+{
+    size_t region_idx;
+
+    /*
+     * Like tcg_splitwx_to_rw, with no assert.  The pc may come from
+     * a signal handler over which the caller has no control.
+     */
+    if (!in_code_gen_buffer(p)) {
+        p -= tcg_splitwx_diff;
+        if (!in_code_gen_buffer(p)) {
+            return NULL;
+        }
+    }
+
+    if (p < region.start_aligned) {
+        region_idx = 0;
+    } else {
+        ptrdiff_t offset = p - region.start_aligned;
+
+        if (offset > region.stride * (region.n - 1)) {
+            region_idx = region.n - 1;
+        } else {
+            region_idx = offset / region.stride;
+        }
+    }
+    return region_trees + region_idx * tree_size;
+}
+
+void tcg_tb_insert(TranslationBlock *tb)
+{
+    struct tcg_region_tree *rt = tc_ptr_to_region_tree(tb->tc.ptr);
+
+    g_assert(rt != NULL);
+    qemu_mutex_lock(&rt->lock);
+    g_tree_insert(rt->tree, &tb->tc, tb);
+    qemu_mutex_unlock(&rt->lock);
+}
+
+void tcg_tb_remove(TranslationBlock *tb)
+{
+    struct tcg_region_tree *rt = tc_ptr_to_region_tree(tb->tc.ptr);
+
+    g_assert(rt != NULL);
+    qemu_mutex_lock(&rt->lock);
+    g_tree_remove(rt->tree, &tb->tc);
+    qemu_mutex_unlock(&rt->lock);
+}
+
+/*
+ * Find the TB 'tb' such that
+ * tb->tc.ptr <= tc_ptr < tb->tc.ptr + tb->tc.size
+ * Return NULL if not found.
+ */
+TranslationBlock *tcg_tb_lookup(uintptr_t tc_ptr)
+{
+    struct tcg_region_tree *rt = tc_ptr_to_region_tree((void *)tc_ptr);
+    TranslationBlock *tb;
+    struct tb_tc s = { .ptr = (void *)tc_ptr };
+
+    if (rt == NULL) {
+        return NULL;
+    }
+
+    qemu_mutex_lock(&rt->lock);
+    tb = g_tree_lookup(rt->tree, &s);
+    qemu_mutex_unlock(&rt->lock);
+    return tb;
+}
+
+static void tcg_region_tree_lock_all(void)
+{
+    size_t i;
+
+    for (i = 0; i < region.n; i++) {
+        struct tcg_region_tree *rt = region_trees + i * tree_size;
+
+        qemu_mutex_lock(&rt->lock);
+    }
+}
+
+static void tcg_region_tree_unlock_all(void)
+{
+    size_t i;
+
+    for (i = 0; i < region.n; i++) {
+        struct tcg_region_tree *rt = region_trees + i * tree_size;
+
+        qemu_mutex_unlock(&rt->lock);
+    }
+}
+
+void tcg_tb_foreach(GTraverseFunc func, gpointer user_data)
+{
+    size_t i;
+
+    tcg_region_tree_lock_all();
+    for (i = 0; i < region.n; i++) {
+        struct tcg_region_tree *rt = region_trees + i * tree_size;
+
+        g_tree_foreach(rt->tree, func, user_data);
+    }
+    tcg_region_tree_unlock_all();
+}
+
+size_t tcg_nb_tbs(void)
+{
+    size_t nb_tbs = 0;
+    size_t i;
+
+    tcg_region_tree_lock_all();
+    for (i = 0; i < region.n; i++) {
+        struct tcg_region_tree *rt = region_trees + i * tree_size;
+
+        nb_tbs += g_tree_nnodes(rt->tree);
+    }
+    tcg_region_tree_unlock_all();
+    return nb_tbs;
+}
+
+static gboolean tcg_region_tree_traverse(gpointer k, gpointer v, gpointer data)
+{
+    TranslationBlock *tb = v;
+
+    tb_destroy(tb);
+    return FALSE;
+}
+
+static void tcg_region_tree_reset_all(void)
+{
+    size_t i;
+
+    tcg_region_tree_lock_all();
+    for (i = 0; i < region.n; i++) {
+        struct tcg_region_tree *rt = region_trees + i * tree_size;
+
+        g_tree_foreach(rt->tree, tcg_region_tree_traverse, NULL);
+        /* Increment the refcount first so that destroy acts as a reset */
+        g_tree_ref(rt->tree);
+        g_tree_destroy(rt->tree);
+    }
+    tcg_region_tree_unlock_all();
+}
+
+static void tcg_region_bounds(size_t curr_region, void **pstart, void **pend)
+{
+    void *start, *end;
+
+    start = region.start_aligned + curr_region * region.stride;
+    end = start + region.size;
+
+    if (curr_region == 0) {
+        start = region.start;
+    }
+    if (curr_region == region.n - 1) {
+        end = region.end;
+    }
+
+    *pstart = start;
+    *pend = end;
+}
+
+static void tcg_region_assign(TCGContext *s, size_t curr_region)
+{
+    void *start, *end;
+
+    tcg_region_bounds(curr_region, &start, &end);
+
+    s->code_gen_buffer = start;
+    s->code_gen_ptr = start;
+    s->code_gen_buffer_size = end - start;
+    s->code_gen_highwater = end - TCG_HIGHWATER;
+}
+
+static bool tcg_region_alloc__locked(TCGContext *s)
+{
+    if (region.current == region.n) {
+        return true;
+    }
+    tcg_region_assign(s, region.current);
+    region.current++;
+    return false;
+}
+
+/*
+ * Request a new region once the one in use has filled up.
+ * Returns true on error.
+ */
+bool tcg_region_alloc(TCGContext *s)
+{
+    bool err;
+    /* read the region size now; alloc__locked will overwrite it on success */
+    size_t size_full = s->code_gen_buffer_size;
+
+    qemu_mutex_lock(&region.lock);
+    err = tcg_region_alloc__locked(s);
+    if (!err) {
+        region.agg_size_full += size_full - TCG_HIGHWATER;
+    }
+    qemu_mutex_unlock(&region.lock);
+    return err;
+}
+
+/*
+ * Perform a context's first region allocation.
+ * This function does _not_ increment region.agg_size_full.
+ */
+static void tcg_region_initial_alloc__locked(TCGContext *s)
+{
+    bool err = tcg_region_alloc__locked(s);
+    g_assert(!err);
+}
+
+void tcg_region_initial_alloc(TCGContext *s)
+{
+    qemu_mutex_lock(&region.lock);
+    tcg_region_initial_alloc__locked(s);
+    qemu_mutex_unlock(&region.lock);
+}
+
+/* Call from a safe-work context */
+void tcg_region_reset_all(void)
+{
+    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
+    unsigned int i;
+
+    qemu_mutex_lock(&region.lock);
+    region.current = 0;
+    region.agg_size_full = 0;
+
+    for (i = 0; i < n_ctxs; i++) {
+        TCGContext *s = qatomic_read(&tcg_ctxs[i]);
+        tcg_region_initial_alloc__locked(s);
+    }
+    qemu_mutex_unlock(&region.lock);
+
+    tcg_region_tree_reset_all();
+}
+
+#ifdef CONFIG_USER_ONLY
+static size_t tcg_n_regions(void)
+{
+    return 1;
+}
+#else
+/*
+ * It is likely that some vCPUs will translate more code than others, so we
+ * first try to set more regions than max_cpus, with those regions being of
+ * reasonable size. If that's not possible we make do by evenly dividing
+ * the code_gen_buffer among the vCPUs.
+ */
+static size_t tcg_n_regions(void)
+{
+    size_t i;
+
+    /* Use a single region if all we have is one vCPU thread */
+#if !defined(CONFIG_USER_ONLY)
+    MachineState *ms = MACHINE(qdev_get_machine());
+    unsigned int max_cpus = ms->smp.max_cpus;
+#endif
+    if (max_cpus == 1 || !qemu_tcg_mttcg_enabled()) {
+        return 1;
+    }
+
+    /* Try to have more regions than max_cpus, with each region being >= 2 MB */
+    for (i = 8; i > 0; i--) {
+        size_t regions_per_thread = i;
+        size_t region_size;
+
+        region_size = tcg_init_ctx.code_gen_buffer_size;
+        region_size /= max_cpus * regions_per_thread;
+
+        if (region_size >= 2 * 1024u * 1024) {
+            return max_cpus * regions_per_thread;
+        }
+    }
+    /* If we can't, then just allocate one region per vCPU thread */
+    return max_cpus;
+}
+#endif
+
+/*
+ * Initializes region partitioning.
+ *
+ * Called at init time from the parent thread (i.e. the one calling
+ * tcg_context_init), after the target's TCG globals have been set.
+ *
+ * Region partitioning works by splitting code_gen_buffer into separate regions,
+ * and then assigning regions to TCG threads so that the threads can translate
+ * code in parallel without synchronization.
+ *
+ * In softmmu the number of TCG threads is bounded by max_cpus, so we use at
+ * least max_cpus regions in MTTCG. In !MTTCG we use a single region.
+ * Note that the TCG options from the command-line (i.e. -accel accel=tcg,[...])
+ * must have been parsed before calling this function, since it calls
+ * qemu_tcg_mttcg_enabled().
+ *
+ * In user-mode we use a single region.  Having multiple regions in user-mode
+ * is not supported, because the number of vCPU threads (recall that each thread
+ * spawned by the guest corresponds to a vCPU thread) is only bounded by the
+ * OS, and usually this number is huge (tens of thousands is not uncommon).
+ * Thus, given this large bound on the number of vCPU threads and the fact
+ * that code_gen_buffer is allocated at compile-time, we cannot guarantee
+ * that the availability of at least one region per vCPU thread.
+ *
+ * However, this user-mode limitation is unlikely to be a significant problem
+ * in practice. Multi-threaded guests share most if not all of their translated
+ * code, which makes parallel code generation less appealing than in softmmu.
+ */
+void tcg_region_init(void)
+{
+    void *buf = tcg_init_ctx.code_gen_buffer;
+    void *aligned;
+    size_t size = tcg_init_ctx.code_gen_buffer_size;
+    size_t page_size = qemu_real_host_page_size;
+    size_t region_size;
+    size_t n_regions;
+    size_t i;
+
+    n_regions = tcg_n_regions();
+
+    /* The first region will be 'aligned - buf' bytes larger than the others */
+    aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
+    g_assert(aligned < tcg_init_ctx.code_gen_buffer + size);
+    /*
+     * Make region_size a multiple of page_size, using aligned as the start.
+     * As a result of this we might end up with a few extra pages at the end of
+     * the buffer; we will assign those to the last region.
+     */
+    region_size = (size - (aligned - buf)) / n_regions;
+    region_size = QEMU_ALIGN_DOWN(region_size, page_size);
+
+    /* A region must have at least 2 pages; one code, one guard */
+    g_assert(region_size >= 2 * page_size);
+
+    /* init the region struct */
+    qemu_mutex_init(&region.lock);
+    region.n = n_regions;
+    region.size = region_size - page_size;
+    region.stride = region_size;
+    region.start = buf;
+    region.start_aligned = aligned;
+    /* page-align the end, since its last page will be a guard page */
+    region.end = QEMU_ALIGN_PTR_DOWN(buf + size, page_size);
+    /* account for that last guard page */
+    region.end -= page_size;
+
+    /*
+     * Set guard pages in the rw buffer, as that's the one into which
+     * buffer overruns could occur.  Do not set guard pages in the rx
+     * buffer -- let that one use hugepages throughout.
+     */
+    for (i = 0; i < region.n; i++) {
+        void *start, *end;
+
+        tcg_region_bounds(i, &start, &end);
+
+        /*
+         * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
+         * rejects a permission change from RWX -> NONE.  Guard pages are
+         * nice for bug detection but are not essential; ignore any failure.
+         */
+        (void)qemu_mprotect_none(end, page_size);
+    }
+
+    tcg_region_trees_init();
+
+    /*
+     * Leave the initial context initialized to the first region.
+     * This will be the context into which we generate the prologue.
+     * It is also the only context for CONFIG_USER_ONLY.
+     */
+    tcg_region_initial_alloc__locked(&tcg_init_ctx);
+}
+
+void tcg_region_prologue_set(TCGContext *s)
+{
+    /* Deduct the prologue from the first region.  */
+    g_assert(region.start == s->code_gen_buffer);
+    region.start = s->code_ptr;
+
+    /* Recompute boundaries of the first region. */
+    tcg_region_assign(s, 0);
+
+    /* Register the balance of the buffer with gdb. */
+    tcg_register_jit(tcg_splitwx_to_rx(region.start),
+                     region.end - region.start);
+}
+
+/*
+ * Returns the size (in bytes) of all translated code (i.e. from all regions)
+ * currently in the cache.
+ * See also: tcg_code_capacity()
+ * Do not confuse with tcg_current_code_size(); that one applies to a single
+ * TCG context.
+ */
+size_t tcg_code_size(void)
+{
+    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
+    unsigned int i;
+    size_t total;
+
+    qemu_mutex_lock(&region.lock);
+    total = region.agg_size_full;
+    for (i = 0; i < n_ctxs; i++) {
+        const TCGContext *s = qatomic_read(&tcg_ctxs[i]);
+        size_t size;
+
+        size = qatomic_read(&s->code_gen_ptr) - s->code_gen_buffer;
+        g_assert(size <= s->code_gen_buffer_size);
+        total += size;
+    }
+    qemu_mutex_unlock(&region.lock);
+    return total;
+}
+
+/*
+ * Returns the code capacity (in bytes) of the entire cache, i.e. including all
+ * regions.
+ * See also: tcg_code_size()
+ */
+size_t tcg_code_capacity(void)
+{
+    size_t guard_size, capacity;
+
+    /* no need for synchronization; these variables are set at init time */
+    guard_size = region.stride - region.size;
+    capacity = region.end + guard_size - region.start;
+    capacity -= region.n * (guard_size + TCG_HIGHWATER);
+    return capacity;
+}
+
+size_t tcg_tb_phys_invalidate_count(void)
+{
+    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
+    unsigned int i;
+    size_t total = 0;
+
+    for (i = 0; i < n_ctxs; i++) {
+        const TCGContext *s = qatomic_read(&tcg_ctxs[i]);
+
+        total += qatomic_read(&s->tb_phys_invalidate_count);
+    }
+    return total;
+}
diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@
 
 #include "elf.h"
 #include "exec/log.h"
+#include "tcg-internal.h"
 
 /* Forward declarations for functions declared in tcg-target.c.inc and
    used here. */
@@ -XXX,XX +XXX,XX @@ static bool tcg_target_const_match(int64_t val, TCGType type, int ct);
 static int tcg_out_ldst_finalize(TCGContext *s);
 #endif
 
-#define TCG_HIGHWATER 1024
-
-static TCGContext **tcg_ctxs;
-static unsigned int n_tcg_ctxs;
+TCGContext **tcg_ctxs;
+unsigned int n_tcg_ctxs;
 TCGv_env cpu_env = 0;
 const void *tcg_code_gen_epilogue;
 uintptr_t tcg_splitwx_diff;
@@ -XXX,XX +XXX,XX @@ uintptr_t tcg_splitwx_diff;
 tcg_prologue_fn *tcg_qemu_tb_exec;
 #endif
 
-struct tcg_region_tree {
-    QemuMutex lock;
-    GTree *tree;
-    /* padding to avoid false sharing is computed at run-time */
-};
-
-/*
- * We divide code_gen_buffer into equally-sized "regions" that TCG threads
- * dynamically allocate from as demand dictates. Given appropriate region
- * sizing, this minimizes flushes even when some TCG threads generate a lot
- * more code than others.
- */
-struct tcg_region_state {
-    QemuMutex lock;
-
-    /* fields set at init time */
-    void *start;
-    void *start_aligned;
-    void *end;
-    size_t n;
-    size_t size; /* size of one region */
-    size_t stride; /* .size + guard size */
-
-    /* fields protected by the lock */
-    size_t current; /* current region index */
-    size_t agg_size_full; /* aggregate size of full regions */
-};
-
-static struct tcg_region_state region;
-/*
- * This is an array of struct tcg_region_tree's, with padding.
- * We use void * to simplify the computation of region_trees[i]; each
- * struct is found every tree_size bytes.
- */
-static void *region_trees;
-static size_t tree_size;
 static TCGRegSet tcg_target_available_regs[TCG_TYPE_COUNT];
 static TCGRegSet tcg_target_call_clobber_regs;
 
@@ -XXX,XX +XXX,XX @@ static const TCGTargetOpDef constraint_sets[] = {
 
 #include "tcg-target.c.inc"
 
-/* compare a pointer @ptr and a tb_tc @s */
-static int ptr_cmp_tb_tc(const void *ptr, const struct tb_tc *s)
-{
-    if (ptr >= s->ptr + s->size) {
-        return 1;
-    } else if (ptr < s->ptr) {
-        return -1;
-    }
-    return 0;
-}
-
-static gint tb_tc_cmp(gconstpointer ap, gconstpointer bp)
-{
-    const struct tb_tc *a = ap;
-    const struct tb_tc *b = bp;
-
-    /*
-     * When both sizes are set, we know this isn't a lookup.
-     * This is the most likely case: every TB must be inserted; lookups
-     * are a lot less frequent.
-     */
-    if (likely(a->size && b->size)) {
-        if (a->ptr > b->ptr) {
-            return 1;
-        } else if (a->ptr < b->ptr) {
-            return -1;
-        }
-        /* a->ptr == b->ptr should happen only on deletions */
-        g_assert(a->size == b->size);
-        return 0;
-    }
-    /*
-     * All lookups have either .size field set to 0.
-     * From the glib sources we see that @ap is always the lookup key. However
-     * the docs provide no guarantee, so we just mark this case as likely.
-     */
-    if (likely(a->size == 0)) {
-        return ptr_cmp_tb_tc(a->ptr, b);
-    }
-    return ptr_cmp_tb_tc(b->ptr, a);
-}
-
-static void tcg_region_trees_init(void)
-{
-    size_t i;
-
-    tree_size = ROUND_UP(sizeof(struct tcg_region_tree), qemu_dcache_linesize);
-    region_trees = qemu_memalign(qemu_dcache_linesize, region.n * tree_size);
-    for (i = 0; i < region.n; i++) {
-        struct tcg_region_tree *rt = region_trees + i * tree_size;
-
-        qemu_mutex_init(&rt->lock);
-        rt->tree = g_tree_new(tb_tc_cmp);
-    }
-}
-
-static struct tcg_region_tree *tc_ptr_to_region_tree(const void *p)
-{
-    size_t region_idx;
-
-    /*
-     * Like tcg_splitwx_to_rw, with no assert.  The pc may come from
-     * a signal handler over which the caller has no control.
-     */
-    if (!in_code_gen_buffer(p)) {
-        p -= tcg_splitwx_diff;
-        if (!in_code_gen_buffer(p)) {
-            return NULL;
-        }
-    }
-
-    if (p < region.start_aligned) {
-        region_idx = 0;
-    } else {
-        ptrdiff_t offset = p - region.start_aligned;
-
-        if (offset > region.stride * (region.n - 1)) {
-            region_idx = region.n - 1;
-        } else {
-            region_idx = offset / region.stride;
-        }
-    }
-    return region_trees + region_idx * tree_size;
-}
-
-void tcg_tb_insert(TranslationBlock *tb)
-{
-    struct tcg_region_tree *rt = tc_ptr_to_region_tree(tb->tc.ptr);
-
-    g_assert(rt != NULL);
-    qemu_mutex_lock(&rt->lock);
-    g_tree_insert(rt->tree, &tb->tc, tb);
-    qemu_mutex_unlock(&rt->lock);
-}
-
-void tcg_tb_remove(TranslationBlock *tb)
-{
-    struct tcg_region_tree *rt = tc_ptr_to_region_tree(tb->tc.ptr);
-
-    g_assert(rt != NULL);
-    qemu_mutex_lock(&rt->lock);
-    g_tree_remove(rt->tree, &tb->tc);
-    qemu_mutex_unlock(&rt->lock);
-}
-
-/*
- * Find the TB 'tb' such that
- * tb->tc.ptr <= tc_ptr < tb->tc.ptr + tb->tc.size
- * Return NULL if not found.
- */
-TranslationBlock *tcg_tb_lookup(uintptr_t tc_ptr)
-{
-    struct tcg_region_tree *rt = tc_ptr_to_region_tree((void *)tc_ptr);
-    TranslationBlock *tb;
-    struct tb_tc s = { .ptr = (void *)tc_ptr };
-
-    if (rt == NULL) {
-        return NULL;
-    }
-
-    qemu_mutex_lock(&rt->lock);
-    tb = g_tree_lookup(rt->tree, &s);
-    qemu_mutex_unlock(&rt->lock);
-    return tb;
-}
-
-static void tcg_region_tree_lock_all(void)
-{
-    size_t i;
-
-    for (i = 0; i < region.n; i++) {
-        struct tcg_region_tree *rt = region_trees + i * tree_size;
-
-        qemu_mutex_lock(&rt->lock);
-    }
-}
-
-static void tcg_region_tree_unlock_all(void)
-{
-    size_t i;
-
-    for (i = 0; i < region.n; i++) {
-        struct tcg_region_tree *rt = region_trees + i * tree_size;
-
-        qemu_mutex_unlock(&rt->lock);
-    }
-}
-
-void tcg_tb_foreach(GTraverseFunc func, gpointer user_data)
-{
-    size_t i;
-
-    tcg_region_tree_lock_all();
-    for (i = 0; i < region.n; i++) {
-        struct tcg_region_tree *rt = region_trees + i * tree_size;
-
-        g_tree_foreach(rt->tree, func, user_data);
-    }
-    tcg_region_tree_unlock_all();
-}
-
-size_t tcg_nb_tbs(void)
-{
-    size_t nb_tbs = 0;
-    size_t i;
-
-    tcg_region_tree_lock_all();
-    for (i = 0; i < region.n; i++) {
-        struct tcg_region_tree *rt = region_trees + i * tree_size;
-
-        nb_tbs += g_tree_nnodes(rt->tree);
-    }
-    tcg_region_tree_unlock_all();
-    return nb_tbs;
-}
-
-static gboolean tcg_region_tree_traverse(gpointer k, gpointer v, gpointer data)
-{
-    TranslationBlock *tb = v;
-
-    tb_destroy(tb);
-    return FALSE;
-}
-
-static void tcg_region_tree_reset_all(void)
-{
-    size_t i;
-
-    tcg_region_tree_lock_all();
-    for (i = 0; i < region.n; i++) {
-        struct tcg_region_tree *rt = region_trees + i * tree_size;
-
-        g_tree_foreach(rt->tree, tcg_region_tree_traverse, NULL);
-        /* Increment the refcount first so that destroy acts as a reset */
-        g_tree_ref(rt->tree);
-        g_tree_destroy(rt->tree);
-    }
-    tcg_region_tree_unlock_all();
-}
-
-static void tcg_region_bounds(size_t curr_region, void **pstart, void **pend)
-{
-    void *start, *end;
-
-    start = region.start_aligned + curr_region * region.stride;
-    end = start + region.size;
-
-    if (curr_region == 0) {
-        start = region.start;
-    }
-    if (curr_region == region.n - 1) {
-        end = region.end;
-    }
-
-    *pstart = start;
-    *pend = end;
-}
-
-static void tcg_region_assign(TCGContext *s, size_t curr_region)
-{
-    void *start, *end;
-
-    tcg_region_bounds(curr_region, &start, &end);
-
-    s->code_gen_buffer = start;
-    s->code_gen_ptr = start;
-    s->code_gen_buffer_size = end - start;
-    s->code_gen_highwater = end - TCG_HIGHWATER;
-}
-
-static bool tcg_region_alloc__locked(TCGContext *s)
-{
-    if (region.current == region.n) {
-        return true;
-    }
-    tcg_region_assign(s, region.current);
-    region.current++;
-    return false;
-}
-
-/*
- * Request a new region once the one in use has filled up.
- * Returns true on error.
- */
-static bool tcg_region_alloc(TCGContext *s)
-{
-    bool err;
-    /* read the region size now; alloc__locked will overwrite it on success */
-    size_t size_full = s->code_gen_buffer_size;
-
-    qemu_mutex_lock(&region.lock);
-    err = tcg_region_alloc__locked(s);
-    if (!err) {
-        region.agg_size_full += size_full - TCG_HIGHWATER;
-    }
-    qemu_mutex_unlock(&region.lock);
-    return err;
-}
-
-/*
- * Perform a context's first region allocation.
- * This function does _not_ increment region.agg_size_full.
- */
-static void tcg_region_initial_alloc__locked(TCGContext *s)
-{
-    bool err = tcg_region_alloc__locked(s);
-    g_assert(!err);
-}
-
-#ifndef CONFIG_USER_ONLY
-static void tcg_region_initial_alloc(TCGContext *s)
-{
-    qemu_mutex_lock(&region.lock);
-    tcg_region_initial_alloc__locked(s);
-    qemu_mutex_unlock(&region.lock);
-}
-#endif
-
-/* Call from a safe-work context */
-void tcg_region_reset_all(void)
-{
-    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
-    unsigned int i;
-
-    qemu_mutex_lock(&region.lock);
-    region.current = 0;
-    region.agg_size_full = 0;
-
-    for (i = 0; i < n_ctxs; i++) {
-        TCGContext *s = qatomic_read(&tcg_ctxs[i]);
-        tcg_region_initial_alloc__locked(s);
-    }
-    qemu_mutex_unlock(&region.lock);
-
-    tcg_region_tree_reset_all();
-}
-
-#ifdef CONFIG_USER_ONLY
-static size_t tcg_n_regions(void)
-{
-    return 1;
-}
-#else
-/*
- * It is likely that some vCPUs will translate more code than others, so we
- * first try to set more regions than max_cpus, with those regions being of
- * reasonable size. If that's not possible we make do by evenly dividing
- * the code_gen_buffer among the vCPUs.
- */
-static size_t tcg_n_regions(void)
-{
-    size_t i;
-
-    /* Use a single region if all we have is one vCPU thread */
-#if !defined(CONFIG_USER_ONLY)
-    MachineState *ms = MACHINE(qdev_get_machine());
-    unsigned int max_cpus = ms->smp.max_cpus;
-#endif
-    if (max_cpus == 1 || !qemu_tcg_mttcg_enabled()) {
-        return 1;
-    }
-
-    /* Try to have more regions than max_cpus, with each region being >= 2 MB */
-    for (i = 8; i > 0; i--) {
-        size_t regions_per_thread = i;
-        size_t region_size;
-
-        region_size = tcg_init_ctx.code_gen_buffer_size;
-        region_size /= max_cpus * regions_per_thread;
-
-        if (region_size >= 2 * 1024u * 1024) {
-            return max_cpus * regions_per_thread;
-        }
-    }
-    /* If we can't, then just allocate one region per vCPU thread */
-    return max_cpus;
-}
-#endif
-
-/*
- * Initializes region partitioning.
- *
- * Called at init time from the parent thread (i.e. the one calling
- * tcg_context_init), after the target's TCG globals have been set.
- *
- * Region partitioning works by splitting code_gen_buffer into separate regions,
- * and then assigning regions to TCG threads so that the threads can translate
- * code in parallel without synchronization.
- *
- * In softmmu the number of TCG threads is bounded by max_cpus, so we use at
- * least max_cpus regions in MTTCG. In !MTTCG we use a single region.
- * Note that the TCG options from the command-line (i.e. -accel accel=tcg,[...])
- * must have been parsed before calling this function, since it calls
- * qemu_tcg_mttcg_enabled().
- *
- * In user-mode we use a single region.  Having multiple regions in user-mode
- * is not supported, because the number of vCPU threads (recall that each thread
- * spawned by the guest corresponds to a vCPU thread) is only bounded by the
- * OS, and usually this number is huge (tens of thousands is not uncommon).
- * Thus, given this large bound on the number of vCPU threads and the fact
- * that code_gen_buffer is allocated at compile-time, we cannot guarantee
- * that the availability of at least one region per vCPU thread.
- *
- * However, this user-mode limitation is unlikely to be a significant problem
- * in practice. Multi-threaded guests share most if not all of their translated
- * code, which makes parallel code generation less appealing than in softmmu.
- */
-void tcg_region_init(void)
-{
-    void *buf = tcg_init_ctx.code_gen_buffer;
-    void *aligned;
-    size_t size = tcg_init_ctx.code_gen_buffer_size;
-    size_t page_size = qemu_real_host_page_size;
-    size_t region_size;
-    size_t n_regions;
-    size_t i;
-
-    n_regions = tcg_n_regions();
-
-    /* The first region will be 'aligned - buf' bytes larger than the others */
-    aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
-    g_assert(aligned < tcg_init_ctx.code_gen_buffer + size);
-    /*
-     * Make region_size a multiple of page_size, using aligned as the start.
-     * As a result of this we might end up with a few extra pages at the end of
-     * the buffer; we will assign those to the last region.
-     */
-    region_size = (size - (aligned - buf)) / n_regions;
-    region_size = QEMU_ALIGN_DOWN(region_size, page_size);
-
-    /* A region must have at least 2 pages; one code, one guard */
-    g_assert(region_size >= 2 * page_size);
-
-    /* init the region struct */
-    qemu_mutex_init(&region.lock);
-    region.n = n_regions;
-    region.size = region_size - page_size;
-    region.stride = region_size;
-    region.start = buf;
-    region.start_aligned = aligned;
-    /* page-align the end, since its last page will be a guard page */
-    region.end = QEMU_ALIGN_PTR_DOWN(buf + size, page_size);
-    /* account for that last guard page */
-    region.end -= page_size;
-
-    /*
-     * Set guard pages in the rw buffer, as that's the one into which
-     * buffer overruns could occur.  Do not set guard pages in the rx
-     * buffer -- let that one use hugepages throughout.
-     */
-    for (i = 0; i < region.n; i++) {
-        void *start, *end;
-
-        tcg_region_bounds(i, &start, &end);
-
-        /*
-         * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
-         * rejects a permission change from RWX -> NONE.  Guard pages are
-         * nice for bug detection but are not essential; ignore any failure.
-         */
-        (void)qemu_mprotect_none(end, page_size);
-    }
-
-    tcg_region_trees_init();
-
-    /*
-     * Leave the initial context initialized to the first region.
-     * This will be the context into which we generate the prologue.
-     * It is also the only context for CONFIG_USER_ONLY.
-     */
-    tcg_region_initial_alloc__locked(&tcg_init_ctx);
-}
-
-static void tcg_region_prologue_set(TCGContext *s)
-{
-    /* Deduct the prologue from the first region.  */
-    g_assert(region.start == s->code_gen_buffer);
-    region.start = s->code_ptr;
-
-    /* Recompute boundaries of the first region. */
-    tcg_region_assign(s, 0);
-
-    /* Register the balance of the buffer with gdb. */
-    tcg_register_jit(tcg_splitwx_to_rx(region.start),
-                     region.end - region.start);
-}
-
 #ifdef CONFIG_DEBUG_TCG
 const void *tcg_splitwx_to_rx(void *rw)
 {
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
 }
 #endif /* !CONFIG_USER_ONLY */
 
-/*
- * Returns the size (in bytes) of all translated code (i.e. from all regions)
- * currently in the cache.
- * See also: tcg_code_capacity()
- * Do not confuse with tcg_current_code_size(); that one applies to a single
- * TCG context.
- */
-size_t tcg_code_size(void)
-{
-    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
-    unsigned int i;
-    size_t total;
-
-    qemu_mutex_lock(&region.lock);
-    total = region.agg_size_full;
-    for (i = 0; i < n_ctxs; i++) {
-        const TCGContext *s = qatomic_read(&tcg_ctxs[i]);
-        size_t size;
-
-        size = qatomic_read(&s->code_gen_ptr) - s->code_gen_buffer;
-        g_assert(size <= s->code_gen_buffer_size);
-        total += size;
-    }
-    qemu_mutex_unlock(&region.lock);
-    return total;
-}
-
-/*
- * Returns the code capacity (in bytes) of the entire cache, i.e. including all
- * regions.
- * See also: tcg_code_size()
- */
-size_t tcg_code_capacity(void)
-{
-    size_t guard_size, capacity;
-
-    /* no need for synchronization; these variables are set at init time */
-    guard_size = region.stride - region.size;
-    capacity = region.end + guard_size - region.start;
-    capacity -= region.n * (guard_size + TCG_HIGHWATER);
-    return capacity;
-}
-
-size_t tcg_tb_phys_invalidate_count(void)
-{
-    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
-    unsigned int i;
-    size_t total = 0;
-
-    for (i = 0; i < n_ctxs; i++) {
-        const TCGContext *s = qatomic_read(&tcg_ctxs[i]);
-
-        total += qatomic_read(&s->tb_phys_invalidate_count);
-    }
-    return total;
-}
-
 /* pool based memory allocation */
 void *tcg_malloc_internal(TCGContext *s, int size)
 {
diff --git a/tcg/meson.build b/tcg/meson.build
index XXXXXXX..XXXXXXX 100644
--- a/tcg/meson.build
+++ b/tcg/meson.build
@@ -XXX,XX +XXX,XX @@ tcg_ss = ss.source_set()
 
 tcg_ss.add(files(
   'optimize.c',
+  'region.c',
   'tcg.c',
   'tcg-common.c',
   'tcg-op.c',
-- 
2.25.1

It consists of one function call and has only one caller.

diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/translate-all.c
+++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@ static void page_table_config_init(void)
     assert(v_l2_levels >= 0);
 }
 
-static void cpu_gen_init(void)
-{
-    tcg_context_init(&tcg_init_ctx);
-}
-
 /* Encode VAL as a signed leb128 sequence at P.
    Return P incremented past the encoded value.  */
 static uint8_t *encode_sleb128(uint8_t *p, target_long val)
@@ -XXX,XX +XXX,XX @@ void tcg_exec_init(unsigned long tb_size, int splitwx)
     bool ok;
 
     tcg_allowed = true;
-    cpu_gen_init();
+    tcg_context_init(&tcg_init_ctx);
     page_init();
     tb_htable_init();
 
-- 
2.25.1

Buffer management is integral to tcg.  Do not leave the allocation
to code outside of tcg/.  This is code movement, with further
cleanups to follow.

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/tcg/tcg.h         |   2 +-
 accel/tcg/translate-all.c | 414 +-----------------------------------
 tcg/region.c              | 431 +++++++++++++++++++++++++++++++++++++-
 3 files changed, 428 insertions(+), 419 deletions(-)

diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
index XXXXXXX..XXXXXXX 100644
--- a/include/tcg/tcg.h
+++ b/include/tcg/tcg.h
@@ -XXX,XX +XXX,XX @@ void *tcg_malloc_internal(TCGContext *s, int size);
 void tcg_pool_reset(TCGContext *s);
 TranslationBlock *tcg_tb_alloc(TCGContext *s);
 
-void tcg_region_init(void);
+void tcg_region_init(size_t tb_size, int splitwx);
 void tb_destroy(TranslationBlock *tb);
 void tcg_region_reset_all(void);
 
diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/translate-all.c
+++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@
  */
 
 #include "qemu/osdep.h"
-#include "qemu/units.h"
 #include "qemu-common.h"
 
 #define NO_CPU_IO_DEFS
@@ -XXX,XX +XXX,XX @@
 #include "exec/cputlb.h"
 #include "exec/translate-all.h"
 #include "qemu/bitmap.h"
-#include "qemu/error-report.h"
 #include "qemu/qemu-print.h"
 #include "qemu/timer.h"
 #include "qemu/main-loop.h"
@@ -XXX,XX +XXX,XX @@ static void page_lock_pair(PageDesc **ret_p1, tb_page_addr_t phys1,
     }
 }
 
-/* Minimum size of the code gen buffer.  This number is randomly chosen,
-   but not so small that we can't have a fair number of TB's live.  */
-#define MIN_CODE_GEN_BUFFER_SIZE     (1 * MiB)
-
-/* Maximum size of the code gen buffer we'd like to use.  Unless otherwise
-   indicated, this is constrained by the range of direct branches on the
-   host cpu, as used by the TCG implementation of goto_tb.  */
-#if defined(__x86_64__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
-#elif defined(__sparc__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
-#elif defined(__powerpc64__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
-#elif defined(__powerpc__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (32 * MiB)
-#elif defined(__aarch64__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
-#elif defined(__s390x__)
-  /* We have a +- 4GB range on the branches; leave some slop.  */
-# define MAX_CODE_GEN_BUFFER_SIZE  (3 * GiB)
-#elif defined(__mips__)
-  /* We have a 256MB branch region, but leave room to make sure the
-     main executable is also within that region.  */
-# define MAX_CODE_GEN_BUFFER_SIZE  (128 * MiB)
-#else
-# define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
-#endif
-
-#if TCG_TARGET_REG_BITS == 32
-#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (32 * MiB)
-#ifdef CONFIG_USER_ONLY
-/*
- * For user mode on smaller 32 bit systems we may run into trouble
- * allocating big chunks of data in the right place. On these systems
- * we utilise a static code generation buffer directly in the binary.
- */
-#define USE_STATIC_CODE_GEN_BUFFER
-#endif
-#else /* TCG_TARGET_REG_BITS == 64 */
-#ifdef CONFIG_USER_ONLY
-/*
- * As user-mode emulation typically means running multiple instances
- * of the translator don't go too nuts with our default code gen
- * buffer lest we make things too hard for the OS.
- */
-#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (128 * MiB)
-#else
-/*
- * We expect most system emulation to run one or two guests per host.
- * Users running large scale system emulation may want to tweak their
- * runtime setup via the tb-size control on the command line.
- */
-#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (1 * GiB)
-#endif
-#endif
-
-#define DEFAULT_CODE_GEN_BUFFER_SIZE \
-  (DEFAULT_CODE_GEN_BUFFER_SIZE_1 < MAX_CODE_GEN_BUFFER_SIZE \
-   ? DEFAULT_CODE_GEN_BUFFER_SIZE_1 : MAX_CODE_GEN_BUFFER_SIZE)
-
-static size_t size_code_gen_buffer(size_t tb_size)
-{
-    /* Size the buffer.  */
-    if (tb_size == 0) {
-        size_t phys_mem = qemu_get_host_physmem();
-        if (phys_mem == 0) {
-            tb_size = DEFAULT_CODE_GEN_BUFFER_SIZE;
-        } else {
-            tb_size = MIN(DEFAULT_CODE_GEN_BUFFER_SIZE, phys_mem / 8);
-        }
-    }
-    if (tb_size < MIN_CODE_GEN_BUFFER_SIZE) {
-        tb_size = MIN_CODE_GEN_BUFFER_SIZE;
-    }
-    if (tb_size > MAX_CODE_GEN_BUFFER_SIZE) {
-        tb_size = MAX_CODE_GEN_BUFFER_SIZE;
-    }
-    return tb_size;
-}
-
-#ifdef __mips__
-/* In order to use J and JAL within the code_gen_buffer, we require
-   that the buffer not cross a 256MB boundary.  */
-static inline bool cross_256mb(void *addr, size_t size)
-{
-    return ((uintptr_t)addr ^ ((uintptr_t)addr + size)) & ~0x0ffffffful;
-}
-
-/* We weren't able to allocate a buffer without crossing that boundary,
-   so make do with the larger portion of the buffer that doesn't cross.
-   Returns the new base of the buffer, and adjusts code_gen_buffer_size.  */
-static inline void *split_cross_256mb(void *buf1, size_t size1)
-{
-    void *buf2 = (void *)(((uintptr_t)buf1 + size1) & ~0x0ffffffful);
-    size_t size2 = buf1 + size1 - buf2;
-
-    size1 = buf2 - buf1;
-    if (size1 < size2) {
-        size1 = size2;
-        buf1 = buf2;
-    }
-
-    tcg_ctx->code_gen_buffer_size = size1;
-    return buf1;
-}
-#endif
-
-#ifdef USE_STATIC_CODE_GEN_BUFFER
-static uint8_t static_code_gen_buffer[DEFAULT_CODE_GEN_BUFFER_SIZE]
-    __attribute__((aligned(CODE_GEN_ALIGN)));
-
-static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
-{
-    void *buf, *end;
-    size_t size;
-
-    if (splitwx > 0) {
-        error_setg(errp, "jit split-wx not supported");
-        return false;
-    }
-
-    /* page-align the beginning and end of the buffer */
-    buf = static_code_gen_buffer;
-    end = static_code_gen_buffer + sizeof(static_code_gen_buffer);
-    buf = QEMU_ALIGN_PTR_UP(buf, qemu_real_host_page_size);
-    end = QEMU_ALIGN_PTR_DOWN(end, qemu_real_host_page_size);
-
-    size = end - buf;
-
-    /* Honor a command-line option limiting the size of the buffer.  */
-    if (size > tb_size) {
-        size = QEMU_ALIGN_DOWN(tb_size, qemu_real_host_page_size);
-    }
-    tcg_ctx->code_gen_buffer_size = size;
-
-#ifdef __mips__
-    if (cross_256mb(buf, size)) {
-        buf = split_cross_256mb(buf, size);
-        size = tcg_ctx->code_gen_buffer_size;
-    }
-#endif
-
-    if (qemu_mprotect_rwx(buf, size)) {
-        error_setg_errno(errp, errno, "mprotect of jit buffer");
-        return false;
-    }
-    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
-
-    tcg_ctx->code_gen_buffer = buf;
-    return true;
-}
-#elif defined(_WIN32)
-static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
-{
-    void *buf;
-
-    if (splitwx > 0) {
-        error_setg(errp, "jit split-wx not supported");
-        return false;
-    }
-
-    buf = VirtualAlloc(NULL, size, MEM_RESERVE | MEM_COMMIT,
-                             PAGE_EXECUTE_READWRITE);
-    if (buf == NULL) {
-        error_setg_win32(errp, GetLastError(),
-                         "allocate %zu bytes for jit buffer", size);
-        return false;
-    }
-
-    tcg_ctx->code_gen_buffer = buf;
-    tcg_ctx->code_gen_buffer_size = size;
-    return true;
-}
-#else
-static bool alloc_code_gen_buffer_anon(size_t size, int prot,
-                                       int flags, Error **errp)
-{
-    void *buf;
-
-    buf = mmap(NULL, size, prot, flags, -1, 0);
-    if (buf == MAP_FAILED) {
-        error_setg_errno(errp, errno,
-                         "allocate %zu bytes for jit buffer", size);
-        return false;
-    }
-    tcg_ctx->code_gen_buffer_size = size;
-
-#ifdef __mips__
-    if (cross_256mb(buf, size)) {
-        /*
-         * Try again, with the original still mapped, to avoid re-acquiring
-         * the same 256mb crossing.
-         */
-        size_t size2;
-        void *buf2 = mmap(NULL, size, prot, flags, -1, 0);
-        switch ((int)(buf2 != MAP_FAILED)) {
-        case 1:
-            if (!cross_256mb(buf2, size)) {
-                /* Success!  Use the new buffer.  */
-                munmap(buf, size);
-                break;
-            }
-            /* Failure.  Work with what we had.  */
-            munmap(buf2, size);
-            /* fallthru */
-        default:
-            /* Split the original buffer.  Free the smaller half.  */
-            buf2 = split_cross_256mb(buf, size);
-            size2 = tcg_ctx->code_gen_buffer_size;
-            if (buf == buf2) {
-                munmap(buf + size2, size - size2);
-            } else {
-                munmap(buf, size - size2);
-            }
-            size = size2;
-            break;
-        }
-        buf = buf2;
-    }
-#endif
-
-    /* Request large pages for the buffer.  */
-    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
-
-    tcg_ctx->code_gen_buffer = buf;
-    return true;
-}
-
-#ifndef CONFIG_TCG_INTERPRETER
-#ifdef CONFIG_POSIX
-#include "qemu/memfd.h"
-
-static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
-{
-    void *buf_rw = NULL, *buf_rx = MAP_FAILED;
-    int fd = -1;
-
-#ifdef __mips__
-    /* Find space for the RX mapping, vs the 256MiB regions. */
-    if (!alloc_code_gen_buffer_anon(size, PROT_NONE,
-                                    MAP_PRIVATE | MAP_ANONYMOUS |
-                                    MAP_NORESERVE, errp)) {
-        return false;
-    }
-    /* The size of the mapping may have been adjusted. */
-    size = tcg_ctx->code_gen_buffer_size;
-    buf_rx = tcg_ctx->code_gen_buffer;
-#endif
-
-    buf_rw = qemu_memfd_alloc("tcg-jit", size, 0, &fd, errp);
-    if (buf_rw == NULL) {
-        goto fail;
-    }
-
-#ifdef __mips__
-    void *tmp = mmap(buf_rx, size, PROT_READ | PROT_EXEC,
-                     MAP_SHARED | MAP_FIXED, fd, 0);
-    if (tmp != buf_rx) {
-        goto fail_rx;
-    }
-#else
-    buf_rx = mmap(NULL, size, PROT_READ | PROT_EXEC, MAP_SHARED, fd, 0);
-    if (buf_rx == MAP_FAILED) {
-        goto fail_rx;
-    }
-#endif
-
-    close(fd);
-    tcg_ctx->code_gen_buffer = buf_rw;
-    tcg_ctx->code_gen_buffer_size = size;
-    tcg_splitwx_diff = buf_rx - buf_rw;
-
-    /* Request large pages for the buffer and the splitwx.  */
-    qemu_madvise(buf_rw, size, QEMU_MADV_HUGEPAGE);
-    qemu_madvise(buf_rx, size, QEMU_MADV_HUGEPAGE);
-    return true;
-
- fail_rx:
-    error_setg_errno(errp, errno, "failed to map shared memory for execute");
- fail:
-    if (buf_rx != MAP_FAILED) {
-        munmap(buf_rx, size);
-    }
-    if (buf_rw) {
-        munmap(buf_rw, size);
-    }
-    if (fd >= 0) {
-        close(fd);
-    }
-    return false;
-}
-#endif /* CONFIG_POSIX */
-
-#ifdef CONFIG_DARWIN
-#include <mach/mach.h>
-
-extern kern_return_t mach_vm_remap(vm_map_t target_task,
-                                   mach_vm_address_t *target_address,
-                                   mach_vm_size_t size,
-                                   mach_vm_offset_t mask,
-                                   int flags,
-                                   vm_map_t src_task,
-                                   mach_vm_address_t src_address,
-                                   boolean_t copy,
-                                   vm_prot_t *cur_protection,
-                                   vm_prot_t *max_protection,
-                                   vm_inherit_t inheritance);
-
-static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
-{
-    kern_return_t ret;
-    mach_vm_address_t buf_rw, buf_rx;
-    vm_prot_t cur_prot, max_prot;
-
-    /* Map the read-write portion via normal anon memory. */
-    if (!alloc_code_gen_buffer_anon(size, PROT_READ | PROT_WRITE,
-                                    MAP_PRIVATE | MAP_ANONYMOUS, errp)) {
-        return false;
-    }
-
-    buf_rw = (mach_vm_address_t)tcg_ctx->code_gen_buffer;
-    buf_rx = 0;
-    ret = mach_vm_remap(mach_task_self(),
-                        &buf_rx,
-                        size,
-                        0,
-                        VM_FLAGS_ANYWHERE,
-                        mach_task_self(),
-                        buf_rw,
-                        false,
-                        &cur_prot,
-                        &max_prot,
-                        VM_INHERIT_NONE);
-    if (ret != KERN_SUCCESS) {
-        /* TODO: Convert "ret" to a human readable error message. */
-        error_setg(errp, "vm_remap for jit splitwx failed");
-        munmap((void *)buf_rw, size);
-        return false;
-    }
-
-    if (mprotect((void *)buf_rx, size, PROT_READ | PROT_EXEC) != 0) {
-        error_setg_errno(errp, errno, "mprotect for jit splitwx");
-        munmap((void *)buf_rx, size);
-        munmap((void *)buf_rw, size);
-        return false;
-    }
-
-    tcg_splitwx_diff = buf_rx - buf_rw;
-    return true;
-}
-#endif /* CONFIG_DARWIN */
-#endif /* CONFIG_TCG_INTERPRETER */
-
-static bool alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
-{
-#ifndef CONFIG_TCG_INTERPRETER
-# ifdef CONFIG_DARWIN
-    return alloc_code_gen_buffer_splitwx_vmremap(size, errp);
-# endif
-# ifdef CONFIG_POSIX
-    return alloc_code_gen_buffer_splitwx_memfd(size, errp);
-# endif
-#endif
-    error_setg(errp, "jit split-wx not supported");
-    return false;
-}
-
-static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
-{
-    ERRP_GUARD();
-    int prot, flags;
-
-    if (splitwx) {
-        if (alloc_code_gen_buffer_splitwx(size, errp)) {
-            return true;
-        }
-        /*
-         * If splitwx force-on (1), fail;
-         * if splitwx default-on (-1), fall through to splitwx off.
-         */
-        if (splitwx > 0) {
-            return false;
-        }
-        error_free_or_abort(errp);
-    }
-
-    prot = PROT_READ | PROT_WRITE | PROT_EXEC;
-    flags = MAP_PRIVATE | MAP_ANONYMOUS;
-#ifdef CONFIG_TCG_INTERPRETER
-    /* The tcg interpreter does not need execute permission. */
-    prot = PROT_READ | PROT_WRITE;
-#elif defined(CONFIG_DARWIN)
-    /* Applicable to both iOS and macOS (Apple Silicon). */
-    if (!splitwx) {
-        flags |= MAP_JIT;
-    }
-#endif
-
-    return alloc_code_gen_buffer_anon(size, prot, flags, errp);
-}
-#endif /* USE_STATIC_CODE_GEN_BUFFER, WIN32, POSIX */
-
 static bool tb_cmp(const void *ap, const void *bp)
 {
     const TranslationBlock *a = ap;
@@ -XXX,XX +XXX,XX @@ static void tb_htable_init(void)
    size. */
 void tcg_exec_init(unsigned long tb_size, int splitwx)
 {
-    bool ok;
-
     tcg_allowed = true;
     tcg_context_init(&tcg_init_ctx);
     page_init();
     tb_htable_init();
-
-    ok = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
-                               splitwx, &error_fatal);
-    assert(ok);
-
-    /* TODO: allocating regions is hand-in-glove with code_gen_buffer. */
-    tcg_region_init();
+    tcg_region_init(tb_size, splitwx);
 
 #if defined(CONFIG_SOFTMMU)
     /* There's no guest base to take into account, so go ahead and
diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@
  */
 
 #include "qemu/osdep.h"
+#include "qemu/units.h"
+#include "qapi/error.h"
 #include "exec/exec-all.h"
 #include "tcg/tcg.h"
 #if !defined(CONFIG_USER_ONLY)
@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(void)
 }
 #endif
 
+/*
+ * Minimum size of the code gen buffer.  This number is randomly chosen,
+ * but not so small that we can't have a fair number of TB's live.
+ */
+#define MIN_CODE_GEN_BUFFER_SIZE     (1 * MiB)
+
+/*
+ * Maximum size of the code gen buffer we'd like to use.  Unless otherwise
+ * indicated, this is constrained by the range of direct branches on the
+ * host cpu, as used by the TCG implementation of goto_tb.
+ */
+#if defined(__x86_64__)
+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
+#elif defined(__sparc__)
+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
+#elif defined(__powerpc64__)
+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
+#elif defined(__powerpc__)
+# define MAX_CODE_GEN_BUFFER_SIZE  (32 * MiB)
+#elif defined(__aarch64__)
+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
+#elif defined(__s390x__)
+  /* We have a +- 4GB range on the branches; leave some slop.  */
+# define MAX_CODE_GEN_BUFFER_SIZE  (3 * GiB)
+#elif defined(__mips__)
+  /*
+   * We have a 256MB branch region, but leave room to make sure the
+   * main executable is also within that region.
+   */
+# define MAX_CODE_GEN_BUFFER_SIZE  (128 * MiB)
+#else
+# define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
+#endif
+
+#if TCG_TARGET_REG_BITS == 32
+#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (32 * MiB)
+#ifdef CONFIG_USER_ONLY
+/*
+ * For user mode on smaller 32 bit systems we may run into trouble
+ * allocating big chunks of data in the right place. On these systems
+ * we utilise a static code generation buffer directly in the binary.
+ */
+#define USE_STATIC_CODE_GEN_BUFFER
+#endif
+#else /* TCG_TARGET_REG_BITS == 64 */
+#ifdef CONFIG_USER_ONLY
+/*
+ * As user-mode emulation typically means running multiple instances
+ * of the translator don't go too nuts with our default code gen
+ * buffer lest we make things too hard for the OS.
+ */
+#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (128 * MiB)
+#else
+/*
+ * We expect most system emulation to run one or two guests per host.
+ * Users running large scale system emulation may want to tweak their
+ * runtime setup via the tb-size control on the command line.
+ */
+#define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (1 * GiB)
+#endif
+#endif
+
+#define DEFAULT_CODE_GEN_BUFFER_SIZE \
+  (DEFAULT_CODE_GEN_BUFFER_SIZE_1 < MAX_CODE_GEN_BUFFER_SIZE \
+   ? DEFAULT_CODE_GEN_BUFFER_SIZE_1 : MAX_CODE_GEN_BUFFER_SIZE)
+
+static size_t size_code_gen_buffer(size_t tb_size)
+{
+    /* Size the buffer.  */
+    if (tb_size == 0) {
+        size_t phys_mem = qemu_get_host_physmem();
+        if (phys_mem == 0) {
+            tb_size = DEFAULT_CODE_GEN_BUFFER_SIZE;
+        } else {
+            tb_size = MIN(DEFAULT_CODE_GEN_BUFFER_SIZE, phys_mem / 8);
+        }
+    }
+    if (tb_size < MIN_CODE_GEN_BUFFER_SIZE) {
+        tb_size = MIN_CODE_GEN_BUFFER_SIZE;
+    }
+    if (tb_size > MAX_CODE_GEN_BUFFER_SIZE) {
+        tb_size = MAX_CODE_GEN_BUFFER_SIZE;
+    }
+    return tb_size;
+}
+
+#ifdef __mips__
+/*
+ * In order to use J and JAL within the code_gen_buffer, we require
+ * that the buffer not cross a 256MB boundary.
+ */
+static inline bool cross_256mb(void *addr, size_t size)
+{
+    return ((uintptr_t)addr ^ ((uintptr_t)addr + size)) & ~0x0ffffffful;
+}
+
+/*
+ * We weren't able to allocate a buffer without crossing that boundary,
+ * so make do with the larger portion of the buffer that doesn't cross.
+ * Returns the new base of the buffer, and adjusts code_gen_buffer_size.
+ */
+static inline void *split_cross_256mb(void *buf1, size_t size1)
+{
+    void *buf2 = (void *)(((uintptr_t)buf1 + size1) & ~0x0ffffffful);
+    size_t size2 = buf1 + size1 - buf2;
+
+    size1 = buf2 - buf1;
+    if (size1 < size2) {
+        size1 = size2;
+        buf1 = buf2;
+    }
+
+    tcg_ctx->code_gen_buffer_size = size1;
+    return buf1;
+}
+#endif
+
+#ifdef USE_STATIC_CODE_GEN_BUFFER
+static uint8_t static_code_gen_buffer[DEFAULT_CODE_GEN_BUFFER_SIZE]
+    __attribute__((aligned(CODE_GEN_ALIGN)));
+
+static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
+{
+    void *buf, *end;
+    size_t size;
+
+    if (splitwx > 0) {
+        error_setg(errp, "jit split-wx not supported");
+        return false;
+    }
+
+    /* page-align the beginning and end of the buffer */
+    buf = static_code_gen_buffer;
+    end = static_code_gen_buffer + sizeof(static_code_gen_buffer);
+    buf = QEMU_ALIGN_PTR_UP(buf, qemu_real_host_page_size);
+    end = QEMU_ALIGN_PTR_DOWN(end, qemu_real_host_page_size);
+
+    size = end - buf;
+
+    /* Honor a command-line option limiting the size of the buffer.  */
+    if (size > tb_size) {
+        size = QEMU_ALIGN_DOWN(tb_size, qemu_real_host_page_size);
+    }
+    tcg_ctx->code_gen_buffer_size = size;
+
+#ifdef __mips__
+    if (cross_256mb(buf, size)) {
+        buf = split_cross_256mb(buf, size);
+        size = tcg_ctx->code_gen_buffer_size;
+    }
+#endif
+
+    if (qemu_mprotect_rwx(buf, size)) {
+        error_setg_errno(errp, errno, "mprotect of jit buffer");
+        return false;
+    }
+    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
+
+    tcg_ctx->code_gen_buffer = buf;
+    return true;
+}
+#elif defined(_WIN32)
+static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
+{
+    void *buf;
+
+    if (splitwx > 0) {
+        error_setg(errp, "jit split-wx not supported");
+        return false;
+    }
+
+    buf = VirtualAlloc(NULL, size, MEM_RESERVE | MEM_COMMIT,
+                             PAGE_EXECUTE_READWRITE);
+    if (buf == NULL) {
+        error_setg_win32(errp, GetLastError(),
+                         "allocate %zu bytes for jit buffer", size);
+        return false;
+    }
+
+    tcg_ctx->code_gen_buffer = buf;
+    tcg_ctx->code_gen_buffer_size = size;
+    return true;
+}
+#else
+static bool alloc_code_gen_buffer_anon(size_t size, int prot,
+                                       int flags, Error **errp)
+{
+    void *buf;
+
+    buf = mmap(NULL, size, prot, flags, -1, 0);
+    if (buf == MAP_FAILED) {
+        error_setg_errno(errp, errno,
+                         "allocate %zu bytes for jit buffer", size);
+        return false;
+    }
+    tcg_ctx->code_gen_buffer_size = size;
+
+#ifdef __mips__
+    if (cross_256mb(buf, size)) {
+        /*
+         * Try again, with the original still mapped, to avoid re-acquiring
+         * the same 256mb crossing.
+         */
+        size_t size2;
+        void *buf2 = mmap(NULL, size, prot, flags, -1, 0);
+        switch ((int)(buf2 != MAP_FAILED)) {
+        case 1:
+            if (!cross_256mb(buf2, size)) {
+                /* Success!  Use the new buffer.  */
+                munmap(buf, size);
+                break;
+            }
+            /* Failure.  Work with what we had.  */
+            munmap(buf2, size);
+            /* fallthru */
+        default:
+            /* Split the original buffer.  Free the smaller half.  */
+            buf2 = split_cross_256mb(buf, size);
+            size2 = tcg_ctx->code_gen_buffer_size;
+            if (buf == buf2) {
+                munmap(buf + size2, size - size2);
+            } else {
+                munmap(buf, size - size2);
+            }
+            size = size2;
+            break;
+        }
+        buf = buf2;
+    }
+#endif
+
+    /* Request large pages for the buffer.  */
+    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
+
+    tcg_ctx->code_gen_buffer = buf;
+    return true;
+}
+
+#ifndef CONFIG_TCG_INTERPRETER
+#ifdef CONFIG_POSIX
+#include "qemu/memfd.h"
+
+static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
+{
+    void *buf_rw = NULL, *buf_rx = MAP_FAILED;
+    int fd = -1;
+
+#ifdef __mips__
+    /* Find space for the RX mapping, vs the 256MiB regions. */
+    if (!alloc_code_gen_buffer_anon(size, PROT_NONE,
+                                    MAP_PRIVATE | MAP_ANONYMOUS |
+                                    MAP_NORESERVE, errp)) {
+        return false;
+    }
+    /* The size of the mapping may have been adjusted. */
+    size = tcg_ctx->code_gen_buffer_size;
+    buf_rx = tcg_ctx->code_gen_buffer;
+#endif
+
+    buf_rw = qemu_memfd_alloc("tcg-jit", size, 0, &fd, errp);
+    if (buf_rw == NULL) {
+        goto fail;
+    }
+
+#ifdef __mips__
+    void *tmp = mmap(buf_rx, size, PROT_READ | PROT_EXEC,
+                     MAP_SHARED | MAP_FIXED, fd, 0);
+    if (tmp != buf_rx) {
+        goto fail_rx;
+    }
+#else
+    buf_rx = mmap(NULL, size, PROT_READ | PROT_EXEC, MAP_SHARED, fd, 0);
+    if (buf_rx == MAP_FAILED) {
+        goto fail_rx;
+    }
+#endif
+
+    close(fd);
+    tcg_ctx->code_gen_buffer = buf_rw;
+    tcg_ctx->code_gen_buffer_size = size;
+    tcg_splitwx_diff = buf_rx - buf_rw;
+
+    /* Request large pages for the buffer and the splitwx.  */
+    qemu_madvise(buf_rw, size, QEMU_MADV_HUGEPAGE);
+    qemu_madvise(buf_rx, size, QEMU_MADV_HUGEPAGE);
+    return true;
+
+ fail_rx:
+    error_setg_errno(errp, errno, "failed to map shared memory for execute");
+ fail:
+    if (buf_rx != MAP_FAILED) {
+        munmap(buf_rx, size);
+    }
+    if (buf_rw) {
+        munmap(buf_rw, size);
+    }
+    if (fd >= 0) {
+        close(fd);
+    }
+    return false;
+}
+#endif /* CONFIG_POSIX */
+
+#ifdef CONFIG_DARWIN
+#include <mach/mach.h>
+
+extern kern_return_t mach_vm_remap(vm_map_t target_task,
+                                   mach_vm_address_t *target_address,
+                                   mach_vm_size_t size,
+                                   mach_vm_offset_t mask,
+                                   int flags,
+                                   vm_map_t src_task,
+                                   mach_vm_address_t src_address,
+                                   boolean_t copy,
+                                   vm_prot_t *cur_protection,
+                                   vm_prot_t *max_protection,
+                                   vm_inherit_t inheritance);
+
+static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
+{
+    kern_return_t ret;
+    mach_vm_address_t buf_rw, buf_rx;
+    vm_prot_t cur_prot, max_prot;
+
+    /* Map the read-write portion via normal anon memory. */
+    if (!alloc_code_gen_buffer_anon(size, PROT_READ | PROT_WRITE,
+                                    MAP_PRIVATE | MAP_ANONYMOUS, errp)) {
+        return false;
+    }
+
+    buf_rw = (mach_vm_address_t)tcg_ctx->code_gen_buffer;
+    buf_rx = 0;
+    ret = mach_vm_remap(mach_task_self(),
+                        &buf_rx,
+                        size,
+                        0,
+                        VM_FLAGS_ANYWHERE,
+                        mach_task_self(),
+                        buf_rw,
+                        false,
+                        &cur_prot,
+                        &max_prot,
+                        VM_INHERIT_NONE);
+    if (ret != KERN_SUCCESS) {
+        /* TODO: Convert "ret" to a human readable error message. */
+        error_setg(errp, "vm_remap for jit splitwx failed");
+        munmap((void *)buf_rw, size);
+        return false;
+    }
+
+    if (mprotect((void *)buf_rx, size, PROT_READ | PROT_EXEC) != 0) {
+        error_setg_errno(errp, errno, "mprotect for jit splitwx");
+        munmap((void *)buf_rx, size);
+        munmap((void *)buf_rw, size);
+        return false;
+    }
+
+    tcg_splitwx_diff = buf_rx - buf_rw;
+    return true;
+}
+#endif /* CONFIG_DARWIN */
+#endif /* CONFIG_TCG_INTERPRETER */
+
+static bool alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
+{
+#ifndef CONFIG_TCG_INTERPRETER
+# ifdef CONFIG_DARWIN
+    return alloc_code_gen_buffer_splitwx_vmremap(size, errp);
+# endif
+# ifdef CONFIG_POSIX
+    return alloc_code_gen_buffer_splitwx_memfd(size, errp);
+# endif
+#endif
+    error_setg(errp, "jit split-wx not supported");
+    return false;
+}
+
+static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
+{
+    ERRP_GUARD();
+    int prot, flags;
+
+    if (splitwx) {
+        if (alloc_code_gen_buffer_splitwx(size, errp)) {
+            return true;
+        }
+        /*
+         * If splitwx force-on (1), fail;
+         * if splitwx default-on (-1), fall through to splitwx off.
+         */
+        if (splitwx > 0) {
+            return false;
+        }
+        error_free_or_abort(errp);
+    }
+
+    prot = PROT_READ | PROT_WRITE | PROT_EXEC;
+    flags = MAP_PRIVATE | MAP_ANONYMOUS;
+#ifdef CONFIG_TCG_INTERPRETER
+    /* The tcg interpreter does not need execute permission. */
+    prot = PROT_READ | PROT_WRITE;
+#elif defined(CONFIG_DARWIN)
+    /* Applicable to both iOS and macOS (Apple Silicon). */
+    if (!splitwx) {
+        flags |= MAP_JIT;
+    }
+#endif
+
+    return alloc_code_gen_buffer_anon(size, prot, flags, errp);
+}
+#endif /* USE_STATIC_CODE_GEN_BUFFER, WIN32, POSIX */
+
 /*
  * Initializes region partitioning.
  *
@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(void)
  * in practice. Multi-threaded guests share most if not all of their translated
  * code, which makes parallel code generation less appealing than in softmmu.
  */
-void tcg_region_init(void)
+void tcg_region_init(size_t tb_size, int splitwx)
 {
-    void *buf = tcg_init_ctx.code_gen_buffer;
-    void *aligned;
-    size_t size = tcg_init_ctx.code_gen_buffer_size;
-    size_t page_size = qemu_real_host_page_size;
+    void *buf, *aligned;
+    size_t size;
+    size_t page_size;
     size_t region_size;
     size_t n_regions;
     size_t i;
+    bool ok;
 
+    ok = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
+                               splitwx, &error_fatal);
+    assert(ok);
+
+    buf = tcg_init_ctx.code_gen_buffer;
+    size = tcg_init_ctx.code_gen_buffer_size;
+    page_size = qemu_real_host_page_size;
     n_regions = tcg_n_regions();
 
     /* The first region will be 'aligned - buf' bytes larger than the others */
-- 
2.25.1

We shortly want to use tcg_init for something else.
Since the hook is called init_machine, match that.

diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/tcg-all.c
+++ b/accel/tcg/tcg-all.c
@@ -XXX,XX +XXX,XX @@ static void tcg_accel_instance_init(Object *obj)
 
 bool mttcg_enabled;
 
-static int tcg_init(MachineState *ms)
+static int tcg_init_machine(MachineState *ms)
 {
     TCGState *s = TCG_STATE(current_accel());
 
@@ -XXX,XX +XXX,XX @@ static void tcg_accel_class_init(ObjectClass *oc, void *data)
 {
     AccelClass *ac = ACCEL_CLASS(oc);
     ac->name = "tcg";
-    ac->init_machine = tcg_init;
+    ac->init_machine = tcg_init_machine;
     ac->allowed = &tcg_allowed;
 
     object_class_property_add_str(oc, "thread",
-- 
2.25.1

Perform both tcg_context_init and tcg_region_init.
Do not leave this split to the caller.

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/tcg/tcg.h         | 3 +--
 tcg/tcg-internal.h        | 1 +
 accel/tcg/translate-all.c | 3 +--
 tcg/tcg.c                 | 9 ++++++++-
 4 files changed, 11 insertions(+), 5 deletions(-)

There is only one caller, and shortly we will need access
to the MachineState, which tcg_init_machine already has.

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 accel/tcg/internal.h      |  2 ++
 include/sysemu/tcg.h      |  2 --
 accel/tcg/tcg-all.c       | 16 +++++++++++++++-
 accel/tcg/translate-all.c | 21 ++-------------------
 bsd-user/main.c           |  2 +-
 5 files changed, 20 insertions(+), 23 deletions(-)

diff --git a/accel/tcg/internal.h b/accel/tcg/internal.h
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/internal.h
+++ b/accel/tcg/internal.h
@@ -XXX,XX +XXX,XX @@ TranslationBlock *tb_gen_code(CPUState *cpu, target_ulong pc,
                               int cflags);
 
 void QEMU_NORETURN cpu_io_recompile(CPUState *cpu, uintptr_t retaddr);
+void page_init(void);
+void tb_htable_init(void);
 
 #endif /* ACCEL_TCG_INTERNAL_H */
diff --git a/include/sysemu/tcg.h b/include/sysemu/tcg.h
index XXXXXXX..XXXXXXX 100644
--- a/include/sysemu/tcg.h
+++ b/include/sysemu/tcg.h
@@ -XXX,XX +XXX,XX @@
 #ifndef SYSEMU_TCG_H
 #define SYSEMU_TCG_H
 
-void tcg_exec_init(unsigned long tb_size, int splitwx);
-
 #ifdef CONFIG_TCG
 extern bool tcg_allowed;
 #define tcg_enabled() (tcg_allowed)
diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/tcg-all.c
+++ b/accel/tcg/tcg-all.c
@@ -XXX,XX +XXX,XX @@
 #include "qemu/error-report.h"
 #include "qemu/accel.h"
 #include "qapi/qapi-builtin-visit.h"
+#include "internal.h"
 
 struct TCGState {
     AccelState parent_obj;
@@ -XXX,XX +XXX,XX @@ static int tcg_init_machine(MachineState *ms)
 {
     TCGState *s = TCG_STATE(current_accel());
 
-    tcg_exec_init(s->tb_size * 1024 * 1024, s->splitwx_enabled);
+    tcg_allowed = true;
     mttcg_enabled = s->mttcg_enabled;
+
+    page_init();
+    tb_htable_init();
+    tcg_init(s->tb_size * 1024 * 1024, s->splitwx_enabled);
+
+#if defined(CONFIG_SOFTMMU)
+    /*
+     * There's no guest base to take into account, so go ahead and
+     * initialize the prologue now.
+     */
+    tcg_prologue_init(tcg_ctx);
+#endif
+
     return 0;
 }
 
diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/translate-all.c
+++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@ bool cpu_restore_state(CPUState *cpu, uintptr_t host_pc, bool will_exit)
     return false;
 }
 
-static void page_init(void)
+void page_init(void)
 {
     page_size_init();
     page_table_config_init();
@@ -XXX,XX +XXX,XX @@ static bool tb_cmp(const void *ap, const void *bp)
         a->page_addr[1] == b->page_addr[1];
 }
 
-static void tb_htable_init(void)
+void tb_htable_init(void)
 {
     unsigned int mode = QHT_MODE_AUTO_RESIZE;
 
     qht_init(&tb_ctx.htable, tb_cmp, CODE_GEN_HTABLE_SIZE, mode);
 }
 
-/* Must be called before using the QEMU cpus. 'tb_size' is the size
-   (in bytes) allocated to the translation buffer. Zero means default
-   size. */
-void tcg_exec_init(unsigned long tb_size, int splitwx)
-{
-    tcg_allowed = true;
-    page_init();
-    tb_htable_init();
-    tcg_init(tb_size, splitwx);
-
-#if defined(CONFIG_SOFTMMU)
-    /* There's no guest base to take into account, so go ahead and
-       initialize the prologue now.  */
-    tcg_prologue_init(tcg_ctx);
-#endif
-}
-
 /* call with @p->lock held */
 static inline void invalidate_page_bitmap(PageDesc *p)
 {
diff --git a/bsd-user/main.c b/bsd-user/main.c
index XXXXXXX..XXXXXXX 100644
--- a/bsd-user/main.c
+++ b/bsd-user/main.c
@@ -XXX,XX +XXX,XX @@ int main(int argc, char **argv)
     envlist_free(envlist);
 
     /*
-     * Now that page sizes are configured in tcg_exec_init() we can do
+     * Now that page sizes are configured we can do
      * proper page alignment for guest_base.
      */
     guest_base = HOST_PAGE_ALIGN(guest_base);
-- 
2.25.1

Start removing the include of hw/boards.h from tcg/.
Pass down the max_cpus value from tcg_init_machine,
where we have the MachineState already.

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/tcg/tcg.h   |  2 +-
 tcg/tcg-internal.h  |  2 +-
 accel/tcg/tcg-all.c | 10 +++++++++-
 tcg/region.c        | 32 +++++++++++---------------------
 tcg/tcg.c           | 10 ++++------
 5 files changed, 26 insertions(+), 30 deletions(-)

diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
index XXXXXXX..XXXXXXX 100644
--- a/include/tcg/tcg.h
+++ b/include/tcg/tcg.h
@@ -XXX,XX +XXX,XX @@ static inline void *tcg_malloc(int size)
     }
 }
 
-void tcg_init(size_t tb_size, int splitwx);
+void tcg_init(size_t tb_size, int splitwx, unsigned max_cpus);
 void tcg_register_thread(void);
 void tcg_prologue_init(TCGContext *s);
 void tcg_func_start(TCGContext *s);
diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg-internal.h
+++ b/tcg/tcg-internal.h
@@ -XXX,XX +XXX,XX @@
 extern TCGContext **tcg_ctxs;
 extern unsigned int n_tcg_ctxs;
 
-void tcg_region_init(size_t tb_size, int splitwx);
+void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus);
 bool tcg_region_alloc(TCGContext *s);
 void tcg_region_initial_alloc(TCGContext *s);
 void tcg_region_prologue_set(TCGContext *s);
diff --git a/accel/tcg/tcg-all.c b/accel/tcg/tcg-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/tcg-all.c
+++ b/accel/tcg/tcg-all.c
@@ -XXX,XX +XXX,XX @@
 #include "qemu/accel.h"
 #include "qapi/qapi-builtin-visit.h"
 #include "qemu/units.h"
+#if !defined(CONFIG_USER_ONLY)
+#include "hw/boards.h"
+#endif
 #include "internal.h"
 
 struct TCGState {
@@ -XXX,XX +XXX,XX @@ bool mttcg_enabled;
 static int tcg_init_machine(MachineState *ms)
 {
     TCGState *s = TCG_STATE(current_accel());
+#ifdef CONFIG_USER_ONLY
+    unsigned max_cpus = 1;
+#else
+    unsigned max_cpus = ms->smp.max_cpus;
+#endif
 
     tcg_allowed = true;
     mttcg_enabled = s->mttcg_enabled;
 
     page_init();
     tb_htable_init();
-    tcg_init(s->tb_size * MiB, s->splitwx_enabled);
+    tcg_init(s->tb_size * MiB, s->splitwx_enabled, max_cpus);
 
 #if defined(CONFIG_SOFTMMU)
     /*
diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@
 #include "qapi/error.h"
 #include "exec/exec-all.h"
 #include "tcg/tcg.h"
-#if !defined(CONFIG_USER_ONLY)
-#include "hw/boards.h"
-#endif
 #include "tcg-internal.h"
 
 
@@ -XXX,XX +XXX,XX @@ void tcg_region_reset_all(void)
     tcg_region_tree_reset_all();
 }
 
+static size_t tcg_n_regions(unsigned max_cpus)
+{
 #ifdef CONFIG_USER_ONLY
-static size_t tcg_n_regions(void)
-{
     return 1;
-}
 #else
-/*
- * It is likely that some vCPUs will translate more code than others, so we
- * first try to set more regions than max_cpus, with those regions being of
- * reasonable size. If that's not possible we make do by evenly dividing
- * the code_gen_buffer among the vCPUs.
- */
-static size_t tcg_n_regions(void)
-{
+    /*
+     * It is likely that some vCPUs will translate more code than others,
+     * so we first try to set more regions than max_cpus, with those regions
+     * being of reasonable size. If that's not possible we make do by evenly
+     * dividing the code_gen_buffer among the vCPUs.
+     */
     size_t i;
 
     /* Use a single region if all we have is one vCPU thread */
-#if !defined(CONFIG_USER_ONLY)
-    MachineState *ms = MACHINE(qdev_get_machine());
-    unsigned int max_cpus = ms->smp.max_cpus;
-#endif
     if (max_cpus == 1 || !qemu_tcg_mttcg_enabled()) {
         return 1;
     }
@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(void)
     }
     /* If we can't, then just allocate one region per vCPU thread */
     return max_cpus;
-}
 #endif
+}
 
 /*
  * Minimum size of the code gen buffer.  This number is randomly chosen,
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
  * in practice. Multi-threaded guests share most if not all of their translated
  * code, which makes parallel code generation less appealing than in softmmu.
  */
-void tcg_region_init(size_t tb_size, int splitwx)
+void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
 {
     void *buf, *aligned;
     size_t size;
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx)
     buf = tcg_init_ctx.code_gen_buffer;
     size = tcg_init_ctx.code_gen_buffer_size;
     page_size = qemu_real_host_page_size;
-    n_regions = tcg_n_regions();
+    n_regions = tcg_n_regions(max_cpus);
 
     /* The first region will be 'aligned - buf' bytes larger than the others */
     aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ static void process_op_defs(TCGContext *s);
 static TCGTemp *tcg_global_reg_new_internal(TCGContext *s, TCGType type,
                                             TCGReg reg, const char *name);
 
-static void tcg_context_init(void)
+static void tcg_context_init(unsigned max_cpus)
 {
     TCGContext *s = &tcg_init_ctx;
     int op, total_args, n, i;
@@ -XXX,XX +XXX,XX @@ static void tcg_context_init(void)
     tcg_ctxs = &tcg_ctx;
     n_tcg_ctxs = 1;
 #else
-    MachineState *ms = MACHINE(qdev_get_machine());
-    unsigned int max_cpus = ms->smp.max_cpus;
     tcg_ctxs = g_new(TCGContext *, max_cpus);
 #endif
 
@@ -XXX,XX +XXX,XX @@ static void tcg_context_init(void)
     cpu_env = temp_tcgv_ptr(ts);
 }
 
-void tcg_init(size_t tb_size, int splitwx)
+void tcg_init(size_t tb_size, int splitwx, unsigned max_cpus)
 {
-    tcg_context_init();
-    tcg_region_init(tb_size, splitwx);
+    tcg_context_init(max_cpus);
+    tcg_region_init(tb_size, splitwx, max_cpus);
 }
 
 /*
-- 
2.25.1

Finish the divorce of tcg/ from hw/, and do not take
the max cpu value from MachineState; just remember what
we were passed in tcg_init.

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/tcg-internal.h |  3 ++-
 tcg/region.c       |  6 +++---
 tcg/tcg.c          | 23 ++++++++++-------------
 3 files changed, 15 insertions(+), 17 deletions(-)

diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg-internal.h
+++ b/tcg/tcg-internal.h
@@ -XXX,XX +XXX,XX @@
 #define TCG_HIGHWATER 1024
 
 extern TCGContext **tcg_ctxs;
-extern unsigned int n_tcg_ctxs;
+extern unsigned int tcg_cur_ctxs;
+extern unsigned int tcg_max_ctxs;
 
 void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus);
 bool tcg_region_alloc(TCGContext *s);
diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ void tcg_region_initial_alloc(TCGContext *s)
 /* Call from a safe-work context */
 void tcg_region_reset_all(void)
 {
-    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
+    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
     unsigned int i;
 
     qemu_mutex_lock(&region.lock);
@@ -XXX,XX +XXX,XX @@ void tcg_region_prologue_set(TCGContext *s)
  */
 size_t tcg_code_size(void)
 {
-    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
+    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
     unsigned int i;
     size_t total;
 
@@ -XXX,XX +XXX,XX @@ size_t tcg_code_capacity(void)
 
 size_t tcg_tb_phys_invalidate_count(void)
 {
-    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
+    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
     unsigned int i;
     size_t total = 0;
 
diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@
 #define NO_CPU_IO_DEFS
 
 #include "exec/exec-all.h"
-
-#if !defined(CONFIG_USER_ONLY)
-#include "hw/boards.h"
-#endif
-
 #include "tcg/tcg-op.h"
 
 #if UINTPTR_MAX == UINT32_MAX
@@ -XXX,XX +XXX,XX @@ static int tcg_out_ldst_finalize(TCGContext *s);
 #endif
 
 TCGContext **tcg_ctxs;
-unsigned int n_tcg_ctxs;
+unsigned int tcg_cur_ctxs;
+unsigned int tcg_max_ctxs;
 TCGv_env cpu_env = 0;
 const void *tcg_code_gen_epilogue;
 uintptr_t tcg_splitwx_diff;
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
 #else
 void tcg_register_thread(void)
 {
-    MachineState *ms = MACHINE(qdev_get_machine());
     TCGContext *s = g_malloc(sizeof(*s));
     unsigned int i, n;
 
@@ -XXX,XX +XXX,XX @@ void tcg_register_thread(void)
     }
 
     /* Claim an entry in tcg_ctxs */
-    n = qatomic_fetch_inc(&n_tcg_ctxs);
-    g_assert(n < ms->smp.max_cpus);
+    n = qatomic_fetch_inc(&tcg_cur_ctxs);
+    g_assert(n < tcg_max_ctxs);
     qatomic_set(&tcg_ctxs[n], s);
 
     if (n > 0) {
@@ -XXX,XX +XXX,XX @@ static void tcg_context_init(unsigned max_cpus)
      */
 #ifdef CONFIG_USER_ONLY
     tcg_ctxs = &tcg_ctx;
-    n_tcg_ctxs = 1;
+    tcg_cur_ctxs = 1;
+    tcg_max_ctxs = 1;
 #else
-    tcg_ctxs = g_new(TCGContext *, max_cpus);
+    tcg_max_ctxs = max_cpus;
+    tcg_ctxs = g_new0(TCGContext *, max_cpus);
 #endif
 
     tcg_debug_assert(!tcg_regset_test_reg(s->reserved_regs, TCG_AREG0));
@@ -XXX,XX +XXX,XX @@ static void tcg_reg_alloc_call(TCGContext *s, TCGOp *op)
 static inline
 void tcg_profile_snapshot(TCGProfile *prof, bool counters, bool table)
 {
-    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
+    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
     unsigned int i;
 
     for (i = 0; i < n_ctxs; i++) {
@@ -XXX,XX +XXX,XX @@ void tcg_dump_op_count(void)
 
 int64_t tcg_cpu_exec_time(void)
 {
-    unsigned int n_ctxs = qatomic_read(&n_tcg_ctxs);
+    unsigned int n_ctxs = qatomic_read(&tcg_cur_ctxs);
     unsigned int i;
     int64_t ret = 0;
 
-- 
2.25.1

Remove the ifdef ladder and move each define into the
appropriate header file.

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/aarch64/tcg-target.h |  1 +
 tcg/arm/tcg-target.h     |  1 +
 tcg/i386/tcg-target.h    |  2 ++
 tcg/mips/tcg-target.h    |  6 ++++++
 tcg/ppc/tcg-target.h     |  2 ++
 tcg/riscv/tcg-target.h   |  1 +
 tcg/s390/tcg-target.h    |  3 +++
 tcg/sparc/tcg-target.h   |  1 +
 tcg/tci/tcg-target.h     |  1 +
 tcg/region.c             | 33 +++++----------------------------
 10 files changed, 23 insertions(+), 28 deletions(-)

diff --git a/tcg/aarch64/tcg-target.h b/tcg/aarch64/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/aarch64/tcg-target.h
+++ b/tcg/aarch64/tcg-target.h
@@ -XXX,XX +XXX,XX @@
 
 #define TCG_TARGET_INSN_UNIT_SIZE  4
 #define TCG_TARGET_TLB_DISPLACEMENT_BITS 24
+#define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
 #undef TCG_TARGET_STACK_GROWSUP
 
 typedef enum {
diff --git a/tcg/arm/tcg-target.h b/tcg/arm/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/arm/tcg-target.h
+++ b/tcg/arm/tcg-target.h
@@ -XXX,XX +XXX,XX @@ extern int arm_arch;
 #undef TCG_TARGET_STACK_GROWSUP
 #define TCG_TARGET_INSN_UNIT_SIZE 4
 #define TCG_TARGET_TLB_DISPLACEMENT_BITS 16
+#define MAX_CODE_GEN_BUFFER_SIZE  UINT32_MAX
 
 typedef enum {
     TCG_REG_R0 = 0,
diff --git a/tcg/i386/tcg-target.h b/tcg/i386/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/i386/tcg-target.h
+++ b/tcg/i386/tcg-target.h
@@ -XXX,XX +XXX,XX @@
 #ifdef __x86_64__
 # define TCG_TARGET_REG_BITS  64
 # define TCG_TARGET_NB_REGS   32
+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
 #else
 # define TCG_TARGET_REG_BITS  32
 # define TCG_TARGET_NB_REGS   24
+# define MAX_CODE_GEN_BUFFER_SIZE  UINT32_MAX
 #endif
 
 typedef enum {
diff --git a/tcg/mips/tcg-target.h b/tcg/mips/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/mips/tcg-target.h
+++ b/tcg/mips/tcg-target.h
@@ -XXX,XX +XXX,XX @@
 #define TCG_TARGET_TLB_DISPLACEMENT_BITS 16
 #define TCG_TARGET_NB_REGS 32
 
+/*
+ * We have a 256MB branch region, but leave room to make sure the
+ * main executable is also within that region.
+ */
+#define MAX_CODE_GEN_BUFFER_SIZE  (128 * MiB)
+
 typedef enum {
     TCG_REG_ZERO = 0,
     TCG_REG_AT,
diff --git a/tcg/ppc/tcg-target.h b/tcg/ppc/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/ppc/tcg-target.h
+++ b/tcg/ppc/tcg-target.h
@@ -XXX,XX +XXX,XX @@
 
 #ifdef _ARCH_PPC64
 # define TCG_TARGET_REG_BITS  64
+# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
 #else
 # define TCG_TARGET_REG_BITS  32
+# define MAX_CODE_GEN_BUFFER_SIZE  (32 * MiB)
 #endif
 
 #define TCG_TARGET_NB_REGS 64
diff --git a/tcg/riscv/tcg-target.h b/tcg/riscv/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/riscv/tcg-target.h
+++ b/tcg/riscv/tcg-target.h
@@ -XXX,XX +XXX,XX @@
 #define TCG_TARGET_INSN_UNIT_SIZE 4
 #define TCG_TARGET_TLB_DISPLACEMENT_BITS 20
 #define TCG_TARGET_NB_REGS 32
+#define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
 
 typedef enum {
     TCG_REG_ZERO,
diff --git a/tcg/s390/tcg-target.h b/tcg/s390/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/s390/tcg-target.h
+++ b/tcg/s390/tcg-target.h
@@ -XXX,XX +XXX,XX @@
 #define TCG_TARGET_INSN_UNIT_SIZE 2
 #define TCG_TARGET_TLB_DISPLACEMENT_BITS 19
 
+/* We have a +- 4GB range on the branches; leave some slop.  */
+#define MAX_CODE_GEN_BUFFER_SIZE  (3 * GiB)
+
 typedef enum TCGReg {
     TCG_REG_R0 = 0,
     TCG_REG_R1,
diff --git a/tcg/sparc/tcg-target.h b/tcg/sparc/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/sparc/tcg-target.h
+++ b/tcg/sparc/tcg-target.h
@@ -XXX,XX +XXX,XX @@
 #define TCG_TARGET_INSN_UNIT_SIZE 4
 #define TCG_TARGET_TLB_DISPLACEMENT_BITS 32
 #define TCG_TARGET_NB_REGS 32
+#define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
 
 typedef enum {
     TCG_REG_G0 = 0,
diff --git a/tcg/tci/tcg-target.h b/tcg/tci/tcg-target.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tci/tcg-target.h
+++ b/tcg/tci/tcg-target.h
@@ -XXX,XX +XXX,XX @@
 #define TCG_TARGET_INTERPRETER 1
 #define TCG_TARGET_INSN_UNIT_SIZE 1
 #define TCG_TARGET_TLB_DISPLACEMENT_BITS 32
+#define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
 
 #if UINTPTR_MAX == UINT32_MAX
 # define TCG_TARGET_REG_BITS 32
diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(unsigned max_cpus)
 /*
  * Minimum size of the code gen buffer.  This number is randomly chosen,
  * but not so small that we can't have a fair number of TB's live.
+ *
+ * Maximum size, MAX_CODE_GEN_BUFFER_SIZE, is defined in tcg-target.h.
+ * Unless otherwise indicated, this is constrained by the range of
+ * direct branches on the host cpu, as used by the TCG implementation
+ * of goto_tb.
  */
 #define MIN_CODE_GEN_BUFFER_SIZE     (1 * MiB)
 
-/*
- * Maximum size of the code gen buffer we'd like to use.  Unless otherwise
- * indicated, this is constrained by the range of direct branches on the
- * host cpu, as used by the TCG implementation of goto_tb.
- */
-#if defined(__x86_64__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
-#elif defined(__sparc__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
-#elif defined(__powerpc64__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
-#elif defined(__powerpc__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (32 * MiB)
-#elif defined(__aarch64__)
-# define MAX_CODE_GEN_BUFFER_SIZE  (2 * GiB)
-#elif defined(__s390x__)
-  /* We have a +- 4GB range on the branches; leave some slop.  */
-# define MAX_CODE_GEN_BUFFER_SIZE  (3 * GiB)
-#elif defined(__mips__)
-  /*
-   * We have a 256MB branch region, but leave room to make sure the
-   * main executable is also within that region.
-   */
-# define MAX_CODE_GEN_BUFFER_SIZE  (128 * MiB)
-#else
-# define MAX_CODE_GEN_BUFFER_SIZE  ((size_t)-1)
-#endif
-
 #if TCG_TARGET_REG_BITS == 32
 #define DEFAULT_CODE_GEN_BUFFER_SIZE_1 (32 * MiB)
 #ifdef CONFIG_USER_ONLY
-- 
2.25.1

A size is easier to work with than an end point,
particularly during initial buffer allocation.

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ struct tcg_region_state {
     /* fields set at init time */
     void *start;
     void *start_aligned;
-    void *end;
     size_t n;
     size_t size; /* size of one region */
     size_t stride; /* .size + guard size */
+    size_t total_size; /* size of entire buffer, >= n * stride */
 
     /* fields protected by the lock */
     size_t current; /* current region index */
@@ -XXX,XX +XXX,XX @@ static void tcg_region_bounds(size_t curr_region, void **pstart, void **pend)
     if (curr_region == 0) {
         start = region.start;
     }
+    /* The final region may have a few extra pages due to earlier rounding. */
     if (curr_region == region.n - 1) {
-        end = region.end;
+        end = region.start_aligned + region.total_size;
     }
 
     *pstart = start;
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
  */
 void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
 {
-    void *buf, *aligned;
-    size_t size;
+    void *buf, *aligned, *end;
+    size_t total_size;
     size_t page_size;
     size_t region_size;
     size_t n_regions;
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
     assert(ok);
 
     buf = tcg_init_ctx.code_gen_buffer;
-    size = tcg_init_ctx.code_gen_buffer_size;
+    total_size = tcg_init_ctx.code_gen_buffer_size;
     page_size = qemu_real_host_page_size;
     n_regions = tcg_n_regions(max_cpus);
 
     /* The first region will be 'aligned - buf' bytes larger than the others */
     aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
-    g_assert(aligned < tcg_init_ctx.code_gen_buffer + size);
+    g_assert(aligned < tcg_init_ctx.code_gen_buffer + total_size);
+
     /*
      * Make region_size a multiple of page_size, using aligned as the start.
      * As a result of this we might end up with a few extra pages at the end of
      * the buffer; we will assign those to the last region.
      */
-    region_size = (size - (aligned - buf)) / n_regions;
+    region_size = (total_size - (aligned - buf)) / n_regions;
     region_size = QEMU_ALIGN_DOWN(region_size, page_size);
 
     /* A region must have at least 2 pages; one code, one guard */
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
     region.start = buf;
     region.start_aligned = aligned;
     /* page-align the end, since its last page will be a guard page */
-    region.end = QEMU_ALIGN_PTR_DOWN(buf + size, page_size);
+    end = QEMU_ALIGN_PTR_DOWN(buf + total_size, page_size);
     /* account for that last guard page */
-    region.end -= page_size;
+    end -= page_size;
+    total_size = end - aligned;
+    region.total_size = total_size;
 
     /*
      * Set guard pages in the rw buffer, as that's the one into which
@@ -XXX,XX +XXX,XX @@ void tcg_region_prologue_set(TCGContext *s)
 
     /* Register the balance of the buffer with gdb. */
     tcg_register_jit(tcg_splitwx_to_rx(region.start),
-                     region.end - region.start);
+                     region.start_aligned + region.total_size - region.start);
 }
 
 /*
@@ -XXX,XX +XXX,XX @@ size_t tcg_code_capacity(void)
 
     /* no need for synchronization; these variables are set at init time */
     guard_size = region.stride - region.size;
-    capacity = region.end + guard_size - region.start;
-    capacity -= region.n * (guard_size + TCG_HIGHWATER);
+    capacity = region.total_size;
+    capacity -= (region.n - 1) * guard_size;
+    capacity -= region.n * TCG_HIGHWATER;
+
     return capacity;
 }
 
-- 
2.25.1

Give the field a name reflecting its actual meaning.

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ struct tcg_region_state {
     QemuMutex lock;
 
     /* fields set at init time */
-    void *start;
     void *start_aligned;
+    void *after_prologue;
     size_t n;
     size_t size; /* size of one region */
     size_t stride; /* .size + guard size */
@@ -XXX,XX +XXX,XX @@ static void tcg_region_bounds(size_t curr_region, void **pstart, void **pend)
     end = start + region.size;
 
     if (curr_region == 0) {
-        start = region.start;
+        start = region.after_prologue;
     }
     /* The final region may have a few extra pages due to earlier rounding. */
     if (curr_region == region.n - 1) {
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
     region.n = n_regions;
     region.size = region_size - page_size;
     region.stride = region_size;
-    region.start = buf;
+    region.after_prologue = buf;
     region.start_aligned = aligned;
     /* page-align the end, since its last page will be a guard page */
     end = QEMU_ALIGN_PTR_DOWN(buf + total_size, page_size);
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
 void tcg_region_prologue_set(TCGContext *s)
 {
     /* Deduct the prologue from the first region.  */
-    g_assert(region.start == s->code_gen_buffer);
-    region.start = s->code_ptr;
+    g_assert(region.start_aligned == s->code_gen_buffer);
+    region.after_prologue = s->code_ptr;
 
     /* Recompute boundaries of the first region. */
     tcg_region_assign(s, 0);
 
     /* Register the balance of the buffer with gdb. */
-    tcg_register_jit(tcg_splitwx_to_rx(region.start),
-                     region.start_aligned + region.total_size - region.start);
+    tcg_register_jit(tcg_splitwx_to_rx(region.after_prologue),
+                     region.start_aligned + region.total_size -
+                     region.after_prologue);
 }
 
 /*
-- 
2.25.1

Compute the value using straight division and bounds,
rather than a loop.  Pass in tb_size rather than reading
from tcg_init_ctx.code_gen_buffer_size,

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ void tcg_region_reset_all(void)
     tcg_region_tree_reset_all();
 }
 
-static size_t tcg_n_regions(unsigned max_cpus)
+static size_t tcg_n_regions(size_t tb_size, unsigned max_cpus)
 {
 #ifdef CONFIG_USER_ONLY
     return 1;
 #else
+    size_t n_regions;
+
     /*
      * It is likely that some vCPUs will translate more code than others,
      * so we first try to set more regions than max_cpus, with those regions
      * being of reasonable size. If that's not possible we make do by evenly
      * dividing the code_gen_buffer among the vCPUs.
      */
-    size_t i;
-
     /* Use a single region if all we have is one vCPU thread */
     if (max_cpus == 1 || !qemu_tcg_mttcg_enabled()) {
         return 1;
     }
 
-    /* Try to have more regions than max_cpus, with each region being >= 2 MB */
-    for (i = 8; i > 0; i--) {
-        size_t regions_per_thread = i;
-        size_t region_size;
-
-        region_size = tcg_init_ctx.code_gen_buffer_size;
-        region_size /= max_cpus * regions_per_thread;
-
-        if (region_size >= 2 * 1024u * 1024) {
-            return max_cpus * regions_per_thread;
-        }
+    /*
+     * Try to have more regions than max_cpus, with each region being >= 2 MB.
+     * If we can't, then just allocate one region per vCPU thread.
+     */
+    n_regions = tb_size / (2 * MiB);
+    if (n_regions <= max_cpus) {
+        return max_cpus;
     }
-    /* If we can't, then just allocate one region per vCPU thread */
-    return max_cpus;
+    return MIN(n_regions, max_cpus * 8);
 #endif
 }
 
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
     buf = tcg_init_ctx.code_gen_buffer;
     total_size = tcg_init_ctx.code_gen_buffer_size;
     page_size = qemu_real_host_page_size;
-    n_regions = tcg_n_regions(max_cpus);
+    n_regions = tcg_n_regions(total_size, max_cpus);
 
     /* The first region will be 'aligned - buf' bytes larger than the others */
     aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
-- 
2.25.1

Return output buffer and size via output pointer arguments,
rather than returning size via tcg_ctx->code_gen_buffer_size.

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static inline bool cross_256mb(void *addr, size_t size)
 /*
  * We weren't able to allocate a buffer without crossing that boundary,
  * so make do with the larger portion of the buffer that doesn't cross.
- * Returns the new base of the buffer, and adjusts code_gen_buffer_size.
+ * Returns the new base and size of the buffer in *obuf and *osize.
  */
-static inline void *split_cross_256mb(void *buf1, size_t size1)
+static inline void split_cross_256mb(void **obuf, size_t *osize,
+                                     void *buf1, size_t size1)
 {
     void *buf2 = (void *)(((uintptr_t)buf1 + size1) & ~0x0ffffffful);
     size_t size2 = buf1 + size1 - buf2;
@@ -XXX,XX +XXX,XX @@ static inline void *split_cross_256mb(void *buf1, size_t size1)
         buf1 = buf2;
     }
 
-    tcg_ctx->code_gen_buffer_size = size1;
-    return buf1;
+    *obuf = buf1;
+    *osize = size1;
 }
 #endif
 
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
     if (size > tb_size) {
         size = QEMU_ALIGN_DOWN(tb_size, qemu_real_host_page_size);
     }
-    tcg_ctx->code_gen_buffer_size = size;
 
 #ifdef __mips__
     if (cross_256mb(buf, size)) {
-        buf = split_cross_256mb(buf, size);
-        size = tcg_ctx->code_gen_buffer_size;
+        split_cross_256mb(&buf, &size, buf, size);
     }
 #endif
 
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
     qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 
     tcg_ctx->code_gen_buffer = buf;
+    tcg_ctx->code_gen_buffer_size = size;
     return true;
 }
 #elif defined(_WIN32)
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
                          "allocate %zu bytes for jit buffer", size);
         return false;
     }
-    tcg_ctx->code_gen_buffer_size = size;
 
 #ifdef __mips__
     if (cross_256mb(buf, size)) {
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
             /* fallthru */
         default:
             /* Split the original buffer.  Free the smaller half.  */
-            buf2 = split_cross_256mb(buf, size);
-            size2 = tcg_ctx->code_gen_buffer_size;
+            split_cross_256mb(&buf2, &size2, buf, size);
             if (buf == buf2) {
                 munmap(buf + size2, size - size2);
             } else {
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
     qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 
     tcg_ctx->code_gen_buffer = buf;
+    tcg_ctx->code_gen_buffer_size = size;
     return true;
 }
 
-- 
2.25.1

Shortly, the full code_gen_buffer will only be visible
to region.c, so move in_code_gen_buffer out-of-line.

Move the debugging versions of tcg_splitwx_to_{rx,rw}
to region.c as well, so that the compiler gets to see
the implementation of in_code_gen_buffer.

This leaves exactly one use of in_code_gen_buffer outside
of region.c, in cpu_restore_state.  Which, being on the
exception path, is not performance critical.

Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/tcg/tcg.h | 11 +----------
 tcg/region.c      | 34 ++++++++++++++++++++++++++++++++++
 tcg/tcg.c         | 23 -----------------------
 3 files changed, 35 insertions(+), 33 deletions(-)

diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
index XXXXXXX..XXXXXXX 100644
--- a/include/tcg/tcg.h
+++ b/include/tcg/tcg.h
@@ -XXX,XX +XXX,XX @@ extern const void *tcg_code_gen_epilogue;
 extern uintptr_t tcg_splitwx_diff;
 extern TCGv_env cpu_env;
 
-static inline bool in_code_gen_buffer(const void *p)
-{
-    const TCGContext *s = &tcg_init_ctx;
-    /*
-     * Much like it is valid to have a pointer to the byte past the
-     * end of an array (so long as you don't dereference it), allow
-     * a pointer to the byte past the end of the code gen buffer.
-     */
-    return (size_t)(p - s->code_gen_buffer) <= s->code_gen_buffer_size;
-}
+bool in_code_gen_buffer(const void *p);
 
 #ifdef CONFIG_DEBUG_TCG
 const void *tcg_splitwx_to_rx(void *rw);
diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static struct tcg_region_state region;
 static void *region_trees;
 static size_t tree_size;
 
+bool in_code_gen_buffer(const void *p)
+{
+    const TCGContext *s = &tcg_init_ctx;
+    /*
+     * Much like it is valid to have a pointer to the byte past the
+     * end of an array (so long as you don't dereference it), allow
+     * a pointer to the byte past the end of the code gen buffer.
+     */
+    return (size_t)(p - s->code_gen_buffer) <= s->code_gen_buffer_size;
+}
+
+#ifdef CONFIG_DEBUG_TCG
+const void *tcg_splitwx_to_rx(void *rw)
+{
+    /* Pass NULL pointers unchanged. */
+    if (rw) {
+        g_assert(in_code_gen_buffer(rw));
+        rw += tcg_splitwx_diff;
+    }
+    return rw;
+}
+
+void *tcg_splitwx_to_rw(const void *rx)
+{
+    /* Pass NULL pointers unchanged. */
+    if (rx) {
+        rx -= tcg_splitwx_diff;
+        /* Assert that we end with a pointer in the rw region. */
+        g_assert(in_code_gen_buffer(rx));
+    }
+    return (void *)rx;
+}
+#endif /* CONFIG_DEBUG_TCG */
+
 /* compare a pointer @ptr and a tb_tc @s */
 static int ptr_cmp_tb_tc(const void *ptr, const struct tb_tc *s)
 {
diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ static const TCGTargetOpDef constraint_sets[] = {
 
 #include "tcg-target.c.inc"
 
-#ifdef CONFIG_DEBUG_TCG
-const void *tcg_splitwx_to_rx(void *rw)
-{
-    /* Pass NULL pointers unchanged. */
-    if (rw) {
-        g_assert(in_code_gen_buffer(rw));
-        rw += tcg_splitwx_diff;
-    }
-    return rw;
-}
-
-void *tcg_splitwx_to_rw(const void *rx)
-{
-    /* Pass NULL pointers unchanged. */
-    if (rx) {
-        rx -= tcg_splitwx_diff;
-        /* Assert that we end with a pointer in the rw region. */
-        g_assert(in_code_gen_buffer(rx));
-    }
-    return (void *)rx;
-}
-#endif /* CONFIG_DEBUG_TCG */
-
 static void alloc_tcg_plugin_context(TCGContext *s)
 {
 #ifdef CONFIG_PLUGIN
-- 
2.25.1

Do not mess around with setting values within tcg_init_ctx.
Put the values into 'region' directly, which is where they
will live for the lifetime of the program.

Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/region.c | 64 ++++++++++++++++++++++------------------------------
 1 file changed, 27 insertions(+), 37 deletions(-)

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static size_t tree_size;
 
 bool in_code_gen_buffer(const void *p)
 {
-    const TCGContext *s = &tcg_init_ctx;
     /*
      * Much like it is valid to have a pointer to the byte past the
      * end of an array (so long as you don't dereference it), allow
      * a pointer to the byte past the end of the code gen buffer.
      */
-    return (size_t)(p - s->code_gen_buffer) <= s->code_gen_buffer_size;
+    return (size_t)(p - region.start_aligned) <= region.total_size;
 }
 
 #ifdef CONFIG_DEBUG_TCG
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
     }
     qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 
-    tcg_ctx->code_gen_buffer = buf;
-    tcg_ctx->code_gen_buffer_size = size;
+    region.start_aligned = buf;
+    region.total_size = size;
     return true;
 }
 #elif defined(_WIN32)
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
         return false;
     }
 
-    tcg_ctx->code_gen_buffer = buf;
-    tcg_ctx->code_gen_buffer_size = size;
+    region.start_aligned = buf;
+    region.total_size = size;
     return true;
 }
 #else
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
     /* Request large pages for the buffer.  */
     qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 
-    tcg_ctx->code_gen_buffer = buf;
-    tcg_ctx->code_gen_buffer_size = size;
+    region.start_aligned = buf;
+    region.total_size = size;
     return true;
 }
 
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
         return false;
     }
     /* The size of the mapping may have been adjusted. */
-    size = tcg_ctx->code_gen_buffer_size;
-    buf_rx = tcg_ctx->code_gen_buffer;
+    buf_rx = region.start_aligned;
+    size = region.total_size;
 #endif
 
     buf_rw = qemu_memfd_alloc("tcg-jit", size, 0, &fd, errp);
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
 #endif
 
     close(fd);
-    tcg_ctx->code_gen_buffer = buf_rw;
-    tcg_ctx->code_gen_buffer_size = size;
+    region.start_aligned = buf_rw;
+    region.total_size = size;
     tcg_splitwx_diff = buf_rx - buf_rw;
 
     /* Request large pages for the buffer and the splitwx.  */
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
         return false;
     }
 
-    buf_rw = (mach_vm_address_t)tcg_ctx->code_gen_buffer;
+    buf_rw = region.start_aligned;
     buf_rx = 0;
     ret = mach_vm_remap(mach_task_self(),
                         &buf_rx,
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
  */
 void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
 {
-    void *buf, *aligned, *end;
-    size_t total_size;
     size_t page_size;
     size_t region_size;
-    size_t n_regions;
     size_t i;
     bool ok;
 
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
                                splitwx, &error_fatal);
     assert(ok);
 
-    buf = tcg_init_ctx.code_gen_buffer;
-    total_size = tcg_init_ctx.code_gen_buffer_size;
-    page_size = qemu_real_host_page_size;
-    n_regions = tcg_n_regions(total_size, max_cpus);
-
-    /* The first region will be 'aligned - buf' bytes larger than the others */
-    aligned = QEMU_ALIGN_PTR_UP(buf, page_size);
-    g_assert(aligned < tcg_init_ctx.code_gen_buffer + total_size);
-
     /*
      * Make region_size a multiple of page_size, using aligned as the start.
      * As a result of this we might end up with a few extra pages at the end of
      * the buffer; we will assign those to the last region.
      */
-    region_size = (total_size - (aligned - buf)) / n_regions;
+    region.n = tcg_n_regions(region.total_size, max_cpus);
+    page_size = qemu_real_host_page_size;
+    region_size = region.total_size / region.n;
     region_size = QEMU_ALIGN_DOWN(region_size, page_size);
 
     /* A region must have at least 2 pages; one code, one guard */
     g_assert(region_size >= 2 * page_size);
+    region.stride = region_size;
+
+    /* Reserve space for guard pages. */
+    region.size = region_size - page_size;
+    region.total_size -= page_size;
+
+    /*
+     * The first region will be smaller than the others, via the prologue,
+     * which has yet to be allocated.  For now, the first region begins at
+     * the page boundary.
+     */
+    region.after_prologue = region.start_aligned;
 
     /* init the region struct */
     qemu_mutex_init(&region.lock);
-    region.n = n_regions;
-    region.size = region_size - page_size;
-    region.stride = region_size;
-    region.after_prologue = buf;
-    region.start_aligned = aligned;
-    /* page-align the end, since its last page will be a guard page */
-    end = QEMU_ALIGN_PTR_DOWN(buf + total_size, page_size);
-    /* account for that last guard page */
-    end -= page_size;
-    total_size = end - aligned;
-    region.total_size = total_size;
 
     /*
      * Set guard pages in the rw buffer, as that's the one into which
-- 
2.25.1

Change the interface from a boolean error indication to a
negative error vs a non-negative protection.  For the moment
this is only interface change, not making use of the new data.

Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/region.c | 63 +++++++++++++++++++++++++++-------------------------
 1 file changed, 33 insertions(+), 30 deletions(-)

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static inline void split_cross_256mb(void **obuf, size_t *osize,
 static uint8_t static_code_gen_buffer[DEFAULT_CODE_GEN_BUFFER_SIZE]
     __attribute__((aligned(CODE_GEN_ALIGN)));
 
-static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
+static int alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
 {
     void *buf, *end;
     size_t size;
 
     if (splitwx > 0) {
         error_setg(errp, "jit split-wx not supported");
-        return false;
+        return -1;
     }
 
     /* page-align the beginning and end of the buffer */
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
 
     region.start_aligned = buf;
     region.total_size = size;
-    return true;
+
+    return PROT_READ | PROT_WRITE;
 }
 #elif defined(_WIN32)
-static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
+static int alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
 {
     void *buf;
 
     if (splitwx > 0) {
         error_setg(errp, "jit split-wx not supported");
-        return false;
+        return -1;
     }
 
     buf = VirtualAlloc(NULL, size, MEM_RESERVE | MEM_COMMIT,
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
 
     region.start_aligned = buf;
     region.total_size = size;
-    return true;
+
+    return PAGE_READ | PAGE_WRITE | PAGE_EXEC;
 }
 #else
-static bool alloc_code_gen_buffer_anon(size_t size, int prot,
-                                       int flags, Error **errp)
+static int alloc_code_gen_buffer_anon(size_t size, int prot,
+                                      int flags, Error **errp)
 {
     void *buf;
 
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
     if (buf == MAP_FAILED) {
         error_setg_errno(errp, errno,
                          "allocate %zu bytes for jit buffer", size);
-        return false;
+        return -1;
     }
 
 #ifdef __mips__
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_anon(size_t size, int prot,
 
     region.start_aligned = buf;
     region.total_size = size;
-    return true;
+    return prot;
 }
 
 #ifndef CONFIG_TCG_INTERPRETER
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
 
 #ifdef __mips__
     /* Find space for the RX mapping, vs the 256MiB regions. */
-    if (!alloc_code_gen_buffer_anon(size, PROT_NONE,
-                                    MAP_PRIVATE | MAP_ANONYMOUS |
-                                    MAP_NORESERVE, errp)) {
+    if (alloc_code_gen_buffer_anon(size, PROT_NONE,
+                                   MAP_PRIVATE | MAP_ANONYMOUS |
+                                   MAP_NORESERVE, errp) < 0) {
         return false;
     }
     /* The size of the mapping may have been adjusted. */
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
     /* Request large pages for the buffer and the splitwx.  */
     qemu_madvise(buf_rw, size, QEMU_MADV_HUGEPAGE);
     qemu_madvise(buf_rx, size, QEMU_MADV_HUGEPAGE);
-    return true;
+    return PROT_READ | PROT_WRITE;
 
  fail_rx:
     error_setg_errno(errp, errno, "failed to map shared memory for execute");
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
     if (fd >= 0) {
         close(fd);
     }
-    return false;
+    return -1;
 }
 #endif /* CONFIG_POSIX */
 
@@ -XXX,XX +XXX,XX @@ extern kern_return_t mach_vm_remap(vm_map_t target_task,
                                    vm_prot_t *max_protection,
                                    vm_inherit_t inheritance);
 
-static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
+static int alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
 {
     kern_return_t ret;
     mach_vm_address_t buf_rw, buf_rx;
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
     /* Map the read-write portion via normal anon memory. */
     if (!alloc_code_gen_buffer_anon(size, PROT_READ | PROT_WRITE,
                                     MAP_PRIVATE | MAP_ANONYMOUS, errp)) {
-        return false;
+        return -1;
     }
 
     buf_rw = region.start_aligned;
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_vmremap(size_t size, Error **errp)
         /* TODO: Convert "ret" to a human readable error message. */
         error_setg(errp, "vm_remap for jit splitwx failed");
         munmap((void *)buf_rw, size);
-        return false;
+        return -1;
     }
 
     if (mprotect((void *)buf_rx, size, PROT_READ | PROT_EXEC) != 0) {
         error_setg_errno(errp, errno, "mprotect for jit splitwx");
         munmap((void *)buf_rx, size);
         munmap((void *)buf_rw, size);
-        return false;
+        return -1;
     }
 
     tcg_splitwx_diff = buf_rx - buf_rw;
-    return true;
+    return PROT_READ | PROT_WRITE;
 }
 #endif /* CONFIG_DARWIN */
 #endif /* CONFIG_TCG_INTERPRETER */
 
-static bool alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
+static int alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
 {
 #ifndef CONFIG_TCG_INTERPRETER
 # ifdef CONFIG_DARWIN
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx(size_t size, Error **errp)
 # endif
 #endif
     error_setg(errp, "jit split-wx not supported");
-    return false;
+    return -1;
 }
 
-static bool alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
+static int alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
 {
     ERRP_GUARD();
     int prot, flags;
 
     if (splitwx) {
-        if (alloc_code_gen_buffer_splitwx(size, errp)) {
-            return true;
+        prot = alloc_code_gen_buffer_splitwx(size, errp);
+        if (prot >= 0) {
+            return prot;
         }
         /*
          * If splitwx force-on (1), fail;
          * if splitwx default-on (-1), fall through to splitwx off.
          */
         if (splitwx > 0) {
-            return false;
+            return -1;
         }
         error_free_or_abort(errp);
     }
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
     size_t page_size;
     size_t region_size;
     size_t i;
-    bool ok;
+    int have_prot;
 
-    ok = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
-                               splitwx, &error_fatal);
-    assert(ok);
+    have_prot = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
+                                      splitwx, &error_fatal);
+    assert(have_prot >= 0);
 
     /*
      * Make region_size a multiple of page_size, using aligned as the start.
-- 
2.25.1

Move the call out of the N versions of alloc_code_gen_buffer
and into tcg_region_init.

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
         error_setg_errno(errp, errno, "mprotect of jit buffer");
         return false;
     }
-    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
 
     region.start_aligned = buf;
     region.total_size = size;
@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer_anon(size_t size, int prot,
     }
 #endif
 
-    /* Request large pages for the buffer.  */
-    qemu_madvise(buf, size, QEMU_MADV_HUGEPAGE);
-
     region.start_aligned = buf;
     region.total_size = size;
     return prot;
@@ -XXX,XX +XXX,XX @@ static bool alloc_code_gen_buffer_splitwx_memfd(size_t size, Error **errp)
     region.total_size = size;
     tcg_splitwx_diff = buf_rx - buf_rw;
 
-    /* Request large pages for the buffer and the splitwx.  */
-    qemu_madvise(buf_rw, size, QEMU_MADV_HUGEPAGE);
-    qemu_madvise(buf_rx, size, QEMU_MADV_HUGEPAGE);
     return PROT_READ | PROT_WRITE;
 
  fail_rx:
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
                                       splitwx, &error_fatal);
     assert(have_prot >= 0);
 
+    /* Request large pages for the buffer and the splitwx.  */
+    qemu_madvise(region.start_aligned, region.total_size, QEMU_MADV_HUGEPAGE);
+    if (tcg_splitwx_diff) {
+        qemu_madvise(region.start_aligned + tcg_splitwx_diff,
+                     region.total_size, QEMU_MADV_HUGEPAGE);
+    }
+
     /*
      * Make region_size a multiple of page_size, using aligned as the start.
      * As a result of this we might end up with a few extra pages at the end of
-- 
2.25.1

For --enable-tcg-interpreter on Windows, we will need this.

Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/qemu/osdep.h | 1 +
 util/osdep.c         | 9 +++++++++
 2 files changed, 10 insertions(+)

diff --git a/include/qemu/osdep.h b/include/qemu/osdep.h
index XXXXXXX..XXXXXXX 100644
--- a/include/qemu/osdep.h
+++ b/include/qemu/osdep.h
@@ -XXX,XX +XXX,XX @@ void sigaction_invoke(struct sigaction *action,
 #endif
 
 int qemu_madvise(void *addr, size_t len, int advice);
+int qemu_mprotect_rw(void *addr, size_t size);
 int qemu_mprotect_rwx(void *addr, size_t size);
 int qemu_mprotect_none(void *addr, size_t size);
 
diff --git a/util/osdep.c b/util/osdep.c
index XXXXXXX..XXXXXXX 100644
--- a/util/osdep.c
+++ b/util/osdep.c
@@ -XXX,XX +XXX,XX @@ static int qemu_mprotect__osdep(void *addr, size_t size, int prot)
 #endif
 }
 
+int qemu_mprotect_rw(void *addr, size_t size)
+{
+#ifdef _WIN32
+    return qemu_mprotect__osdep(addr, size, PAGE_READWRITE);
+#else
+    return qemu_mprotect__osdep(addr, size, PROT_READ | PROT_WRITE);
+#endif
+}
+
 int qemu_mprotect_rwx(void *addr, size_t size)
 {
 #ifdef _WIN32
-- 
2.25.1

If qemu_get_host_physmem returns an odd number of pages,
then physmem / 8 will not be a multiple of the page size.

The following was observed on a gitlab runner:

ERROR qtest-arm/boot-serial-test - Bail out!
ERROR:../util/osdep.c:80:qemu_mprotect__osdep: \
  assertion failed: (!(size & ~qemu_real_host_page_mask))

Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/region.c | 47 +++++++++++++++++++++--------------------------
 1 file changed, 21 insertions(+), 26 deletions(-)

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static size_t tcg_n_regions(size_t tb_size, unsigned max_cpus)
   (DEFAULT_CODE_GEN_BUFFER_SIZE_1 < MAX_CODE_GEN_BUFFER_SIZE \
    ? DEFAULT_CODE_GEN_BUFFER_SIZE_1 : MAX_CODE_GEN_BUFFER_SIZE)
 
-static size_t size_code_gen_buffer(size_t tb_size)
-{
-    /* Size the buffer.  */
-    if (tb_size == 0) {
-        size_t phys_mem = qemu_get_host_physmem();
-        if (phys_mem == 0) {
-            tb_size = DEFAULT_CODE_GEN_BUFFER_SIZE;
-        } else {
-            tb_size = MIN(DEFAULT_CODE_GEN_BUFFER_SIZE, phys_mem / 8);
-        }
-    }
-    if (tb_size < MIN_CODE_GEN_BUFFER_SIZE) {
-        tb_size = MIN_CODE_GEN_BUFFER_SIZE;
-    }
-    if (tb_size > MAX_CODE_GEN_BUFFER_SIZE) {
-        tb_size = MAX_CODE_GEN_BUFFER_SIZE;
-    }
-    return tb_size;
-}
-
 #ifdef __mips__
 /*
  * In order to use J and JAL within the code_gen_buffer, we require
@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
  */
 void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
 {
-    size_t page_size;
+    const size_t page_size = qemu_real_host_page_size;
     size_t region_size;
     size_t i;
     int have_prot;
 
-    have_prot = alloc_code_gen_buffer(size_code_gen_buffer(tb_size),
-                                      splitwx, &error_fatal);
+    /* Size the buffer.  */
+    if (tb_size == 0) {
+        size_t phys_mem = qemu_get_host_physmem();
+        if (phys_mem == 0) {
+            tb_size = DEFAULT_CODE_GEN_BUFFER_SIZE;
+        } else {
+            tb_size = QEMU_ALIGN_DOWN(phys_mem / 8, page_size);
+            tb_size = MIN(DEFAULT_CODE_GEN_BUFFER_SIZE, tb_size);
+        }
+    }
+    if (tb_size < MIN_CODE_GEN_BUFFER_SIZE) {
+        tb_size = MIN_CODE_GEN_BUFFER_SIZE;
+    }
+    if (tb_size > MAX_CODE_GEN_BUFFER_SIZE) {
+        tb_size = MAX_CODE_GEN_BUFFER_SIZE;
+    }
+
+    have_prot = alloc_code_gen_buffer(tb_size, splitwx, &error_fatal);
     assert(have_prot >= 0);
 
     /* Request large pages for the buffer and the splitwx.  */
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
      * As a result of this we might end up with a few extra pages at the end of
      * the buffer; we will assign those to the last region.
      */
-    region.n = tcg_n_regions(region.total_size, max_cpus);
-    page_size = qemu_real_host_page_size;
-    region_size = region.total_size / region.n;
+    region.n = tcg_n_regions(tb_size, max_cpus);
+    region_size = tb_size / region.n;
     region_size = QEMU_ALIGN_DOWN(region_size, page_size);
 
     /* A region must have at least 2 pages; one code, one guard */
-- 
2.25.1

Do not handle protections on a case-by-case basis in the
various alloc_code_gen_buffer instances; do it within a
single loop in tcg_region_init.

Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/region.c | 45 +++++++++++++++++++++++++++++++--------------
 1 file changed, 31 insertions(+), 14 deletions(-)

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer(size_t tb_size, int splitwx, Error **errp)
     }
 #endif
 
-    if (qemu_mprotect_rwx(buf, size)) {
-        error_setg_errno(errp, errno, "mprotect of jit buffer");
-        return false;
-    }
-
     region.start_aligned = buf;
     region.total_size = size;
 
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
 {
     const size_t page_size = qemu_real_host_page_size;
     size_t region_size;
-    size_t i;
-    int have_prot;
+    int have_prot, need_prot;
 
     /* Size the buffer.  */
     if (tb_size == 0) {
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
      * Set guard pages in the rw buffer, as that's the one into which
      * buffer overruns could occur.  Do not set guard pages in the rx
      * buffer -- let that one use hugepages throughout.
+     * Work with the page protections set up with the initial mapping.
      */
-    for (i = 0; i < region.n; i++) {
+    need_prot = PAGE_READ | PAGE_WRITE;
+#ifndef CONFIG_TCG_INTERPRETER
+    if (tcg_splitwx_diff == 0) {
+        need_prot |= PAGE_EXEC;
+    }
+#endif
+    for (size_t i = 0, n = region.n; i < n; i++) {
         void *start, *end;
 
         tcg_region_bounds(i, &start, &end);
+        if (have_prot != need_prot) {
+            int rc;
 
-        /*
-         * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
-         * rejects a permission change from RWX -> NONE.  Guard pages are
-         * nice for bug detection but are not essential; ignore any failure.
-         */
-        (void)qemu_mprotect_none(end, page_size);
+            if (need_prot == (PAGE_READ | PAGE_WRITE | PAGE_EXEC)) {
+                rc = qemu_mprotect_rwx(start, end - start);
+            } else if (need_prot == (PAGE_READ | PAGE_WRITE)) {
+                rc = qemu_mprotect_rw(start, end - start);
+            } else {
+                g_assert_not_reached();
+            }
+            if (rc) {
+                error_setg_errno(&error_fatal, errno,
+                                 "mprotect of jit buffer");
+            }
+        }
+        if (have_prot != 0) {
+            /*
+             * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
+             * rejects a permission change from RWX -> NONE.  Guard pages are
+             * nice for bug detection but are not essential; ignore any failure.
+             */
+            (void)qemu_mprotect_none(end, page_size);
+        }
     }
 
     tcg_region_trees_init();
-- 
2.25.1

There's a change in mprotect() behaviour [1] in the latest macOS
on M1 and it's not yet clear if it's going to be fixed by Apple.

In this case, instead of changing permissions of N guard pages,
we change permissions of N rwx regions.  The same number of
syscalls are required either way.

[1] https://gist.github.com/hikalium/75ae822466ee4da13cbbe486498a191f

Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/region.c | 19 +++++++++----------
 1 file changed, 9 insertions(+), 10 deletions(-)

diff --git a/tcg/region.c b/tcg/region.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/region.c
+++ b/tcg/region.c
@@ -XXX,XX +XXX,XX @@ static int alloc_code_gen_buffer(size_t size, int splitwx, Error **errp)
         error_free_or_abort(errp);
     }
 
-    prot = PROT_READ | PROT_WRITE | PROT_EXEC;
+    /*
+     * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
+     * rejects a permission change from RWX -> NONE when reserving the
+     * guard pages later.  We can go the other way with the same number
+     * of syscalls, so always begin with PROT_NONE.
+     */
+    prot = PROT_NONE;
     flags = MAP_PRIVATE | MAP_ANONYMOUS;
-#ifdef CONFIG_TCG_INTERPRETER
-    /* The tcg interpreter does not need execute permission. */
-    prot = PROT_READ | PROT_WRITE;
-#elif defined(CONFIG_DARWIN)
+#ifdef CONFIG_DARWIN
     /* Applicable to both iOS and macOS (Apple Silicon). */
     if (!splitwx) {
         flags |= MAP_JIT;
@@ -XXX,XX +XXX,XX @@ void tcg_region_init(size_t tb_size, int splitwx, unsigned max_cpus)
             }
         }
         if (have_prot != 0) {
-            /*
-             * macOS 11.2 has a bug (Apple Feedback FB8994773) in which mprotect
-             * rejects a permission change from RWX -> NONE.  Guard pages are
-             * nice for bug detection but are not essential; ignore any failure.
-             */
+            /* Guard pages are nice for bug detection but are not essential. */
             (void)qemu_mprotect_none(end, page_size);
         }
     }
-- 
2.25.1

These variables belong to the jit side, not the user side.

Since tcg_init_ctx is no longer used outside of tcg/, move
the declaration to tcg-internal.h.

Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Suggested-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/tcg/tcg.h         | 1 -
 tcg/tcg-internal.h        | 1 +
 accel/tcg/translate-all.c | 3 ---
 tcg/tcg.c                 | 3 +++
 4 files changed, 4 insertions(+), 4 deletions(-)

diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
index XXXXXXX..XXXXXXX 100644
--- a/include/tcg/tcg.h
+++ b/include/tcg/tcg.h
@@ -XXX,XX +XXX,XX @@ static inline bool temp_readonly(TCGTemp *ts)
     return ts->kind >= TEMP_FIXED;
 }
 
-extern TCGContext tcg_init_ctx;
 extern __thread TCGContext *tcg_ctx;
 extern const void *tcg_code_gen_epilogue;
 extern uintptr_t tcg_splitwx_diff;
diff --git a/tcg/tcg-internal.h b/tcg/tcg-internal.h
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg-internal.h
+++ b/tcg/tcg-internal.h
@@ -XXX,XX +XXX,XX @@
 
 #define TCG_HIGHWATER 1024
 
+extern TCGContext tcg_init_ctx;
 extern TCGContext **tcg_ctxs;
 extern unsigned int tcg_cur_ctxs;
 extern unsigned int tcg_max_ctxs;
diff --git a/accel/tcg/translate-all.c b/accel/tcg/translate-all.c
index XXXXXXX..XXXXXXX 100644
--- a/accel/tcg/translate-all.c
+++ b/accel/tcg/translate-all.c
@@ -XXX,XX +XXX,XX @@ static int v_l2_levels;
 
 static void *l1_map[V_L1_MAX_SIZE];
 
-/* code generation context */
-TCGContext tcg_init_ctx;
-__thread TCGContext *tcg_ctx;
 TBContext tb_ctx;
 
 static void page_table_config_init(void)
diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ static bool tcg_target_const_match(int64_t val, TCGType type, int ct);
 static int tcg_out_ldst_finalize(TCGContext *s);
 #endif
 
+TCGContext tcg_init_ctx;
+__thread TCGContext *tcg_ctx;
+
 TCGContext **tcg_ctxs;
 unsigned int tcg_cur_ctxs;
 unsigned int tcg_max_ctxs;
-- 
2.25.1

Introduce a function to remove everything emitted
since a given point.

Reviewed-by: Peter Maydell <peter.maydell@linaro.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/tcg/tcg.h | 10 ++++++++++
 tcg/tcg.c         | 13 +++++++++++++
 2 files changed, 23 insertions(+)

diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
index XXXXXXX..XXXXXXX 100644
--- a/include/tcg/tcg.h
+++ b/include/tcg/tcg.h
@@ -XXX,XX +XXX,XX @@ void tcg_op_remove(TCGContext *s, TCGOp *op);
 TCGOp *tcg_op_insert_before(TCGContext *s, TCGOp *op, TCGOpcode opc);
 TCGOp *tcg_op_insert_after(TCGContext *s, TCGOp *op, TCGOpcode opc);
 
+/**
+ * tcg_remove_ops_after:
+ * @op: target operation
+ *
+ * Discard any opcodes emitted since @op.  Expected usage is to save
+ * a starting point with tcg_last_op(), speculatively emit opcodes,
+ * then decide whether or not to keep those opcodes after the fact.
+ */
+void tcg_remove_ops_after(TCGOp *op);
+
 void tcg_optimize(TCGContext *s);
 
 /* Allocate a new temporary and initialize it with a constant. */
diff --git a/tcg/tcg.c b/tcg/tcg.c
index XXXXXXX..XXXXXXX 100644
--- a/tcg/tcg.c
+++ b/tcg/tcg.c
@@ -XXX,XX +XXX,XX @@ void tcg_op_remove(TCGContext *s, TCGOp *op)
 #endif
 }
 
+void tcg_remove_ops_after(TCGOp *op)
+{
+    TCGContext *s = tcg_ctx;
+
+    while (true) {
+        TCGOp *last = tcg_last_op();
+        if (last == op) {
+            return;
+        }
+        tcg_op_remove(s, last);
+    }
+}
+
 static TCGOp *tcg_op_alloc(TCGOpcode opc)
 {
     TCGContext *s = tcg_ctx;
-- 
2.25.1

At some point during the development of tcg_constant_*, I changed
my mind about whether such temps should be able to be passed to
tcg_temp_free_*.  The final version committed allows this, but the
commentary was not updated to match.

Fixes: c0522136adf
Reported-by: Peter Maydell <peter.maydell@linaro.org>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Reviewed-by: Luis Pires <luis.pires@eldorado.org.br>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/tcg/tcg.h | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

diff --git a/include/tcg/tcg.h b/include/tcg/tcg.h
index XXXXXXX..XXXXXXX 100644
--- a/include/tcg/tcg.h
+++ b/include/tcg/tcg.h
@@ -XXX,XX +XXX,XX @@ TCGv_vec tcg_const_ones_vec_matching(TCGv_vec);
 
 /*
  * Locate or create a read-only temporary that is a constant.
- * This kind of temporary need not and should not be freed.
+ * This kind of temporary need not be freed, but for convenience
+ * will be silently ignored by tcg_temp_free_*.
  */
 TCGTemp *tcg_constant_internal(TCGType type, int64_t val);
 
-- 
2.25.1

From: "Jose R. Ziviani" <jziviani@suse.de>

Commit 5e8892db93 fixed several function signatures but tcg_out_op for
arm is missing. This patch fixes it as well.

Signed-off-by: Jose R. Ziviani <jziviani@suse.de>
Message-Id: <20210610224450.23425-1-jziviani@suse.de>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 tcg/arm/tcg-target.c.inc | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

diff --git a/tcg/arm/tcg-target.c.inc b/tcg/arm/tcg-target.c.inc
index XXXXXXX..XXXXXXX 100644
--- a/tcg/arm/tcg-target.c.inc
+++ b/tcg/arm/tcg-target.c.inc
@@ -XXX,XX +XXX,XX @@ static void tcg_out_qemu_st(TCGContext *s, const TCGArg *args, bool is64)
 static void tcg_out_epilogue(TCGContext *s);
 
 static inline void tcg_out_op(TCGContext *s, TCGOpcode opc,
-                const TCGArg *args, const int *const_args)
+                const TCGArg args[TCG_MAX_OP_ARGS],
+                const int const_args[TCG_MAX_OP_ARGS])
 {
     TCGArg a0, a1, a2, a3, a4, a5;
     int c;
-- 
2.25.1

From: Luis Pires <luis.pires@eldorado.org.br>

Signed-off-by: Luis Pires <luis.pires@eldorado.org.br>
Message-Id: <20210601125143.191165-1-luis.pires@eldorado.org.br>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 docs/devel/tcg.rst | 101 ++++++++++++++++++++++++++++++++++++++++-----
 1 file changed, 90 insertions(+), 11 deletions(-)

diff --git a/docs/devel/tcg.rst b/docs/devel/tcg.rst
index XXXXXXX..XXXXXXX 100644
--- a/docs/devel/tcg.rst
+++ b/docs/devel/tcg.rst
@@ -XXX,XX +XXX,XX @@ performances.
 QEMU's dynamic translation backend is called TCG, for "Tiny Code
 Generator". For more information, please take a look at ``tcg/README``.
 
-Some notable features of QEMU's dynamic translator are:
+The following sections outline some notable features and implementation
+details of QEMU's dynamic translator.
 
 CPU state optimisations
 -----------------------
 
-The target CPUs have many internal states which change the way it
-evaluates instructions. In order to achieve a good speed, the
+The target CPUs have many internal states which change the way they
+evaluate instructions. In order to achieve a good speed, the
 translation phase considers that some state information of the virtual
 CPU cannot change in it. The state is recorded in the Translation
 Block (TB). If the state changes (e.g. privilege level), a new TB will
@@ -XXX,XX +XXX,XX @@ Direct block chaining
 ---------------------
 
 After each translated basic block is executed, QEMU uses the simulated
-Program Counter (PC) and other cpu state information (such as the CS
+Program Counter (PC) and other CPU state information (such as the CS
 segment base value) to find the next basic block.
 
-In order to accelerate the most common cases where the new simulated PC
-is known, QEMU can patch a basic block so that it jumps directly to the
-next one.
+In its simplest, less optimized form, this is done by exiting from the
+current TB, going through the TB epilogue, and then back to the
+main loop. That’s where QEMU looks for the next TB to execute,
+translating it from the guest architecture if it isn’t already available
+in memory. Then QEMU proceeds to execute this next TB, starting at the
+prologue and then moving on to the translated instructions.
 
-The most portable code uses an indirect jump. An indirect jump makes
-it easier to make the jump target modification atomic. On some host
-architectures (such as x86 or PowerPC), the ``JUMP`` opcode is
-directly patched so that the block chaining has no overhead.
+Exiting from the TB this way will cause the ``cpu_exec_interrupt()``
+callback to be re-evaluated before executing additional instructions.
+It is mandatory to exit this way after any CPU state changes that may
+unmask interrupts.
+
+In order to accelerate the cases where the TB for the new
+simulated PC is already available, QEMU has mechanisms that allow
+multiple TBs to be chained directly, without having to go back to the
+main loop as described above. These mechanisms are:
+
+``lookup_and_goto_ptr``
+^^^^^^^^^^^^^^^^^^^^^^^
+
+Calling ``tcg_gen_lookup_and_goto_ptr()`` will emit a call to
+``helper_lookup_tb_ptr``. This helper will look for an existing TB that
+matches the current CPU state. If the destination TB is available its
+code address is returned, otherwise the address of the JIT epilogue is
+returned. The call to the helper is always followed by the tcg ``goto_ptr``
+opcode, which branches to the returned address. In this way, we either
+branch to the next TB or return to the main loop.
+
+``goto_tb + exit_tb``
+^^^^^^^^^^^^^^^^^^^^^
+
+The translation code usually implements branching by performing the
+following steps:
+
+1. Call ``tcg_gen_goto_tb()`` passing a jump slot index (either 0 or 1)
+   as a parameter.
+
+2. Emit TCG instructions to update the CPU state with any information
+   that has been assumed constant and is required by the main loop to
+   correctly locate and execute the next TB. For most guests, this is
+   just the PC of the branch destination, but others may store additional
+   data. The information updated in this step must be inferable from both
+   ``cpu_get_tb_cpu_state()`` and ``cpu_restore_state()``.
+
+3. Call ``tcg_gen_exit_tb()`` passing the address of the current TB and
+   the jump slot index again.
+
+Step 1, ``tcg_gen_goto_tb()``, will emit a ``goto_tb`` TCG
+instruction that later on gets translated to a jump to an address
+associated with the specified jump slot. Initially, this is the address
+of step 2's instructions, which update the CPU state information. Step 3,
+``tcg_gen_exit_tb()``, exits from the current TB returning a tagged
+pointer composed of the last executed TB’s address and the jump slot
+index.
+
+The first time this whole sequence is executed, step 1 simply jumps
+to step 2. Then the CPU state information gets updated and we exit from
+the current TB. As a result, the behavior is very similar to the less
+optimized form described earlier in this section.
+
+Next, the main loop looks for the next TB to execute using the
+current CPU state information (creating the TB if it wasn’t already
+available) and, before starting to execute the new TB’s instructions,
+patches the previously executed TB by associating one of its jump
+slots (the one specified in the call to ``tcg_gen_exit_tb()``) with the
+address of the new TB.
+
+The next time this previous TB is executed and we get to that same
+``goto_tb`` step, it will already be patched (assuming the destination TB
+is still in memory) and will jump directly to the first instruction of
+the destination TB, without going back to the main loop.
+
+For the ``goto_tb + exit_tb`` mechanism to be used, the following
+conditions need to be satisfied:
+
+* The change in CPU state must be constant, e.g., a direct branch and
+  not an indirect branch.
+
+* The direct branch cannot cross a page boundary. Memory mappings
+  may change, causing the code at the destination address to change.
+
+Note that, on step 3 (``tcg_gen_exit_tb()``), in addition to the
+jump slot index, the address of the TB just executed is also returned.
+This address corresponds to the TB that will be patched; it may be
+different than the one that was directly executed from the main loop
+if the latter had already been chained to other TBs.
 
 Self-modifying code and translated code invalidation
 ----------------------------------------------------
-- 
2.25.1

The following changes since commit 6587b0c1331d427b0939c37e763842550ed581db:

Merge remote-tracking branch 'remotes/ericb/tags/pull-nbd-2021-10-15' into staging (2021-10-15 14:16:28 -0700)

are available in the Git repository at:

https://gitlab.com/rth7680/qemu.git tags/pull-tcg-20211016

for you to fetch changes up to 995b87dedc78b0467f5f18bbc3546072ba97516a:

Revert "cpu: Move cpu_common_props to hw/core/cpu.c" (2021-10-15 16:39:15 -0700)

----------------------------------------------------------------
Move gdb singlestep to generic code
Fix cpu_common_props

----------------------------------------------------------------
Richard Henderson (24):
      accel/tcg: Handle gdb singlestep in cpu_tb_exec
      target/alpha: Drop checks for singlestep_enabled
      target/avr: Drop checks for singlestep_enabled
      target/cris: Drop checks for singlestep_enabled
      target/hexagon: Drop checks for singlestep_enabled
      target/arm: Drop checks for singlestep_enabled
      target/hppa: Drop checks for singlestep_enabled
      target/i386: Check CF_NO_GOTO_TB for dc->jmp_opt
      target/i386: Drop check for singlestep_enabled
      target/m68k: Drop checks for singlestep_enabled
      target/microblaze: Check CF_NO_GOTO_TB for DISAS_JUMP
      target/microblaze: Drop checks for singlestep_enabled
      target/mips: Fix single stepping
      target/mips: Drop exit checks for singlestep_enabled
      target/openrisc: Drop checks for singlestep_enabled
      target/ppc: Drop exit checks for singlestep_enabled
      target/riscv: Remove dead code after exception
      target/riscv: Remove exit_tb and lookup_and_goto_ptr
      target/rx: Drop checks for singlestep_enabled
      target/s390x: Drop check for singlestep_enabled
      target/sh4: Drop check for singlestep_enabled
      target/tricore: Drop check for singlestep_enabled
      target/xtensa: Drop check for singlestep_enabled
      Revert "cpu: Move cpu_common_props to hw/core/cpu.c"

GDB single-stepping is now handled generically.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/alpha/translate.c | 13 +++----------
 1 file changed, 3 insertions(+), 10 deletions(-)

diff --git a/target/alpha/translate.c b/target/alpha/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/alpha/translate.c
+++ b/target/alpha/translate.c
@@ -XXX,XX +XXX,XX @@ static void alpha_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
         tcg_gen_movi_i64(cpu_pc, ctx->base.pc_next);
         /* FALLTHRU */
     case DISAS_PC_UPDATED:
-        if (!ctx->base.singlestep_enabled) {
-            tcg_gen_lookup_and_goto_ptr();
-            break;
-        }
-        /* FALLTHRU */
+        tcg_gen_lookup_and_goto_ptr();
+        break;
     case DISAS_PC_UPDATED_NOCHAIN:
-        if (ctx->base.singlestep_enabled) {
-            gen_excp_1(EXCP_DEBUG, 0);
-        } else {
-            tcg_gen_exit_tb(NULL, 0);
-        }
+        tcg_gen_exit_tb(NULL, 0);
         break;
     default:
         g_assert_not_reached();
-- 
2.25.1

GDB single-stepping is now handled generically.

Tested-by: Michael Rolnik <mrolnik@gmail.com>
Reviewed-by: Michael Rolnik <mrolnik@gmail.com>
Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/avr/translate.c | 19 ++++---------------
 1 file changed, 4 insertions(+), 15 deletions(-)

diff --git a/target/avr/translate.c b/target/avr/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/avr/translate.c
+++ b/target/avr/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
         tcg_gen_exit_tb(tb, n);
     } else {
         tcg_gen_movi_i32(cpu_pc, dest);
-        if (ctx->base.singlestep_enabled) {
-            gen_helper_debug(cpu_env);
-        } else {
-            tcg_gen_lookup_and_goto_ptr();
-        }
+        tcg_gen_lookup_and_goto_ptr();
     }
     ctx->base.is_jmp = DISAS_NORETURN;
 }
@@ -XXX,XX +XXX,XX @@ static void avr_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
         tcg_gen_movi_tl(cpu_pc, ctx->npc);
         /* fall through */
     case DISAS_LOOKUP:
-        if (!ctx->base.singlestep_enabled) {
-            tcg_gen_lookup_and_goto_ptr();
-            break;
-        }
-        /* fall through */
+        tcg_gen_lookup_and_goto_ptr();
+        break;
     case DISAS_EXIT:
-        if (ctx->base.singlestep_enabled) {
-            gen_helper_debug(cpu_env);
-        } else {
-            tcg_gen_exit_tb(NULL, 0);
-        }
+        tcg_gen_exit_tb(NULL, 0);
         break;
     default:
         g_assert_not_reached();
-- 
2.25.1

GDB single-stepping is now handled generically.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/hexagon/translate.c | 12 ++----------
 1 file changed, 2 insertions(+), 10 deletions(-)

diff --git a/target/hexagon/translate.c b/target/hexagon/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/hexagon/translate.c
+++ b/target/hexagon/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_end_tb(DisasContext *ctx)
 {
     gen_exec_counters(ctx);
     tcg_gen_mov_tl(hex_gpr[HEX_REG_PC], hex_next_PC);
-    if (ctx->base.singlestep_enabled) {
-        gen_exception_raw(EXCP_DEBUG);
-    } else {
-        tcg_gen_exit_tb(NULL, 0);
-    }
+    tcg_gen_exit_tb(NULL, 0);
     ctx->base.is_jmp = DISAS_NORETURN;
 }
 
@@ -XXX,XX +XXX,XX @@ static void hexagon_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
     case DISAS_TOO_MANY:
         gen_exec_counters(ctx);
         tcg_gen_movi_tl(hex_gpr[HEX_REG_PC], ctx->base.pc_next);
-        if (ctx->base.singlestep_enabled) {
-            gen_exception_raw(EXCP_DEBUG);
-        } else {
-            tcg_gen_exit_tb(NULL, 0);
-        }
+        tcg_gen_exit_tb(NULL, 0);
         break;
     case DISAS_NORETURN:
         break;
-- 
2.25.1

GDB single-stepping is now handled generically.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/arm/translate-a64.c | 10 ++--------
 target/arm/translate.c     | 36 ++++++------------------------------
 2 files changed, 8 insertions(+), 38 deletions(-)

diff --git a/target/arm/translate-a64.c b/target/arm/translate-a64.c
index XXXXXXX..XXXXXXX 100644
--- a/target/arm/translate-a64.c
+++ b/target/arm/translate-a64.c
@@ -XXX,XX +XXX,XX @@ static inline void gen_goto_tb(DisasContext *s, int n, uint64_t dest)
         gen_a64_set_pc_im(dest);
         if (s->ss_active) {
             gen_step_complete_exception(s);
-        } else if (s->base.singlestep_enabled) {
-            gen_exception_internal(EXCP_DEBUG);
         } else {
             tcg_gen_lookup_and_goto_ptr();
             s->base.is_jmp = DISAS_NORETURN;
@@ -XXX,XX +XXX,XX @@ static void aarch64_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
 {
     DisasContext *dc = container_of(dcbase, DisasContext, base);
 
-    if (unlikely(dc->base.singlestep_enabled || dc->ss_active)) {
+    if (unlikely(dc->ss_active)) {
         /* Note that this means single stepping WFI doesn't halt the CPU.
          * For conditional branch insns this is harmless unreachable code as
          * gen_goto_tb() has already handled emitting the debug exception
@@ -XXX,XX +XXX,XX @@ static void aarch64_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
             /* fall through */
         case DISAS_EXIT:
         case DISAS_JUMP:
-            if (dc->base.singlestep_enabled) {
-                gen_exception_internal(EXCP_DEBUG);
-            } else {
-                gen_step_complete_exception(dc);
-            }
+            gen_step_complete_exception(dc);
             break;
         case DISAS_NORETURN:
             break;
diff --git a/target/arm/translate.c b/target/arm/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/arm/translate.c
+++ b/target/arm/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_exception_internal(int excp)
     tcg_temp_free_i32(tcg_excp);
 }
 
-static void gen_step_complete_exception(DisasContext *s)
+static void gen_singlestep_exception(DisasContext *s)
 {
     /* We just completed step of an insn. Move from Active-not-pending
      * to Active-pending, and then also take the swstep exception.
@@ -XXX,XX +XXX,XX @@ static void gen_step_complete_exception(DisasContext *s)
     s->base.is_jmp = DISAS_NORETURN;
 }
 
-static void gen_singlestep_exception(DisasContext *s)
-{
-    /* Generate the right kind of exception for singlestep, which is
-     * either the architectural singlestep or EXCP_DEBUG for QEMU's
-     * gdb singlestepping.
-     */
-    if (s->ss_active) {
-        gen_step_complete_exception(s);
-    } else {
-        gen_exception_internal(EXCP_DEBUG);
-    }
-}
-
-static inline bool is_singlestepping(DisasContext *s)
-{
-    /* Return true if we are singlestepping either because of
-     * architectural singlestep or QEMU gdbstub singlestep. This does
-     * not include the command line '-singlestep' mode which is rather
-     * misnamed as it only means "one instruction per TB" and doesn't
-     * affect the code we generate.
-     */
-    return s->base.singlestep_enabled || s->ss_active;
-}
-
 void clear_eci_state(DisasContext *s)
 {
     /*
@@ -XXX,XX +XXX,XX @@ static inline void gen_bx_excret_final_code(DisasContext *s)
     /* Is the new PC value in the magic range indicating exception return? */
     tcg_gen_brcondi_i32(TCG_COND_GEU, cpu_R[15], min_magic, excret_label);
     /* No: end the TB as we would for a DISAS_JMP */
-    if (is_singlestepping(s)) {
+    if (s->ss_active) {
         gen_singlestep_exception(s);
     } else {
         tcg_gen_exit_tb(NULL, 0);
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *s, int n, target_ulong dest)
 /* Jump, specifying which TB number to use if we gen_goto_tb() */
 static inline void gen_jmp_tb(DisasContext *s, uint32_t dest, int tbno)
 {
-    if (unlikely(is_singlestepping(s))) {
+    if (unlikely(s->ss_active)) {
         /* An indirect jump so that we still trigger the debug exception.  */
         gen_set_pc_im(s, dest);
         s->base.is_jmp = DISAS_JUMP;
@@ -XXX,XX +XXX,XX @@ static void arm_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cs)
     dc->page_start = dc->base.pc_first & TARGET_PAGE_MASK;
 
     /* If architectural single step active, limit to 1.  */
-    if (is_singlestepping(dc)) {
+    if (dc->ss_active) {
         dc->base.max_insns = 1;
     }
 
@@ -XXX,XX +XXX,XX @@ static void arm_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
          * insn codepath itself.
          */
         gen_bx_excret_final_code(dc);
-    } else if (unlikely(is_singlestepping(dc))) {
+    } else if (unlikely(dc->ss_active)) {
         /* Unconditional and "condition passed" instruction codepath. */
         switch (dc->base.is_jmp) {
         case DISAS_SWI:
@@ -XXX,XX +XXX,XX @@ static void arm_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
         /* "Condition failed" instruction codepath for the branch/trap insn */
         gen_set_label(dc->condlabel);
         gen_set_condexec(dc);
-        if (unlikely(is_singlestepping(dc))) {
+        if (unlikely(dc->ss_active)) {
             gen_set_pc_im(dc, dc->base.pc_next);
             gen_singlestep_exception(dc);
         } else {
-- 
2.25.1

GDB single-stepping is now handled generically.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/hppa/translate.c | 17 ++++-------------
 1 file changed, 4 insertions(+), 13 deletions(-)

diff --git a/target/hppa/translate.c b/target/hppa/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/hppa/translate.c
+++ b/target/hppa/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int which,
     } else {
         copy_iaoq_entry(cpu_iaoq_f, f, cpu_iaoq_b);
         copy_iaoq_entry(cpu_iaoq_b, b, ctx->iaoq_n_var);
-        if (ctx->base.singlestep_enabled) {
-            gen_excp_1(EXCP_DEBUG);
-        } else {
-            tcg_gen_lookup_and_goto_ptr();
-        }
+        tcg_gen_lookup_and_goto_ptr();
     }
 }
 
@@ -XXX,XX +XXX,XX @@ static bool do_rfi(DisasContext *ctx, bool rfi_r)
         gen_helper_rfi(cpu_env);
     }
     /* Exit the TB to recognize new interrupts.  */
-    if (ctx->base.singlestep_enabled) {
-        gen_excp_1(EXCP_DEBUG);
-    } else {
-        tcg_gen_exit_tb(NULL, 0);
-    }
+    tcg_gen_exit_tb(NULL, 0);
     ctx->base.is_jmp = DISAS_NORETURN;
 
     return nullify_end(ctx);
@@ -XXX,XX +XXX,XX @@ static void hppa_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
         nullify_save(ctx);
         /* FALLTHRU */
     case DISAS_IAQ_N_UPDATED:
-        if (ctx->base.singlestep_enabled) {
-            gen_excp_1(EXCP_DEBUG);
-        } else if (is_jmp != DISAS_IAQ_N_STALE_EXIT) {
+        if (is_jmp != DISAS_IAQ_N_STALE_EXIT) {
             tcg_gen_lookup_and_goto_ptr();
+            break;
         }
         /* FALLTHRU */
     case DISAS_EXIT:
-- 
2.25.1

We were using singlestep_enabled as a proxy for whether
translator_use_goto_tb would always return false.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/i386/tcg/translate.c | 5 +++--
 1 file changed, 3 insertions(+), 2 deletions(-)

diff --git a/target/i386/tcg/translate.c b/target/i386/tcg/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/i386/tcg/translate.c
+++ b/target/i386/tcg/translate.c
@@ -XXX,XX +XXX,XX @@ static void i386_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cpu)
     DisasContext *dc = container_of(dcbase, DisasContext, base);
     CPUX86State *env = cpu->env_ptr;
     uint32_t flags = dc->base.tb->flags;
+    uint32_t cflags = tb_cflags(dc->base.tb);
     int cpl = (flags >> HF_CPL_SHIFT) & 3;
     int iopl = (flags >> IOPL_SHIFT) & 3;
 
@@ -XXX,XX +XXX,XX @@ static void i386_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cpu)
     dc->cpuid_ext3_features = env->features[FEAT_8000_0001_ECX];
     dc->cpuid_7_0_ebx_features = env->features[FEAT_7_0_EBX];
     dc->cpuid_xsave_features = env->features[FEAT_XSAVE];
-    dc->jmp_opt = !(dc->base.singlestep_enabled ||
+    dc->jmp_opt = !((cflags & CF_NO_GOTO_TB) ||
                     (flags & (HF_TF_MASK | HF_INHIBIT_IRQ_MASK)));
     /*
      * If jmp_opt, we want to handle each string instruction individually.
      * For icount also disable repz optimization so that each iteration
      * is accounted separately.
      */
-    dc->repz_opt = !dc->jmp_opt && !(tb_cflags(dc->base.tb) & CF_USE_ICOUNT);
+    dc->repz_opt = !dc->jmp_opt && !(cflags & CF_USE_ICOUNT);
 
     dc->T0 = tcg_temp_new();
     dc->T1 = tcg_temp_new();
-- 
2.25.1

GDB single-stepping is now handled generically.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/i386/helper.h          | 1 -
 target/i386/tcg/misc_helper.c | 8 --------
 target/i386/tcg/translate.c   | 4 +---
 3 files changed, 1 insertion(+), 12 deletions(-)

diff --git a/target/i386/helper.h b/target/i386/helper.h
index XXXXXXX..XXXXXXX 100644
--- a/target/i386/helper.h
+++ b/target/i386/helper.h
@@ -XXX,XX +XXX,XX @@ DEF_HELPER_2(syscall, void, env, int)
 DEF_HELPER_2(sysret, void, env, int)
 #endif
 DEF_HELPER_FLAGS_2(pause, TCG_CALL_NO_WG, noreturn, env, int)
-DEF_HELPER_FLAGS_1(debug, TCG_CALL_NO_WG, noreturn, env)
 DEF_HELPER_1(reset_rf, void, env)
 DEF_HELPER_FLAGS_3(raise_interrupt, TCG_CALL_NO_WG, noreturn, env, int, int)
 DEF_HELPER_FLAGS_2(raise_exception, TCG_CALL_NO_WG, noreturn, env, int)
diff --git a/target/i386/tcg/misc_helper.c b/target/i386/tcg/misc_helper.c
index XXXXXXX..XXXXXXX 100644
--- a/target/i386/tcg/misc_helper.c
+++ b/target/i386/tcg/misc_helper.c
@@ -XXX,XX +XXX,XX @@ void QEMU_NORETURN helper_pause(CPUX86State *env, int next_eip_addend)
     do_pause(env);
 }
 
-void QEMU_NORETURN helper_debug(CPUX86State *env)
-{
-    CPUState *cs = env_cpu(env);
-
-    cs->exception_index = EXCP_DEBUG;
-    cpu_loop_exit(cs);
-}
-
 uint64_t helper_rdpkru(CPUX86State *env, uint32_t ecx)
 {
     if ((env->cr[4] & CR4_PKE_MASK) == 0) {
diff --git a/target/i386/tcg/translate.c b/target/i386/tcg/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/i386/tcg/translate.c
+++ b/target/i386/tcg/translate.c
@@ -XXX,XX +XXX,XX @@ do_gen_eob_worker(DisasContext *s, bool inhibit, bool recheck_tf, bool jr)
     if (s->base.tb->flags & HF_RF_MASK) {
         gen_helper_reset_rf(cpu_env);
     }
-    if (s->base.singlestep_enabled) {
-        gen_helper_debug(cpu_env);
-    } else if (recheck_tf) {
+    if (recheck_tf) {
         gen_helper_rechecking_single_step(cpu_env);
         tcg_gen_exit_tb(NULL, 0);
     } else if (s->flags & HF_TF_MASK) {
-- 
2.25.1

GDB single-stepping is now handled generically.

Acked-by: Laurent Vivier <laurent@vivier.eu>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/m68k/translate.c | 44 +++++++++--------------------------------
 1 file changed, 9 insertions(+), 35 deletions(-)

diff --git a/target/m68k/translate.c b/target/m68k/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/m68k/translate.c
+++ b/target/m68k/translate.c
@@ -XXX,XX +XXX,XX @@ static void do_writebacks(DisasContext *s)
     }
 }
 
-static bool is_singlestepping(DisasContext *s)
-{
-    /*
-     * Return true if we are singlestepping either because of
-     * architectural singlestep or QEMU gdbstub singlestep. This does
-     * not include the command line '-singlestep' mode which is rather
-     * misnamed as it only means "one instruction per TB" and doesn't
-     * affect the code we generate.
-     */
-    return s->base.singlestep_enabled || s->ss_active;
-}
-
 /* is_jmp field values */
 #define DISAS_JUMP      DISAS_TARGET_0 /* only pc was modified dynamically */
 #define DISAS_EXIT      DISAS_TARGET_1 /* cpu state was modified dynamically */
@@ -XXX,XX +XXX,XX @@ static void gen_exception(DisasContext *s, uint32_t dest, int nr)
     s->base.is_jmp = DISAS_NORETURN;
 }
 
-static void gen_singlestep_exception(DisasContext *s)
-{
-    /*
-     * Generate the right kind of exception for singlestep, which is
-     * either the architectural singlestep or EXCP_DEBUG for QEMU's
-     * gdb singlestepping.
-     */
-    if (s->ss_active) {
-        gen_raise_exception(EXCP_TRACE);
-    } else {
-        gen_raise_exception(EXCP_DEBUG);
-    }
-}
-
 static inline void gen_addr_fault(DisasContext *s)
 {
     gen_exception(s, s->base.pc_next, EXCP_ADDRESS);
@@ -XXX,XX +XXX,XX @@ static void gen_exit_tb(DisasContext *s)
 /* Generate a jump to an immediate address.  */
 static void gen_jmp_tb(DisasContext *s, int n, uint32_t dest)
 {
-    if (unlikely(is_singlestepping(s))) {
+    if (unlikely(s->ss_active)) {
         update_cc_op(s);
         tcg_gen_movi_i32(QREG_PC, dest);
-        gen_singlestep_exception(s);
+        gen_raise_exception(EXCP_TRACE);
     } else if (translator_use_goto_tb(&s->base, dest)) {
         tcg_gen_goto_tb(n);
         tcg_gen_movi_i32(QREG_PC, dest);
@@ -XXX,XX +XXX,XX @@ static void m68k_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cpu)
 
     dc->ss_active = (M68K_SR_TRACE(env->sr) == M68K_SR_TRACE_ANY_INS);
     /* If architectural single step active, limit to 1 */
-    if (is_singlestepping(dc)) {
+    if (dc->ss_active) {
         dc->base.max_insns = 1;
     }
 }
@@ -XXX,XX +XXX,XX @@ static void m68k_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
         break;
     case DISAS_TOO_MANY:
         update_cc_op(dc);
-        if (is_singlestepping(dc)) {
+        if (dc->ss_active) {
             tcg_gen_movi_i32(QREG_PC, dc->pc);
-            gen_singlestep_exception(dc);
+            gen_raise_exception(EXCP_TRACE);
         } else {
             gen_jmp_tb(dc, 0, dc->pc);
         }
         break;
     case DISAS_JUMP:
         /* We updated CC_OP and PC in gen_jmp/gen_jmp_im.  */
-        if (is_singlestepping(dc)) {
-            gen_singlestep_exception(dc);
+        if (dc->ss_active) {
+            gen_raise_exception(EXCP_TRACE);
         } else {
             tcg_gen_lookup_and_goto_ptr();
         }
@@ -XXX,XX +XXX,XX @@ static void m68k_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
          * We updated CC_OP and PC in gen_exit_tb, but also modified
          * other state that may require returning to the main loop.
          */
-        if (is_singlestepping(dc)) {
-            gen_singlestep_exception(dc);
+        if (dc->ss_active) {
+            gen_raise_exception(EXCP_TRACE);
         } else {
             tcg_gen_exit_tb(NULL, 0);
         }
-- 
2.25.1

We were using singlestep_enabled as a proxy for whether
translator_use_goto_tb would always return false.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/microblaze/translate.c | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/target/microblaze/translate.c b/target/microblaze/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/microblaze/translate.c
+++ b/target/microblaze/translate.c
@@ -XXX,XX +XXX,XX @@ static void mb_tr_tb_stop(DisasContextBase *dcb, CPUState *cs)
         break;
 
     case DISAS_JUMP:
-        if (dc->jmp_dest != -1 && !cs->singlestep_enabled) {
+        if (dc->jmp_dest != -1 && !(tb_cflags(dc->base.tb) & CF_NO_GOTO_TB)) {
             /* Direct jump. */
             tcg_gen_discard_i32(cpu_btarget);
 
@@ -XXX,XX +XXX,XX @@ static void mb_tr_tb_stop(DisasContextBase *dcb, CPUState *cs)
             return;
         }
 
-        /* Indirect jump (or direct jump w/ singlestep) */
+        /* Indirect jump (or direct jump w/ goto_tb disabled) */
         tcg_gen_mov_i32(cpu_pc, cpu_btarget);
         tcg_gen_discard_i32(cpu_btarget);
 
-- 
2.25.1

GDB single-stepping is now handled generically.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/microblaze/translate.c | 14 ++------------
 1 file changed, 2 insertions(+), 12 deletions(-)

diff --git a/target/microblaze/translate.c b/target/microblaze/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/microblaze/translate.c
+++ b/target/microblaze/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_raise_hw_excp(DisasContext *dc, uint32_t esr_ec)
 
 static void gen_goto_tb(DisasContext *dc, int n, target_ulong dest)
 {
-    if (dc->base.singlestep_enabled) {
-        TCGv_i32 tmp = tcg_const_i32(EXCP_DEBUG);
-        tcg_gen_movi_i32(cpu_pc, dest);
-        gen_helper_raise_exception(cpu_env, tmp);
-        tcg_temp_free_i32(tmp);
-    } else if (translator_use_goto_tb(&dc->base, dest)) {
+    if (translator_use_goto_tb(&dc->base, dest)) {
         tcg_gen_goto_tb(n);
         tcg_gen_movi_i32(cpu_pc, dest);
         tcg_gen_exit_tb(dc->base.tb, n);
@@ -XXX,XX +XXX,XX @@ static void mb_tr_tb_stop(DisasContextBase *dcb, CPUState *cs)
         /* Indirect jump (or direct jump w/ goto_tb disabled) */
         tcg_gen_mov_i32(cpu_pc, cpu_btarget);
         tcg_gen_discard_i32(cpu_btarget);
-
-        if (unlikely(cs->singlestep_enabled)) {
-            gen_raise_exception(dc, EXCP_DEBUG);
-        } else {
-            tcg_gen_lookup_and_goto_ptr();
-        }
+        tcg_gen_lookup_and_goto_ptr();
         return;
 
     default:
-- 
2.25.1

As per an ancient comment in mips_tr_translate_insn about the
expectations of gdb, when restarting the insn in a delay slot
we also re-execute the branch.  Which means that we are
expected to execute two insns in this case.

This has been broken since 8b86d6d2580, where we forced max_insns
to 1 while single-stepping.  This resulted in an exit from the
translator loop after the branch but before the delay slot is
translated.

Increase the max_insns to 2 for this case.  In addition, bypass
the end-of-page check, for when the branch itself ends the page.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/mips/tcg/translate.c | 25 ++++++++++++++++---------
 1 file changed, 16 insertions(+), 9 deletions(-)

diff --git a/target/mips/tcg/translate.c b/target/mips/tcg/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/mips/tcg/translate.c
+++ b/target/mips/tcg/translate.c
@@ -XXX,XX +XXX,XX @@ static void mips_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cs)
     ctx->default_tcg_memop_mask = (ctx->insn_flags & (ISA_MIPS_R6 |
                                   INSN_LOONGSON3A)) ? MO_UNALN : MO_ALIGN;
 
+    /*
+     * Execute a branch and its delay slot as a single instruction.
+     * This is what GDB expects and is consistent with what the
+     * hardware does (e.g. if a delay slot instruction faults, the
+     * reported PC is the PC of the branch).
+     */
+    if (ctx->base.singlestep_enabled && (ctx->hflags & MIPS_HFLAG_BMASK)) {
+        ctx->base.max_insns = 2;
+    }
+
     LOG_DISAS("\ntb %p idx %d hflags %04x\n", ctx->base.tb, ctx->mem_idx,
               ctx->hflags);
 }
@@ -XXX,XX +XXX,XX @@ static void mips_tr_translate_insn(DisasContextBase *dcbase, CPUState *cs)
     if (ctx->base.is_jmp != DISAS_NEXT) {
         return;
     }
+
     /*
-     * Execute a branch and its delay slot as a single instruction.
-     * This is what GDB expects and is consistent with what the
-     * hardware does (e.g. if a delay slot instruction faults, the
-     * reported PC is the PC of the branch).
+     * End the TB on (most) page crossings.
+     * See mips_tr_init_disas_context about single-stepping a branch
+     * together with its delay slot.
      */
-    if (ctx->base.singlestep_enabled &&
-        (ctx->hflags & MIPS_HFLAG_BMASK) == 0) {
-        ctx->base.is_jmp = DISAS_TOO_MANY;
-    }
-    if (ctx->base.pc_next - ctx->page_start >= TARGET_PAGE_SIZE) {
+    if (ctx->base.pc_next - ctx->page_start >= TARGET_PAGE_SIZE
+        && !ctx->base.singlestep_enabled) {
         ctx->base.is_jmp = DISAS_TOO_MANY;
     }
 }
-- 
2.25.1

GDB single-stepping is now handled generically.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/mips/tcg/translate.c | 50 +++++++++++++------------------------
 1 file changed, 18 insertions(+), 32 deletions(-)

diff --git a/target/mips/tcg/translate.c b/target/mips/tcg/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/mips/tcg/translate.c
+++ b/target/mips/tcg/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
         tcg_gen_exit_tb(ctx->base.tb, n);
     } else {
         gen_save_pc(dest);
-        if (ctx->base.singlestep_enabled) {
-            save_cpu_state(ctx, 0);
-            gen_helper_raise_exception_debug(cpu_env);
-        } else {
-            tcg_gen_lookup_and_goto_ptr();
-        }
+        tcg_gen_lookup_and_goto_ptr();
     }
 }
 
@@ -XXX,XX +XXX,XX @@ static void gen_branch(DisasContext *ctx, int insn_bytes)
             } else {
                 tcg_gen_mov_tl(cpu_PC, btarget);
             }
-            if (ctx->base.singlestep_enabled) {
-                save_cpu_state(ctx, 0);
-                gen_helper_raise_exception_debug(cpu_env);
-            }
             tcg_gen_lookup_and_goto_ptr();
             break;
         default:
@@ -XXX,XX +XXX,XX @@ static void mips_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
 {
     DisasContext *ctx = container_of(dcbase, DisasContext, base);
 
-    if (ctx->base.singlestep_enabled && ctx->base.is_jmp != DISAS_NORETURN) {
-        save_cpu_state(ctx, ctx->base.is_jmp != DISAS_EXIT);
-        gen_helper_raise_exception_debug(cpu_env);
-    } else {
-        switch (ctx->base.is_jmp) {
-        case DISAS_STOP:
-            gen_save_pc(ctx->base.pc_next);
-            tcg_gen_lookup_and_goto_ptr();
-            break;
-        case DISAS_NEXT:
-        case DISAS_TOO_MANY:
-            save_cpu_state(ctx, 0);
-            gen_goto_tb(ctx, 0, ctx->base.pc_next);
-            break;
-        case DISAS_EXIT:
-            tcg_gen_exit_tb(NULL, 0);
-            break;
-        case DISAS_NORETURN:
-            break;
-        default:
-            g_assert_not_reached();
-        }
+    switch (ctx->base.is_jmp) {
+    case DISAS_STOP:
+        gen_save_pc(ctx->base.pc_next);
+        tcg_gen_lookup_and_goto_ptr();
+        break;
+    case DISAS_NEXT:
+    case DISAS_TOO_MANY:
+        save_cpu_state(ctx, 0);
+        gen_goto_tb(ctx, 0, ctx->base.pc_next);
+        break;
+    case DISAS_EXIT:
+        tcg_gen_exit_tb(NULL, 0);
+        break;
+    case DISAS_NORETURN:
+        break;
+    default:
+        g_assert_not_reached();
     }
 }
 
-- 
2.25.1

GDB single-stepping is now handled generically.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/openrisc/translate.c | 18 +++---------------
 1 file changed, 3 insertions(+), 15 deletions(-)

diff --git a/target/openrisc/translate.c b/target/openrisc/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/openrisc/translate.c
+++ b/target/openrisc/translate.c
@@ -XXX,XX +XXX,XX @@ static void openrisc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
             /* The jump destination is indirect/computed; use jmp_pc.  */
             tcg_gen_mov_tl(cpu_pc, jmp_pc);
             tcg_gen_discard_tl(jmp_pc);
-            if (unlikely(dc->base.singlestep_enabled)) {
-                gen_exception(dc, EXCP_DEBUG);
-            } else {
-                tcg_gen_lookup_and_goto_ptr();
-            }
+            tcg_gen_lookup_and_goto_ptr();
             break;
         }
         /* The jump destination is direct; use jmp_pc_imm.
@@ -XXX,XX +XXX,XX @@ static void openrisc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
             break;
         }
         tcg_gen_movi_tl(cpu_pc, jmp_dest);
-        if (unlikely(dc->base.singlestep_enabled)) {
-            gen_exception(dc, EXCP_DEBUG);
-        } else {
-            tcg_gen_lookup_and_goto_ptr();
-        }
+        tcg_gen_lookup_and_goto_ptr();
         break;
 
     case DISAS_EXIT:
-        if (unlikely(dc->base.singlestep_enabled)) {
-            gen_exception(dc, EXCP_DEBUG);
-        } else {
-            tcg_gen_exit_tb(NULL, 0);
-        }
+        tcg_gen_exit_tb(NULL, 0);
         break;
     default:
         g_assert_not_reached();
-- 
2.25.1

GDB single-stepping is now handled generically.
Reuse gen_debug_exception to handle architectural debug exceptions.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/ppc/translate.c | 38 ++++++++------------------------------
 1 file changed, 8 insertions(+), 30 deletions(-)

diff --git a/target/ppc/translate.c b/target/ppc/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/ppc/translate.c
+++ b/target/ppc/translate.c
@@ -XXX,XX +XXX,XX @@
 
 #define CPU_SINGLE_STEP 0x1
 #define CPU_BRANCH_STEP 0x2
-#define GDBSTUB_SINGLE_STEP 0x4
 
 /* Include definitions for instructions classes and implementations flags */
 /* #define PPC_DEBUG_DISAS */
@@ -XXX,XX +XXX,XX @@ static uint32_t gen_prep_dbgex(DisasContext *ctx)
 
 static void gen_debug_exception(DisasContext *ctx)
 {
-    gen_helper_raise_exception(cpu_env, tcg_constant_i32(EXCP_DEBUG));
+    gen_helper_raise_exception(cpu_env, tcg_constant_i32(gen_prep_dbgex(ctx)));
     ctx->base.is_jmp = DISAS_NORETURN;
 }
 
@@ -XXX,XX +XXX,XX @@ static inline bool use_goto_tb(DisasContext *ctx, target_ulong dest)
 
 static void gen_lookup_and_goto_ptr(DisasContext *ctx)
 {
-    int sse = ctx->singlestep_enabled;
-    if (unlikely(sse)) {
-        if (sse & GDBSTUB_SINGLE_STEP) {
-            gen_debug_exception(ctx);
-        } else if (sse & (CPU_SINGLE_STEP | CPU_BRANCH_STEP)) {
-            gen_helper_raise_exception(cpu_env, tcg_constant_i32(gen_prep_dbgex(ctx)));
-        } else {
-            tcg_gen_exit_tb(NULL, 0);
-        }
+    if (unlikely(ctx->singlestep_enabled)) {
+        gen_debug_exception(ctx);
     } else {
         tcg_gen_lookup_and_goto_ptr();
     }
@@ -XXX,XX +XXX,XX @@ static void ppc_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cs)
     ctx->singlestep_enabled = 0;
     if ((hflags >> HFLAGS_SE) & 1) {
         ctx->singlestep_enabled |= CPU_SINGLE_STEP;
+        ctx->base.max_insns = 1;
     }
     if ((hflags >> HFLAGS_BE) & 1) {
         ctx->singlestep_enabled |= CPU_BRANCH_STEP;
     }
-    if (unlikely(ctx->base.singlestep_enabled)) {
-        ctx->singlestep_enabled |= GDBSTUB_SINGLE_STEP;
-    }
-
-    if (ctx->singlestep_enabled & (CPU_SINGLE_STEP | GDBSTUB_SINGLE_STEP)) {
-        ctx->base.max_insns = 1;
-    }
 }
 
 static void ppc_tr_tb_start(DisasContextBase *db, CPUState *cs)
@@ -XXX,XX +XXX,XX @@ static void ppc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
     DisasContext *ctx = container_of(dcbase, DisasContext, base);
     DisasJumpType is_jmp = ctx->base.is_jmp;
     target_ulong nip = ctx->base.pc_next;
-    int sse;
 
     if (is_jmp == DISAS_NORETURN) {
         /* We have already exited the TB. */
@@ -XXX,XX +XXX,XX @@ static void ppc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
     }
 
     /* Honor single stepping. */
-    sse = ctx->singlestep_enabled & (CPU_SINGLE_STEP | GDBSTUB_SINGLE_STEP);
-    if (unlikely(sse)) {
+    if (unlikely(ctx->singlestep_enabled & CPU_SINGLE_STEP)
+        && (nip <= 0x100 || nip > 0xf00)) {
         switch (is_jmp) {
         case DISAS_TOO_MANY:
         case DISAS_EXIT_UPDATE:
@@ -XXX,XX +XXX,XX @@ static void ppc_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
             g_assert_not_reached();
         }
 
-        if (sse & GDBSTUB_SINGLE_STEP) {
-            gen_debug_exception(ctx);
-            return;
-        }
-        /* else CPU_SINGLE_STEP... */
-        if (nip <= 0x100 || nip > 0xf00) {
-            gen_helper_raise_exception(cpu_env, tcg_constant_i32(gen_prep_dbgex(ctx)));
-            return;
-        }
+        gen_debug_exception(ctx);
+        return;
     }
 
     switch (is_jmp) {
-- 
2.25.1

We have already set DISAS_NORETURN in generate_exception,
which makes the exit_tb unreachable.

Reviewed-by: Alistair Francis <alistair.francis@wdc.com>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/riscv/insn_trans/trans_privileged.c.inc | 6 +-----
 1 file changed, 1 insertion(+), 5 deletions(-)

diff --git a/target/riscv/insn_trans/trans_privileged.c.inc b/target/riscv/insn_trans/trans_privileged.c.inc
index XXXXXXX..XXXXXXX 100644
--- a/target/riscv/insn_trans/trans_privileged.c.inc
+++ b/target/riscv/insn_trans/trans_privileged.c.inc
@@ -XXX,XX +XXX,XX @@ static bool trans_ecall(DisasContext *ctx, arg_ecall *a)
 {
     /* always generates U-level ECALL, fixed in do_interrupt handler */
     generate_exception(ctx, RISCV_EXCP_U_ECALL);
-    exit_tb(ctx); /* no chaining */
-    ctx->base.is_jmp = DISAS_NORETURN;
     return true;
 }
 
@@ -XXX,XX +XXX,XX @@ static bool trans_ebreak(DisasContext *ctx, arg_ebreak *a)
         post   = opcode_at(&ctx->base, post_addr);
     }
 
-    if  (pre == 0x01f01013 && ebreak == 0x00100073 && post == 0x40705013) {
+    if (pre == 0x01f01013 && ebreak == 0x00100073 && post == 0x40705013) {
         generate_exception(ctx, RISCV_EXCP_SEMIHOST);
     } else {
         generate_exception(ctx, RISCV_EXCP_BREAKPOINT);
     }
-    exit_tb(ctx); /* no chaining */
-    ctx->base.is_jmp = DISAS_NORETURN;
     return true;
 }
 
-- 
2.25.1

GDB single-stepping is now handled generically, which means
we don't need to do anything in the wrappers.

Reviewed-by: Alistair Francis <alistair.francis@wdc.com>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/riscv/translate.c                      | 27 +------------------
 .../riscv/insn_trans/trans_privileged.c.inc   |  4 +--
 target/riscv/insn_trans/trans_rvi.c.inc       |  8 +++---
 target/riscv/insn_trans/trans_rvv.c.inc       |  2 +-
 4 files changed, 7 insertions(+), 34 deletions(-)

diff --git a/target/riscv/translate.c b/target/riscv/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/riscv/translate.c
+++ b/target/riscv/translate.c
@@ -XXX,XX +XXX,XX @@ static void generate_exception_mtval(DisasContext *ctx, int excp)
     ctx->base.is_jmp = DISAS_NORETURN;
 }
 
-static void gen_exception_debug(void)
-{
-    gen_helper_raise_exception(cpu_env, tcg_constant_i32(EXCP_DEBUG));
-}
-
-/* Wrapper around tcg_gen_exit_tb that handles single stepping */
-static void exit_tb(DisasContext *ctx)
-{
-    if (ctx->base.singlestep_enabled) {
-        gen_exception_debug();
-    } else {
-        tcg_gen_exit_tb(NULL, 0);
-    }
-}
-
-/* Wrapper around tcg_gen_lookup_and_goto_ptr that handles single stepping */
-static void lookup_and_goto_ptr(DisasContext *ctx)
-{
-    if (ctx->base.singlestep_enabled) {
-        gen_exception_debug();
-    } else {
-        tcg_gen_lookup_and_goto_ptr();
-    }
-}
-
 static void gen_exception_illegal(DisasContext *ctx)
 {
     generate_exception(ctx, RISCV_EXCP_ILLEGAL_INST);
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
         tcg_gen_exit_tb(ctx->base.tb, n);
     } else {
         tcg_gen_movi_tl(cpu_pc, dest);
-        lookup_and_goto_ptr(ctx);
+        tcg_gen_lookup_and_goto_ptr();
     }
 }
 
diff --git a/target/riscv/insn_trans/trans_privileged.c.inc b/target/riscv/insn_trans/trans_privileged.c.inc
index XXXXXXX..XXXXXXX 100644
--- a/target/riscv/insn_trans/trans_privileged.c.inc
+++ b/target/riscv/insn_trans/trans_privileged.c.inc
@@ -XXX,XX +XXX,XX @@ static bool trans_sret(DisasContext *ctx, arg_sret *a)
 
     if (has_ext(ctx, RVS)) {
         gen_helper_sret(cpu_pc, cpu_env, cpu_pc);
-        exit_tb(ctx); /* no chaining */
+        tcg_gen_exit_tb(NULL, 0); /* no chaining */
         ctx->base.is_jmp = DISAS_NORETURN;
     } else {
         return false;
@@ -XXX,XX +XXX,XX @@ static bool trans_mret(DisasContext *ctx, arg_mret *a)
 #ifndef CONFIG_USER_ONLY
     tcg_gen_movi_tl(cpu_pc, ctx->base.pc_next);
     gen_helper_mret(cpu_pc, cpu_env, cpu_pc);
-    exit_tb(ctx); /* no chaining */
+    tcg_gen_exit_tb(NULL, 0); /* no chaining */
     ctx->base.is_jmp = DISAS_NORETURN;
     return true;
 #else
diff --git a/target/riscv/insn_trans/trans_rvi.c.inc b/target/riscv/insn_trans/trans_rvi.c.inc
index XXXXXXX..XXXXXXX 100644
--- a/target/riscv/insn_trans/trans_rvi.c.inc
+++ b/target/riscv/insn_trans/trans_rvi.c.inc
@@ -XXX,XX +XXX,XX @@ static bool trans_jalr(DisasContext *ctx, arg_jalr *a)
     if (a->rd != 0) {
         tcg_gen_movi_tl(cpu_gpr[a->rd], ctx->pc_succ_insn);
     }
-
-    /* No chaining with JALR. */
-    lookup_and_goto_ptr(ctx);
+    tcg_gen_lookup_and_goto_ptr();
 
     if (misaligned) {
         gen_set_label(misaligned);
@@ -XXX,XX +XXX,XX @@ static bool trans_fence_i(DisasContext *ctx, arg_fence_i *a)
      * however we need to end the translation block
      */
     tcg_gen_movi_tl(cpu_pc, ctx->pc_succ_insn);
-    exit_tb(ctx);
+    tcg_gen_exit_tb(NULL, 0);
     ctx->base.is_jmp = DISAS_NORETURN;
     return true;
 }
@@ -XXX,XX +XXX,XX @@ static bool do_csr_post(DisasContext *ctx)
 {
     /* We may have changed important cpu state -- exit to main loop. */
     tcg_gen_movi_tl(cpu_pc, ctx->pc_succ_insn);
-    exit_tb(ctx);
+    tcg_gen_exit_tb(NULL, 0);
     ctx->base.is_jmp = DISAS_NORETURN;
     return true;
 }
diff --git a/target/riscv/insn_trans/trans_rvv.c.inc b/target/riscv/insn_trans/trans_rvv.c.inc
index XXXXXXX..XXXXXXX 100644
--- a/target/riscv/insn_trans/trans_rvv.c.inc
+++ b/target/riscv/insn_trans/trans_rvv.c.inc
@@ -XXX,XX +XXX,XX @@ static bool trans_vsetvl(DisasContext *ctx, arg_vsetvl *a)
     gen_set_gpr(ctx, a->rd, dst);
 
     tcg_gen_movi_tl(cpu_pc, ctx->pc_succ_insn);
-    lookup_and_goto_ptr(ctx);
+    tcg_gen_lookup_and_goto_ptr();
     ctx->base.is_jmp = DISAS_NORETURN;
     return true;
 }
-- 
2.25.1

GDB single-stepping is now handled generically.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/rx/helper.h    |  1 -
 target/rx/op_helper.c |  8 --------
 target/rx/translate.c | 12 ++----------
 3 files changed, 2 insertions(+), 19 deletions(-)

diff --git a/target/rx/helper.h b/target/rx/helper.h
index XXXXXXX..XXXXXXX 100644
--- a/target/rx/helper.h
+++ b/target/rx/helper.h
@@ -XXX,XX +XXX,XX @@ DEF_HELPER_1(raise_illegal_instruction, noreturn, env)
 DEF_HELPER_1(raise_access_fault, noreturn, env)
 DEF_HELPER_1(raise_privilege_violation, noreturn, env)
 DEF_HELPER_1(wait, noreturn, env)
-DEF_HELPER_1(debug, noreturn, env)
 DEF_HELPER_2(rxint, noreturn, env, i32)
 DEF_HELPER_1(rxbrk, noreturn, env)
 DEF_HELPER_FLAGS_3(fadd, TCG_CALL_NO_WG, f32, env, f32, f32)
diff --git a/target/rx/op_helper.c b/target/rx/op_helper.c
index XXXXXXX..XXXXXXX 100644
--- a/target/rx/op_helper.c
+++ b/target/rx/op_helper.c
@@ -XXX,XX +XXX,XX @@ void QEMU_NORETURN helper_wait(CPURXState *env)
     raise_exception(env, EXCP_HLT, 0);
 }
 
-void QEMU_NORETURN helper_debug(CPURXState *env)
-{
-    CPUState *cs = env_cpu(env);
-
-    cs->exception_index = EXCP_DEBUG;
-    cpu_loop_exit(cs);
-}
-
 void QEMU_NORETURN helper_rxint(CPURXState *env, uint32_t vec)
 {
     raise_exception(env, 0x100 + vec, 0);
diff --git a/target/rx/translate.c b/target/rx/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/rx/translate.c
+++ b/target/rx/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *dc, int n, target_ulong dest)
         tcg_gen_exit_tb(dc->base.tb, n);
     } else {
         tcg_gen_movi_i32(cpu_pc, dest);
-        if (dc->base.singlestep_enabled) {
-            gen_helper_debug(cpu_env);
-        } else {
-            tcg_gen_lookup_and_goto_ptr();
-        }
+        tcg_gen_lookup_and_goto_ptr();
     }
     dc->base.is_jmp = DISAS_NORETURN;
 }
@@ -XXX,XX +XXX,XX @@ static void rx_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
         gen_goto_tb(ctx, 0, dcbase->pc_next);
         break;
     case DISAS_JUMP:
-        if (ctx->base.singlestep_enabled) {
-            gen_helper_debug(cpu_env);
-        } else {
-            tcg_gen_lookup_and_goto_ptr();
-        }
+        tcg_gen_lookup_and_goto_ptr();
         break;
     case DISAS_UPDATE:
         tcg_gen_movi_i32(cpu_pc, ctx->base.pc_next);
-- 
2.25.1

GDB single-stepping is now handled generically.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/s390x/tcg/translate.c | 8 ++------
 1 file changed, 2 insertions(+), 6 deletions(-)

diff --git a/target/s390x/tcg/translate.c b/target/s390x/tcg/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/s390x/tcg/translate.c
+++ b/target/s390x/tcg/translate.c
@@ -XXX,XX +XXX,XX @@ struct DisasContext {
     uint64_t pc_tmp;
     uint32_t ilen;
     enum cc_op cc_op;
-    bool do_debug;
 };
 
 /* Information carried about a condition to be evaluated.  */
@@ -XXX,XX +XXX,XX @@ static void s390x_tr_init_disas_context(DisasContextBase *dcbase, CPUState *cs)
 
     dc->cc_op = CC_OP_DYNAMIC;
     dc->ex_value = dc->base.tb->cs_base;
-    dc->do_debug = dc->base.singlestep_enabled;
 }
 
 static void s390x_tr_tb_start(DisasContextBase *db, CPUState *cs)
@@ -XXX,XX +XXX,XX @@ static void s390x_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
         /* FALLTHRU */
     case DISAS_PC_CC_UPDATED:
         /* Exit the TB, either by raising a debug exception or by return.  */
-        if (dc->do_debug) {
-            gen_exception(EXCP_DEBUG);
-        } else if ((dc->base.tb->flags & FLAG_MASK_PER) ||
-                   dc->base.is_jmp == DISAS_PC_STALE_NOCHAIN) {
+        if ((dc->base.tb->flags & FLAG_MASK_PER) ||
+             dc->base.is_jmp == DISAS_PC_STALE_NOCHAIN) {
             tcg_gen_exit_tb(NULL, 0);
         } else {
             tcg_gen_lookup_and_goto_ptr();
-- 
2.25.1

GDB single-stepping is now handled generically.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/sh4/helper.h    |  1 -
 target/sh4/op_helper.c |  5 -----
 target/sh4/translate.c | 14 +++-----------
 3 files changed, 3 insertions(+), 17 deletions(-)

diff --git a/target/sh4/helper.h b/target/sh4/helper.h
index XXXXXXX..XXXXXXX 100644
--- a/target/sh4/helper.h
+++ b/target/sh4/helper.h
@@ -XXX,XX +XXX,XX @@ DEF_HELPER_1(raise_illegal_instruction, noreturn, env)
 DEF_HELPER_1(raise_slot_illegal_instruction, noreturn, env)
 DEF_HELPER_1(raise_fpu_disable, noreturn, env)
 DEF_HELPER_1(raise_slot_fpu_disable, noreturn, env)
-DEF_HELPER_1(debug, noreturn, env)
 DEF_HELPER_1(sleep, noreturn, env)
 DEF_HELPER_2(trapa, noreturn, env, i32)
 DEF_HELPER_1(exclusive, noreturn, env)
diff --git a/target/sh4/op_helper.c b/target/sh4/op_helper.c
index XXXXXXX..XXXXXXX 100644
--- a/target/sh4/op_helper.c
+++ b/target/sh4/op_helper.c
@@ -XXX,XX +XXX,XX @@ void helper_raise_slot_fpu_disable(CPUSH4State *env)
     raise_exception(env, 0x820, 0);
 }
 
-void helper_debug(CPUSH4State *env)
-{
-    raise_exception(env, EXCP_DEBUG, 0);
-}
-
 void helper_sleep(CPUSH4State *env)
 {
     CPUState *cs = env_cpu(env);
diff --git a/target/sh4/translate.c b/target/sh4/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/sh4/translate.c
+++ b/target/sh4/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
         tcg_gen_exit_tb(ctx->base.tb, n);
     } else {
         tcg_gen_movi_i32(cpu_pc, dest);
-        if (ctx->base.singlestep_enabled) {
-            gen_helper_debug(cpu_env);
-        } else if (use_exit_tb(ctx)) {
+        if (use_exit_tb(ctx)) {
             tcg_gen_exit_tb(NULL, 0);
         } else {
             tcg_gen_lookup_and_goto_ptr();
@@ -XXX,XX +XXX,XX @@ static void gen_jump(DisasContext * ctx)
 	   delayed jump as immediate jump are conditinal jumps */
 	tcg_gen_mov_i32(cpu_pc, cpu_delayed_pc);
         tcg_gen_discard_i32(cpu_delayed_pc);
-        if (ctx->base.singlestep_enabled) {
-            gen_helper_debug(cpu_env);
-        } else if (use_exit_tb(ctx)) {
+        if (use_exit_tb(ctx)) {
             tcg_gen_exit_tb(NULL, 0);
         } else {
             tcg_gen_lookup_and_goto_ptr();
@@ -XXX,XX +XXX,XX @@ static void sh4_tr_tb_stop(DisasContextBase *dcbase, CPUState *cs)
     switch (ctx->base.is_jmp) {
     case DISAS_STOP:
         gen_save_cpu_state(ctx, true);
-        if (ctx->base.singlestep_enabled) {
-            gen_helper_debug(cpu_env);
-        } else {
-            tcg_gen_exit_tb(NULL, 0);
-        }
+        tcg_gen_exit_tb(NULL, 0);
         break;
     case DISAS_NEXT:
     case DISAS_TOO_MANY:
-- 
2.25.1

GDB single-stepping is now handled generically.

Reviewed-by: Philippe Mathieu-Daudé <f4bug@amsat.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/tricore/helper.h    |  1 -
 target/tricore/op_helper.c |  7 -------
 target/tricore/translate.c | 14 +-------------
 3 files changed, 1 insertion(+), 21 deletions(-)

diff --git a/target/tricore/helper.h b/target/tricore/helper.h
index XXXXXXX..XXXXXXX 100644
--- a/target/tricore/helper.h
+++ b/target/tricore/helper.h
@@ -XXX,XX +XXX,XX @@ DEF_HELPER_2(psw_write, void, env, i32)
 DEF_HELPER_1(psw_read, i32, env)
 /* Exceptions */
 DEF_HELPER_3(raise_exception_sync, noreturn, env, i32, i32)
-DEF_HELPER_2(qemu_excp, noreturn, env, i32)
diff --git a/target/tricore/op_helper.c b/target/tricore/op_helper.c
index XXXXXXX..XXXXXXX 100644
--- a/target/tricore/op_helper.c
+++ b/target/tricore/op_helper.c
@@ -XXX,XX +XXX,XX @@ static void raise_exception_sync_helper(CPUTriCoreState *env, uint32_t class,
     raise_exception_sync_internal(env, class, tin, pc, 0);
 }
 
-void helper_qemu_excp(CPUTriCoreState *env, uint32_t excp)
-{
-    CPUState *cs = env_cpu(env);
-    cs->exception_index = excp;
-    cpu_loop_exit(cs);
-}
-
 /* Addressing mode helper */
 
 static uint16_t reverse16(uint16_t val)
diff --git a/target/tricore/translate.c b/target/tricore/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/tricore/translate.c
+++ b/target/tricore/translate.c
@@ -XXX,XX +XXX,XX @@ static inline void gen_save_pc(target_ulong pc)
     tcg_gen_movi_tl(cpu_PC, pc);
 }
 
-static void generate_qemu_excp(DisasContext *ctx, int excp)
-{
-    TCGv_i32 tmp = tcg_const_i32(excp);
-    gen_helper_qemu_excp(cpu_env, tmp);
-    ctx->base.is_jmp = DISAS_NORETURN;
-    tcg_temp_free(tmp);
-}
-
 static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
 {
     if (translator_use_goto_tb(&ctx->base, dest)) {
@@ -XXX,XX +XXX,XX @@ static void gen_goto_tb(DisasContext *ctx, int n, target_ulong dest)
         tcg_gen_exit_tb(ctx->base.tb, n);
     } else {
         gen_save_pc(dest);
-        if (ctx->base.singlestep_enabled) {
-            generate_qemu_excp(ctx, EXCP_DEBUG);
-        } else {
-            tcg_gen_lookup_and_goto_ptr();
-        }
+        tcg_gen_lookup_and_goto_ptr();
     }
 }
 
-- 
2.25.1

GDB single-stepping is now handled generically.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/xtensa/translate.c | 25 ++++++++-----------------
 1 file changed, 8 insertions(+), 17 deletions(-)

diff --git a/target/xtensa/translate.c b/target/xtensa/translate.c
index XXXXXXX..XXXXXXX 100644
--- a/target/xtensa/translate.c
+++ b/target/xtensa/translate.c
@@ -XXX,XX +XXX,XX @@ static void gen_jump_slot(DisasContext *dc, TCGv dest, int slot)
     if (dc->icount) {
         tcg_gen_mov_i32(cpu_SR[ICOUNT], dc->next_icount);
     }
-    if (dc->base.singlestep_enabled) {
-        gen_exception(dc, EXCP_DEBUG);
+    if (dc->op_flags & XTENSA_OP_POSTPROCESS) {
+        slot = gen_postprocess(dc, slot);
+    }
+    if (slot >= 0) {
+        tcg_gen_goto_tb(slot);
+        tcg_gen_exit_tb(dc->base.tb, slot);
     } else {
-        if (dc->op_flags & XTENSA_OP_POSTPROCESS) {
-            slot = gen_postprocess(dc, slot);
-        }
-        if (slot >= 0) {
-            tcg_gen_goto_tb(slot);
-            tcg_gen_exit_tb(dc->base.tb, slot);
-        } else {
-            tcg_gen_exit_tb(NULL, 0);
-        }
+        tcg_gen_exit_tb(NULL, 0);
     }
     dc->base.is_jmp = DISAS_NORETURN;
 }
@@ -XXX,XX +XXX,XX @@ static void xtensa_tr_tb_stop(DisasContextBase *dcbase, CPUState *cpu)
     case DISAS_NORETURN:
         break;
     case DISAS_TOO_MANY:
-        if (dc->base.singlestep_enabled) {
-            tcg_gen_movi_i32(cpu_pc, dc->pc);
-            gen_exception(dc, EXCP_DEBUG);
-        } else {
-            gen_jumpi(dc, dc->pc, 0);
-        }
+        gen_jumpi(dc, dc->pc, 0);
         break;
     default:
         g_assert_not_reached();
-- 
2.25.1

This reverts commit 1b36e4f5a5de585210ea95f2257839c2312be28f.

Despite a comment saying why cpu_common_props cannot be placed in
a file that is compiled once, it was moved anyway.  Revert that.

Since then, Property is not defined in hw/core/cpu.h, so it is now
easier to declare a function to install the properties rather than
the Property array itself.

Cc: Eduardo Habkost <ehabkost@redhat.com>
Suggested-by: Peter Maydell <peter.maydell@linaro.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 include/hw/core/cpu.h |  1 +
 cpu.c                 | 21 +++++++++++++++++++++
 hw/core/cpu-common.c  | 17 +----------------
 3 files changed, 23 insertions(+), 16 deletions(-)

diff --git a/include/hw/core/cpu.h b/include/hw/core/cpu.h
index XXXXXXX..XXXXXXX 100644
--- a/include/hw/core/cpu.h
+++ b/include/hw/core/cpu.h
@@ -XXX,XX +XXX,XX @@ void QEMU_NORETURN cpu_abort(CPUState *cpu, const char *fmt, ...)
     GCC_FMT_ATTR(2, 3);
 
 /* $(top_srcdir)/cpu.c */
+void cpu_class_init_props(DeviceClass *dc);
 void cpu_exec_initfn(CPUState *cpu);
 void cpu_exec_realizefn(CPUState *cpu, Error **errp);
 void cpu_exec_unrealizefn(CPUState *cpu);
diff --git a/cpu.c b/cpu.c
index XXXXXXX..XXXXXXX 100644
--- a/cpu.c
+++ b/cpu.c
@@ -XXX,XX +XXX,XX @@ void cpu_exec_unrealizefn(CPUState *cpu)
     cpu_list_remove(cpu);
 }
 
+static Property cpu_common_props[] = {
+#ifndef CONFIG_USER_ONLY
+    /*
+     * Create a memory property for softmmu CPU object,
+     * so users can wire up its memory. (This can't go in hw/core/cpu.c
+     * because that file is compiled only once for both user-mode
+     * and system builds.) The default if no link is set up is to use
+     * the system address space.
+     */
+    DEFINE_PROP_LINK("memory", CPUState, memory, TYPE_MEMORY_REGION,
+                     MemoryRegion *),
+#endif
+    DEFINE_PROP_BOOL("start-powered-off", CPUState, start_powered_off, false),
+    DEFINE_PROP_END_OF_LIST(),
+};
+
+void cpu_class_init_props(DeviceClass *dc)
+{
+    device_class_set_props(dc, cpu_common_props);
+}
+
 void cpu_exec_initfn(CPUState *cpu)
 {
     cpu->as = NULL;
diff --git a/hw/core/cpu-common.c b/hw/core/cpu-common.c
index XXXXXXX..XXXXXXX 100644
--- a/hw/core/cpu-common.c
+++ b/hw/core/cpu-common.c
@@ -XXX,XX +XXX,XX @@ static int64_t cpu_common_get_arch_id(CPUState *cpu)
     return cpu->cpu_index;
 }
 
-static Property cpu_common_props[] = {
-#ifndef CONFIG_USER_ONLY
-    /* Create a memory property for softmmu CPU object,
-     * so users can wire up its memory. (This can't go in hw/core/cpu.c
-     * because that file is compiled only once for both user-mode
-     * and system builds.) The default if no link is set up is to use
-     * the system address space.
-     */
-    DEFINE_PROP_LINK("memory", CPUState, memory, TYPE_MEMORY_REGION,
-                     MemoryRegion *),
-#endif
-    DEFINE_PROP_BOOL("start-powered-off", CPUState, start_powered_off, false),
-    DEFINE_PROP_END_OF_LIST(),
-};
-
 static void cpu_class_init(ObjectClass *klass, void *data)
 {
     DeviceClass *dc = DEVICE_CLASS(klass);
@@ -XXX,XX +XXX,XX @@ static void cpu_class_init(ObjectClass *klass, void *data)
     dc->realize = cpu_common_realizefn;
     dc->unrealize = cpu_common_unrealizefn;
     dc->reset = cpu_common_reset;
-    device_class_set_props(dc, cpu_common_props);
+    cpu_class_init_props(dc);
     /*
      * Reason: CPUs still need special care by board code: wiring up
      * IRQs, adding reset handlers, halting non-first CPUs, ...
-- 
2.25.1