From: Matt Turner <[email protected]>

Every indirect branch that cannot use goto_tb ends in a dispatch that calls
helper_lookup_tb_ptr(). For an emulated compiler that is 8.4 billion helper
calls in a single translation unit: 24.6% of all TB exits take this path,
because jsr/ret/jmp have a register destination and because goto_tb is
restricted to same-page targets.

The helper itself is already tight, but each call pays for a call frame,
the can_do_io store, the get_tb_cpu_state() indirect call through
TCGCPUOps, curr_cflags(), and a breakpoint check, before it gets to the
jump cache probe that almost always hits (95.8% for this workload).

Emit the probe inline instead, for the callers that have migrated to
tcg_gen_goto_jc_*(). Those supply what it needs: the destination PC is in a
TCG temp, and the flags, cflags and cs_base the destination must match are
constants at translation time. The fast path is therefore a hash, four
guarded loads and a goto_ptr. Only a miss calls the helper, which still owns
filling the cache.

Two details matter for the generated code. The flags and cflags guards are
folded into a single aligned 64-bit load and compare, since the fields are
adjacent. And each path emits its own goto_ptr rather than branching to a
shared one: a temp live across the label is spilled and reloaded on every
dispatch, which cost 6.3% on its own.

The flags and cflags constants are safe against the other things that can
change them. CF_PARALLEL is only ever set by begin_parallel_context(),
which flushes first, so no block predating it survives to dispatch. gdb
single-step is only turned on with the CPU stopped, and a block translated
without CF_SINGLE_STEP can only be re-entered through tb_lookup(), which
from then on demands the new cflags -- so a stale-cflags block is never the
one running. Breakpoints are handled by CF_NO_GOTO_JC, added by the previous
patch: while one is set, blocks are translated with a cflags that both keeps
them off the inline path and keeps them unreachable from blocks already on
it. What is left is one_insn_per_tb and -d nochain; see below.

Measured with qemu-alpha running an emulated alpha gcc 16.2.0 compiling
the SQLite 3.45.1 amalgamation (255k lines, -O2) on an x86-64 host, LTO
build, on top of the preceding patches:

    before: 1,402,667,803,616 instructions
    after:    916,415,123,244 instructions   -34.67%

    before: 115.56s wall clock
    after:   85.59s wall clock               -25.94%

The gap between the two is the point at which this stops being a
straight-line win: the helper call was highly predictable work that the
host pipelined well, so removing it retires far fewer instructions than it
saves time. IPC falls from 2.48 to 2.17 across this patch for that reason.

Despite emitting more code, this also reduces instruction cache pressure,
because a dispatch no longer jumps into qemu's .text and evicts translated
code:

    before: 11,735,141,703 L1-icache-load-misses
    after:   7,154,863,292 L1-icache-load-misses   -39.0%

The mechanism is visible directly in a profile: helper_lookup_tb_ptr()
falls from 31.01% of samples to 0.35%, and qemu's own .text falls from
38.8% to 5.3%, with the balance moving into generated code.

Combined with the preceding patches, against an unmodified LTO build,
1,646,994,254,249 instructions fall to 916,415,123,244, or -44.36%. The
emulated compiler produces byte-identical output throughout.

A follow-up worth having: the probe is emitted entirely out of generic TCG
ops, and several backends can do much better than the result. x86_64 and
s390x have memory-operand comparisons; aarch64 can form env + off + h * 16
with a shift-add, load (tb, pc) and (cs_base, flags) with two ldp, and halve
the branches with ccmp. That wants a backend expansion of a dedicated
opcode, which is a separate series.

Signed-off-by: Matt Turner <[email protected]>
Signed-off-by: Richard Henderson <[email protected]>
Message-ID: <[email protected]>
---
 tcg/tcg-op.c | 110 ++++++++++++++++++++++++++++++++++++++++++++++++++-
 1 file changed, 108 insertions(+), 2 deletions(-)

diff --git a/tcg/tcg-op.c b/tcg/tcg-op.c
index 09d9a7963ae..cd10a00bf9e 100644
--- a/tcg/tcg-op.c
+++ b/tcg/tcg-op.c
@@ -28,6 +28,8 @@
 #include "tcg/tcg-op-common.h"
 #include "exec/translation-block.h"
 #include "exec/plugin-gen.h"
+#include "hw/core/cpu.h"
+#include "../accel/tcg/tb-hash.h"
 #include "tcg-internal.h"
 #include "tcg-has.h"
 
@@ -1942,12 +1944,17 @@ void tcg_gen_goto_tb(unsigned idx)
     tcg_gen_op1i(INDEX_op_goto_tb, 0, idx);
 }
 
+static void gen_goto_ptr(TCGTemp *ptr)
+{
+    tcg_gen_op1(INDEX_op_goto_ptr, TCG_TYPE_PTR, temp_arg(ptr));
+}
+
 static void gen_lookup_tb_ptr_and_goto(void)
 {
     TCGv_ptr ptr = tcg_temp_ebb_new_ptr();
 
     gen_helper_lookup_tb_ptr(ptr, tcg_env);
-    tcg_gen_op1i(INDEX_op_goto_ptr, TCG_TYPE_PTR, tcgv_ptr_arg(ptr));
+    gen_goto_ptr(tcgv_ptr_temp(ptr));
     tcg_temp_free_ptr(ptr);
 }
 
@@ -1962,6 +1969,96 @@ void tcg_gen_lookup_and_goto_ptr(void)
     gen_lookup_tb_ptr_and_goto();
 }
 
+static void gen_jc_hash(TCGTemp *h, TCGTemp *t, TCGTemp *pc)
+{
+#ifdef CONFIG_SOFTMMU
+    /*
+     * tb_jmp_cache_hash_func(), softmmu form.  TARGET_PAGE_BITS is a load
+     * from target_page in this translation unit, but it is decided long
+     * before any translation happens, so it is a constant here.
+     */
+    int shift = TARGET_PAGE_BITS - TB_JMP_PAGE_BITS;
+
+    gen_shri(TCG_TYPE_I64, t, pc, shift);
+    gen_xor(TCG_TYPE_I64, t, t, pc);
+    gen_shri(TCG_TYPE_I64, h, t, shift);
+    gen_andi(TCG_TYPE_I64, h, h, TB_JMP_PAGE_MASK);
+    gen_andi(TCG_TYPE_I64, t, t, TB_JMP_ADDR_MASK);
+    gen_or(TCG_TYPE_I64, h, h, t);
+#else
+    /* tb_jmp_cache_hash_func(), user-only form. */
+    gen_shri(TCG_TYPE_I64, h, pc, TB_JMP_CACHE_BITS);
+    gen_xor(TCG_TYPE_I64, h, h, pc);
+    gen_andi(TCG_TYPE_I64, h, h, TB_JMP_CACHE_SIZE - 1);
+#endif
+}
+
+static void gen_jc_probe(TCGTemp *pc, TCGTemp *cs_base, uint32_t flags)
+{
+    const TranslationBlock *tb = tcg_ctx->gen_tb;
+    TCGLabel *slow = gen_new_label();
+    uint64_t fpair;
+
+    QEMU_BUILD_BUG_ON(sizeof_field(CPUJumpCache, array[0]) != 16);
+    QEMU_BUILD_BUG_ON(offsetof(CPUJumpCache, array[0].pc) % 8 != 0);
+    /* One 64-bit load has to cover both, so they must be adjacent... */
+    QEMU_BUILD_BUG_ON(offsetof(TranslationBlock, cflags) !=
+                      offsetof(TranslationBlock, flags) + 4);
+    /* ...and aligned, which nothing else currently relies on. */
+    QEMU_BUILD_BUG_ON(offsetof(TranslationBlock, flags) % 8 != 0);
+
+    {
+        g_autoptr(TCGTemp) t = tcg_temp_new_ebb(TCG_TYPE_I64);
+        g_autoptr(TCGTemp) h = tcg_temp_new_ebb(TCG_TYPE_I64);
+        g_autoptr(TCGTemp) p = tcg_temp_new_ebb(TCG_TYPE_PTR);
+
+        /* p = &jc->array[tb_jmp_cache_hash_func(pc)] */
+        gen_jc_hash(h, t, pc);
+        gen_op_tti(INDEX_op_ld, TCG_TYPE_I64, t, tcgv_ptr_temp(tcg_env),
+                   offsetof(CPUState, tb_jmp_cache) - sizeof(CPUState));
+        gen_lea(TCG_TYPE_I64, p, t, h, 4, 0);
+
+        /*
+         * The pc first: on a hash miss it is the field most likely to differ,
+         * and an entry whose tb is NULL has a zero pc that only pc 0 matches.
+         */
+        gen_ld(TCG_TYPE_I64, t, p, offsetof(CPUJumpCache, array[0].pc));
+        gen_brcond(TCG_TYPE_I64, TCG_COND_NE, t, pc, slow);
+
+        gen_ld(TCG_TYPE_PTR, p, p, offsetof(CPUJumpCache, array[0].tb));
+        gen_brcond(TCG_TYPE_PTR, TCG_COND_EQ,
+                   p, tcg_constant_internal(TCG_TYPE_PTR, 0), slow);
+
+        /*
+         * flags and cflags are adjacent uint32_t, so one aligned 64-bit load
+         * and compare covers both.
+         */
+        fpair = (HOST_BIG_ENDIAN
+                 ? deposit64(tb->cflags, 32, 32, flags)
+                 : deposit64(flags, 32, 32, tb->cflags));
+        gen_ld(TCG_TYPE_I64, t, p, offsetof(TranslationBlock, flags));
+        gen_brcond(TCG_TYPE_I64, TCG_COND_NE,
+                   t, tcg_constant_internal(TCG_TYPE_I64, fpair), slow);
+
+        /*
+         * cs_base is a second word of target-specific flags despite the name,
+         * and the pc alone does not imply it on a target that uses it.
+         */
+        gen_ld(TCG_TYPE_I64, t, p, offsetof(TranslationBlock, cs_base));
+        gen_brcond(TCG_TYPE_I64, TCG_COND_NE, t, cs_base, slow);
+
+        gen_ld(TCG_TYPE_PTR, p, p, offsetof(TranslationBlock, tc.ptr));
+        gen_goto_ptr(p);
+    }
+
+    /*
+     * Emit a second goto_ptr rather than branching to a shared one: a temp
+     * live across the label would be spilled and reloaded on every dispatch.
+     */
+    gen_set_label(slow);
+    gen_lookup_tb_ptr_and_goto();
+}
+
 static void gen_goto_jc3(TCGType type, TCGTemp *pc,
                          TCGTemp *cs_base, unsigned flags)
 {
@@ -1988,7 +2085,16 @@ static void gen_goto_jc3(TCGType type, TCGTemp *pc,
                              tcg_constant_i32(flags));
 #endif
 
-    gen_lookup_tb_ptr_and_goto();
+    /*
+     * A breakpoint is the one thing the probe cannot check for itself, so
+     * while one is set the flag is set too and every dispatch takes the
+     * helper, which does check.  See tcg_update_cflags().
+     */
+    if (tcg_ctx->gen_tb->cflags & (CF_NO_GOTO_PTR | CF_NO_GOTO_JC)) {
+        gen_lookup_tb_ptr_and_goto();
+    } else {
+        gen_jc_probe(pc, cs_base, flags);
+    }
 }
 
 static void gen_goto_jc(TCGType type, TCGTemp *pc)
-- 
2.53.0


Reply via email to