Use ldp to load multiple values at once.
Use ccmp to reduce the number of branhes.

Signed-off-by: Richard Henderson <[email protected]>
---
 tcg/aarch64/tcg-target-con-set.h |  1 +
 tcg/aarch64/tcg-target.c.inc     | 58 +++++++++++++++++++++++++++++++-
 2 files changed, 58 insertions(+), 1 deletion(-)

diff --git a/tcg/aarch64/tcg-target-con-set.h b/tcg/aarch64/tcg-target-con-set.h
index dd84a9af586..4b076962347 100644
--- a/tcg/aarch64/tcg-target-con-set.h
+++ b/tcg/aarch64/tcg-target-con-set.h
@@ -14,6 +14,7 @@ C_O0_I2(r, rC)
 C_O0_I2(rz, r)
 C_O0_I2(w, r)
 C_O0_I3(rz, rz, r)
+C_O0_I4(r, rz, rz, rz)
 C_O1_I1(r, r)
 C_O1_I1(w, r)
 C_O1_I1(w, w)
diff --git a/tcg/aarch64/tcg-target.c.inc b/tcg/aarch64/tcg-target.c.inc
index a0f23914224..2994a3f73e7 100644
--- a/tcg/aarch64/tcg-target.c.inc
+++ b/tcg/aarch64/tcg-target.c.inc
@@ -570,6 +570,9 @@ typedef enum {
     /* Logical shifted register instructions (with a shift).  */
     Iaddsub_realshift_AND_LSR  = Ilogic_shift_AND | (1 << 22),
 
+    /* Conditional compare register. */
+    Iccmp_reg_CCMP     = 0x7a400000,
+
     /* AdvSIMD copy */
     Isimd_copy_DUP     = 0x0e000400,
     Isimd_copy_INS     = 0x4e001c00,
@@ -712,6 +715,14 @@ static void tcg_out_insn_bcond_imm(TCGContext *s, 
AArch64Insn insn,
     tcg_out32(s, insn | tcg_cond_to_aarch64[c] | (imm19 & 0x7ffff) << 5);
 }
 
+static void tcg_out_insn_ccmp_reg(TCGContext *s, AArch64Insn insn,
+                                  TCGType ext, TCGCond c, TCGReg rn,
+                                  TCGReg rm, unsigned nzcv)
+{
+    tcg_out32(s, insn | ext << 31 | tcg_cond_to_aarch64[c] << 12
+              | rm << 16 | rn << 5 | nzcv);
+}
+
 static void tcg_out_insn_tbz(TCGContext *s, AArch64Insn insn,
                               TCGReg rt, int imm6, int imm14)
 {
@@ -2008,8 +2019,53 @@ static const TCGOutOpQemuLdSt2 outop_qemu_st2 = {
     .out = tgen_qemu_st2,
 };
 
+static void tgen_goto_jc(TCGContext *s, TCGReg ptr, TCGReg pc,
+                         TCGArg cs, bool const_cs,
+                         TCGArg fl, bool const_fl, TCGLabel *label)
+{
+    /* ptr is CPUJumpCache */
+
+    QEMU_BUILD_BUG_ON(offsetof(CPUJumpCache, array[0].pc) !=
+                      offsetof(CPUJumpCache, array[0].tb) + 8);
+    tcg_out_insn(s, ldstpair, LDP, ptr, TCG_REG_TMP0, ptr,
+                 offsetof(CPUJumpCache, array[0].tb), 1, 0);
+
+    /*
+     * z=cmp(ptr,0)
+     * if !z, z=cmp(pc,tmp), else z=0
+     * if !z, goto label
+     */
+    tgen_cmpi(s, TCG_TYPE_I64, TCG_COND_NE, ptr, 0);
+    tcg_out_insn(s, ccmp_reg, CCMP, TCG_TYPE_I64, TCG_COND_NE,
+                 pc, TCG_REG_TMP0, 0b0000);
+    tcg_out_reloc(s, s->code_ptr, R_AARCH64_CONDBR19, label, 0);
+    tcg_out_insn(s, bcond_imm, B_C, TCG_COND_NE, 0);
+
+    /* ptr is now TranslationBlock */
+
+    QEMU_BUILD_BUG_ON(offsetof(TranslationBlock, flags) !=
+                      offsetof(TranslationBlock, cs_base) + 8);
+    tcg_out_insn(s, ldstpair, LDP, TCG_REG_TMP0, TCG_REG_TMP1,
+                 ptr, offsetof(TranslationBlock, cs_base), 1, 0);
+    tcg_out_ld(s, TCG_TYPE_PTR, ptr, ptr, offsetof(TranslationBlock, tc.ptr));
+
+    /*
+     * z=cmp(cs,tmp0)
+     * if z, z=cmp(fl,tmp1), else z=0
+     * if !z, goto label
+     */
+    tgen_cmp(s, TCG_TYPE_I64, TCG_COND_EQ, cs, TCG_REG_TMP0);
+    tcg_out_insn(s, ccmp_reg, CCMP, TCG_TYPE_I64, TCG_COND_EQ,
+                 fl, TCG_REG_TMP1, 0x0000);
+    tcg_out_reloc(s, s->code_ptr, R_AARCH64_CONDBR19, label, 0);
+    tcg_out_insn(s, bcond_imm, B_C, TCG_COND_NE, 0);
+
+    tcg_out_goto_ptr(s, ptr);
+}
+
 static const TCGOutOpGotoJC outop_goto_jc = {
-    .base.static_constraint = C_NotImplemented,
+    .base.static_constraint = C_O0_I4(r, rz, rz, rz),
+    .out = tgen_goto_jc,
 };
 
 static const tcg_insn_unit *tb_ret_addr;
-- 
2.53.0


Reply via email to