Use LGM to load multiple values at once.

Signed-off-by: Richard Henderson <[email protected]>
---
 tcg/s390x/tcg-target-con-set.h |  1 +
 tcg/s390x/tcg-target-con-str.h |  1 +
 tcg/s390x/tcg-target.c.inc     | 40 +++++++++++++++++++++++++++++++++-
 3 files changed, 41 insertions(+), 1 deletion(-)

diff --git a/tcg/s390x/tcg-target-con-set.h b/tcg/s390x/tcg-target-con-set.h
index f67fd7898ed..a566a1ca7b4 100644
--- a/tcg/s390x/tcg-target-con-set.h
+++ b/tcg/s390x/tcg-target-con-set.h
@@ -18,6 +18,7 @@ C_O0_I2(r, ri)
 C_O0_I2(r, rC)
 C_O0_I2(v, r)
 C_O0_I3(o, m, r)
+C_O0_I4(r, hJU, hJU, hJU)
 C_O1_I1(r, r)
 C_O1_I1(v, r)
 C_O1_I1(v, v)
diff --git a/tcg/s390x/tcg-target-con-str.h b/tcg/s390x/tcg-target-con-str.h
index 636a38a168a..e44b2a92d52 100644
--- a/tcg/s390x/tcg-target-con-str.h
+++ b/tcg/s390x/tcg-target-con-str.h
@@ -9,6 +9,7 @@
  * REGS(letter, register_mask)
  */
 REGS('r', ALL_GENERAL_REGS)
+REGS('h', 0xfff8) /* general regs other than r0-r2 */
 REGS('v', ALL_VECTOR_REGS)
 REGS('o', 0xaaaa) /* odd numbered general regs */
 
diff --git a/tcg/s390x/tcg-target.c.inc b/tcg/s390x/tcg-target.c.inc
index 3b63383315d..b04c54e26d3 100644
--- a/tcg/s390x/tcg-target.c.inc
+++ b/tcg/s390x/tcg-target.c.inc
@@ -3202,8 +3202,46 @@ static const TCGOutOpLea outop_lea = {
     .out = tgen_lea,
 };
 
+static void tgen_goto_jc(TCGContext *s, TCGReg ptr, TCGReg pc,
+                         TCGArg cs, bool const_cs,
+                         TCGArg fl, bool const_fl, TCGLabel *label)
+{
+    /*
+     * Note that pc, cs, fl are constrained to r3 or higher, r0 and r1
+     * are always reserved as scratch, this is the end of the translation
+     * block so r2 may now be clobbered, and the input ptr * is dead
+     * after the initial LMG.
+     */
+
+    /* ptr is CPUJumpCache */
+
+    QEMU_BUILD_BUG_ON(offsetof(CPUJumpCache, array[0].pc) !=
+                      offsetof(CPUJumpCache, array[0].tb) + 8);
+    /* lmg r1,r2,tb(ptr) */
+    tcg_out_insn(s, RXY, LMG, TCG_REG_R1, ptr, TCG_REG_R2,
+                 offsetof(CPUJumpCache, array[0].tb));
+
+    tgen_brcond(s, TCG_TYPE_I64, TCG_COND_NE, TCG_REG_R2, pc, false, label);
+    tgen_brcond(s, TCG_TYPE_PTR, TCG_COND_EQ, TCG_REG_R1, 0, true, label);
+
+    /* R1 is now TranslationBlock */
+
+    QEMU_BUILD_BUG_ON(offsetof(TranslationBlock, flags) !=
+                      offsetof(TranslationBlock, cs_base) + 8);
+    QEMU_BUILD_BUG_ON(offsetof(TranslationBlock, tc.ptr) !=
+                      offsetof(TranslationBlock, cs_base) + 16);
+    /* lmg r0,r2,cs(r1) */
+    tcg_out_insn(s, RXY, LMG, TCG_REG_R0, TCG_REG_R1, TCG_REG_R2,
+                 offsetof(TranslationBlock, cs_base));
+
+    tgen_brcond(s, TCG_TYPE_I64, TCG_COND_NE, TCG_REG_R0, cs, const_cs, label);
+    tgen_brcond(s, TCG_TYPE_I64, TCG_COND_NE, TCG_REG_R1, fl, const_fl, label);
+    tcg_out_goto_ptr(s, TCG_REG_R2);
+}
+
 static const TCGOutOpGotoJC outop_goto_jc = {
-    .base.static_constraint = C_NotImplemented,
+    .base.static_constraint = C_O0_I4(r, hJU, hJU, hJU),
+    .out = tgen_goto_jc,
 };
 
 static bool tcg_out_dup_vec(TCGContext *s, TCGType type, unsigned vece,
-- 
2.53.0


Reply via email to