Signed-off-by: Molly Chen <[email protected]>
---
 target/riscv/helper.h                       |  43 ++++++
 target/riscv/insn32.decode                  |  57 ++++++++
 target/riscv/tcg/insn_trans/trans_rvp.c.inc |  43 ++++++
 target/riscv/tcg/psimd_helper.c             | 149 ++++++++++++++++++++
 4 files changed, 292 insertions(+)

diff --git a/target/riscv/helper.h b/target/riscv/helper.h
index 1628ca119ac..d72323890f6 100644
--- a/target/riscv/helper.h
+++ b/target/riscv/helper.h
@@ -1455,3 +1455,46 @@ DEF_HELPER_3(pmaxu_w, i64, env, i64, i64)
 DEF_HELPER_3(mseq, i32, env, i32, i32)
 DEF_HELPER_3(mslt, i32, env, i32, i32)
 DEF_HELPER_3(msltu, i32, env, i32, i32)
+
+/* Packed SIMD - Shift Operations */
+DEF_HELPER_3(pslli_b, tl, env, tl, tl)
+DEF_HELPER_3(psll_bs, tl, env, tl, tl)
+DEF_HELPER_3(pslli_h, tl, env, tl, tl)
+DEF_HELPER_3(psll_hs, tl, env, tl, tl)
+DEF_HELPER_3(pslli_w, i64, env, i64, i64)
+DEF_HELPER_3(psll_ws, i64, env, i64, i64)
+DEF_HELPER_3(psrli_b, tl, env, tl, tl)
+DEF_HELPER_3(psrl_bs, tl, env, tl, tl)
+DEF_HELPER_3(psrli_h, tl, env, tl, tl)
+DEF_HELPER_3(psrl_hs, tl, env, tl, tl)
+DEF_HELPER_3(psrli_w, i64, env, i64, i64)
+DEF_HELPER_3(psrl_ws, i64, env, i64, i64)
+DEF_HELPER_3(psrai_b, tl, env, tl, tl)
+DEF_HELPER_3(psra_bs, tl, env, tl, tl)
+DEF_HELPER_3(psrai_h, tl, env, tl, tl)
+DEF_HELPER_3(psra_hs, tl, env, tl, tl)
+DEF_HELPER_3(psrai_w, i64, env, i64, i64)
+DEF_HELPER_3(psra_ws, i64, env, i64, i64)
+DEF_HELPER_3(psslai_h, tl, env, tl, tl)
+DEF_HELPER_3(psslai_w, i64, env, i64, i64)
+DEF_HELPER_3(sslai, i32, env, i32, i32)
+DEF_HELPER_3(psrari_h, tl, env, tl, tl)
+DEF_HELPER_3(psrari_w, i64, env, i64, i64)
+DEF_HELPER_3(srari_32, i32, env, i32, i32)
+DEF_HELPER_3(srari_64, i64, env, i64, i64)
+DEF_HELPER_3(pssha_hs, tl, env, tl, tl)
+DEF_HELPER_3(pssha_ws, i64, env, i64, i64)
+DEF_HELPER_3(psshar_hs, tl, env, tl, tl)
+DEF_HELPER_3(psshar_ws, i64, env, i64, i64)
+DEF_HELPER_3(ssha, i32, env, i32, i32)
+DEF_HELPER_3(sshar, i32, env, i32, i32)
+DEF_HELPER_3(sha, i64, env, i64, i64)
+DEF_HELPER_3(shar, i64, env, i64, i64)
+DEF_HELPER_3(psshl_hs, tl, env, tl, tl)
+DEF_HELPER_3(psshlr_hs, tl, env, tl, tl)
+DEF_HELPER_3(psshl_ws, i64, env, i64, i64)
+DEF_HELPER_3(psshlr_ws, i64, env, i64, i64)
+DEF_HELPER_3(sshl, i32, env, i32, i32)
+DEF_HELPER_3(sshlr, i32, env, i32, i32)
+DEF_HELPER_3(shl, i64, env, i64, i64)
+DEF_HELPER_3(shlr, i64, env, i64, i64)
diff --git a/target/riscv/insn32.decode b/target/riscv/insn32.decode
index 0c60f2eac7a..fb029bb512f 100644
--- a/target/riscv/insn32.decode
+++ b/target/riscv/insn32.decode
@@ -40,6 +40,7 @@
 %imm_z6   26:1 15:5
 %imm_mop5 30:1 26:2 20:2
 %imm_mop3 30:1 26:2
+%imm_p_ui8 20:3
 %imm_p_ui16 20:4
 %imm_p_ui32 20:5
 %imm_p_ui64 20:6
@@ -109,6 +110,7 @@
 @mop5 . . .. .. .... .. ..... ... ..... ....... &mop5 imm=%imm_mop5 %rd %rs1
 @mop3 . . .. .. . ..... ..... ... ..... ....... &mop3 imm=%imm_mop3 %rd %rs1 
%rs2
 
+@p_ui8  ..... .... ... ..... ... ..... ....... &i imm=%imm_p_ui8  %rs1 %rd
 @p_ui16 ..... .... ... ..... ... ..... ....... &i imm=%imm_p_ui16 %rs1 %rd
 @p_ui32 ..... .... ... ..... ... ..... ....... &i imm=%imm_p_ui32 %rs1 %rd
 @p_ui64 ..... .... ... ..... ... ..... ....... &i imm=%imm_p_ui64 %rs1 %rd
@@ -1217,3 +1219,58 @@ pmin_w     1110001 ..... ..... 110 ..... 0111011 @r
 pminu_w    1110101 ..... ..... 110 ..... 0111011 @r
 pmax_w     1111001 ..... ..... 110 ..... 0111011 @r
 pmaxu_w    1111101 ..... ..... 110 ..... 0111011 @r
+
+# Packed SIMD - Shift Operations
+pslli_b    10000 0001... ..... 010 ..... 0011011 @p_ui8
+psll_bs    1000110 ..... ..... 010 ..... 0011011 @r
+pslli_h    10000 001.... ..... 010 ..... 0011011 @p_ui16
+psll_hs    1000100 ..... ..... 010 ..... 0011011 @r
+pslli_w    10000 01..... ..... 010 ..... 0011011 @p_ui32
+psll_ws    1000101 ..... ..... 010 ..... 0011011 @r
+psrli_b    10000 0001... ..... 100 ..... 0011011 @p_ui8
+psrl_bs    1000110 ..... ..... 100 ..... 0011011 @r
+psrli_h    10000 001.... ..... 100 ..... 0011011 @p_ui16
+psrl_hs    1000100 ..... ..... 100 ..... 0011011 @r
+psrli_w    10000 01..... ..... 100 ..... 0011011 @p_ui32
+psrl_ws    1000101 ..... ..... 100 ..... 0011011 @r
+psrai_b    11000 0001... ..... 100 ..... 0011011 @p_ui8
+psra_bs    1100110 ..... ..... 100 ..... 0011011 @r
+psrai_h    11000 001.... ..... 100 ..... 0011011 @p_ui16
+psra_hs    1100100 ..... ..... 100 ..... 0011011 @r
+psrai_w    11000 01..... ..... 100 ..... 0011011 @p_ui32
+psra_ws    1100101 ..... ..... 100 ..... 0011011 @r
+psslai_h   11010 001.... ..... 010 ..... 0011011 @p_ui16
+{
+  sslai    11010 01..... ..... 010 ..... 0011011 @p_ui32
+  psslai_w 11010 01..... ..... 010 ..... 0011011 @p_ui32
+}
+psrari_h   11010 001.... ..... 100 ..... 0011011 @p_ui16
+{
+  srari_32 11010 01..... ..... 100 ..... 0011011 @p_ui32
+  psrari_w 11010 01..... ..... 100 ..... 0011011 @p_ui32
+}
+srari_64   110101 ...... ..... 100 ..... 0011011 @p_ui64
+pssha_hs   1110100 ..... ..... 010 ..... 0011011 @r
+{
+  ssha     1110101 ..... ..... 010 ..... 0011011 @r
+  pssha_ws 1110101 ..... ..... 010 ..... 0011011 @r
+}
+psshar_hs  1111100 ..... ..... 010 ..... 0011011 @r
+{
+  sshar    1111101 ..... ..... 010 ..... 0011011 @r
+  psshar_ws 1111101 ..... ..... 010 ..... 0011011 @r
+}
+sha        1110111 ..... ..... 010 ..... 0011011 @r
+shar       1111111 ..... ..... 010 ..... 0011011 @r
+psshl_hs   1010100 ..... ..... 010 ..... 0011011 @r
+psshlr_hs  1011100 ..... ..... 010 ..... 0011011 @r
+{
+  psshl_ws 1010101 ..... ..... 010 ..... 0011011 @r
+  sshl     1010101 ..... ..... 010 ..... 0011011 @r
+}
+{
+  psshlr_ws 1011101 ..... ..... 010 ..... 0011011 @r
+  sshlr     1011101 ..... ..... 010 ..... 0011011 @r
+}
+shl        1010111 ..... ..... 010 ..... 0011011 @r
+shlr       1011111 ..... ..... 010 ..... 0011011 @r
diff --git a/target/riscv/tcg/insn_trans/trans_rvp.c.inc 
b/target/riscv/tcg/insn_trans/trans_rvp.c.inc
index 2ad8e818ba5..fb82760a60c 100644
--- a/target/riscv/tcg/insn_trans/trans_rvp.c.inc
+++ b/target/riscv/tcg/insn_trans/trans_rvp.c.inc
@@ -601,3 +601,46 @@ GEN_SIMD_TRANS_32(mseq)
 GEN_SIMD_TRANS_32(mslt)
 GEN_SIMD_TRANS_32(msltu)
 
+/* Packed SIMD - Shift Operations */
+GEN_SIMD_TRANS_IMM(pslli_b)
+GEN_SIMD_TRANS(psll_bs)
+GEN_SIMD_TRANS_IMM(pslli_h)
+GEN_SIMD_TRANS(psll_hs)
+GEN_SIMD_TRANS_IMM_64(pslli_w)
+GEN_SIMD_TRANS_64(psll_ws)
+GEN_SIMD_TRANS_IMM(psrli_b)
+GEN_SIMD_TRANS(psrl_bs)
+GEN_SIMD_TRANS_IMM(psrli_h)
+GEN_SIMD_TRANS(psrl_hs)
+GEN_SIMD_TRANS_IMM_64(psrli_w)
+GEN_SIMD_TRANS_64(psrl_ws)
+GEN_SIMD_TRANS_IMM(psrai_b)
+GEN_SIMD_TRANS(psra_bs)
+GEN_SIMD_TRANS_IMM(psrai_h)
+GEN_SIMD_TRANS(psra_hs)
+GEN_SIMD_TRANS_IMM_64(psrai_w)
+GEN_SIMD_TRANS_64(psra_ws)
+GEN_SIMD_TRANS_IMM_VXSAT(psslai_h)
+GEN_SIMD_TRANS_IMM_64_VXSAT(psslai_w)
+GEN_SIMD_TRANS_IMM_32_VXSAT(sslai)
+GEN_SIMD_TRANS_IMM(psrari_h)
+GEN_SIMD_TRANS_IMM_64(psrari_w)
+GEN_SIMD_TRANS_IMM_32(srari_32)
+GEN_SIMD_TRANS_IMM_64(srari_64)
+GEN_SIMD_TRANS_VXSAT(pssha_hs)
+GEN_SIMD_TRANS_64_VXSAT(pssha_ws)
+GEN_SIMD_TRANS_VXSAT(psshar_hs)
+GEN_SIMD_TRANS_64_VXSAT(psshar_ws)
+GEN_SIMD_TRANS_32_VXSAT(ssha)
+GEN_SIMD_TRANS_32_VXSAT(sshar)
+GEN_SIMD_TRANS_64(sha)
+GEN_SIMD_TRANS_64(shar)
+GEN_SIMD_TRANS_VXSAT(psshl_hs)
+GEN_SIMD_TRANS_VXSAT(psshlr_hs)
+GEN_SIMD_TRANS_64_VXSAT(psshl_ws)
+GEN_SIMD_TRANS_64_VXSAT(psshlr_ws)
+GEN_SIMD_TRANS_32_VXSAT(sshl)
+GEN_SIMD_TRANS_32_VXSAT(sshlr)
+GEN_SIMD_TRANS_64(shl)
+GEN_SIMD_TRANS_64(shlr)
+
diff --git a/target/riscv/tcg/psimd_helper.c b/target/riscv/tcg/psimd_helper.c
index 034afe5a054..8b106a336f5 100644
--- a/target/riscv/tcg/psimd_helper.c
+++ b/target/riscv/tcg/psimd_helper.c
@@ -1903,3 +1903,152 @@ GEN_PSIMD_BINOP(mslt, uint32_t, int32_t, uint32_t,
 GEN_PSIMD_BINOP(msltu, uint32_t, uint32_t, uint32_t,
                 EXTRACT32, INSERT32, ELEMS_W, PSIMD_DO_LT_MASK)
 
+/* Shift operations (immediate and register) */
+
+GEN_PSIMD_SHIFTOP(pslli_b, target_ulong, uint8_t, uint8_t,
+                  EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(psll_bs, target_ulong, uint8_t, uint8_t,
+                  EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(pslli_h, target_ulong, uint16_t, uint16_t,
+                  EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(psll_hs, target_ulong, uint16_t, uint16_t,
+                  EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(pslli_w, uint64_t, uint32_t, uint32_t,
+                  EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(psll_ws, uint64_t, uint32_t, uint32_t,
+                  EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SLL)
+
+GEN_PSIMD_SHIFTOP(psrli_b, target_ulong, uint8_t, uint8_t,
+                  EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrl_bs, target_ulong, uint8_t, uint8_t,
+                  EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrli_h, target_ulong, uint16_t, uint16_t,
+                  EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrl_hs, target_ulong, uint16_t, uint16_t,
+                  EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrli_w, uint64_t, uint32_t, uint32_t,
+                  EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrl_ws, uint64_t, uint32_t, uint32_t,
+                  EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SRL)
+
+GEN_PSIMD_SHIFTOP(psrai_b, target_ulong, int8_t, uint8_t,
+                  EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psra_bs, target_ulong, int8_t, uint8_t,
+                  EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psrai_h, target_ulong, int16_t, uint16_t,
+                  EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psra_hs, target_ulong, int16_t, uint16_t,
+                  EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psrai_w, uint64_t, int32_t, uint32_t,
+                  EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psra_ws, uint64_t, int32_t, uint32_t,
+                  EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SRA)
+
+/* Saturating shift operations */
+
+GEN_PSIMD_SAT_SHIFTOP(psslai_h, target_ulong, int16_t, int32_t,
+                      EXTRACT16, INSERT16, ELEMS_H, 0x0f,
+                      signed_saturate_h)
+GEN_PSIMD_SAT_SHIFTOP(psslai_w, uint64_t, int32_t, int64_t,
+                      EXTRACT32, INSERT32, ELEMS_W, 0x1f,
+                      signed_saturate_w)
+
+GEN_PSIMD_SAT_SHIFTOP(sslai, uint32_t, int32_t, int64_t,
+                      EXTRACT32, INSERT32, ELEMS_W, 0x1f,
+                      signed_saturate_w)
+
+/* Rounding shift operations */
+
+GEN_PSIMD_ROUND_SRAI(psrari_h, target_ulong, int16_t, int32_t,
+                     EXTRACT16, INSERT16, ELEMS_H, 0x0f)
+GEN_PSIMD_ROUND_SRAI(psrari_w, uint64_t, int32_t, int64_t,
+                     EXTRACT32, INSERT32, ELEMS_W, 0x1f)
+
+GEN_PSIMD_ROUND_SRAI(srari_32, uint32_t, int32_t, int64_t,
+                     EXTRACT32, INSERT32, ELEMS_W, 0x1f)
+
+/**
+ * SRARI_64 - 64-bit scalar arithmetic shift right with rounding
+ */
+uint64_t HELPER(srari_64)(CPURISCVState *env, uint64_t rs1, uint64_t imm)
+{
+    int64_t a = (int64_t)rs1;
+    uint8_t shamt = imm & 0x3F;
+
+    if (shamt == 0) {
+        return rs1;
+    }
+
+    return (uint64_t)(((a >> (shamt - 1)) + 1) >> 1);
+}
+
+/* Variable shift operations (with saturation and rounding) */
+
+GEN_PSIMD_VAR_SSHA(pssha_hs, target_ulong, int16_t, int32_t,
+                   EXTRACT16, INSERT16, ELEMS_H, 16, signed_saturate_h)
+GEN_PSIMD_VAR_SSHA(pssha_ws, uint64_t, int32_t, int64_t,
+                   EXTRACT32, INSERT32, ELEMS_W, 32, signed_saturate_w)
+
+GEN_PSIMD_VAR_SSHAR(psshar_hs, target_ulong, int16_t, int32_t,
+                    EXTRACT16, INSERT16, ELEMS_H, 16, SAT_MIN_H, SAT_MAX_H,
+                    signed_saturate_h)
+GEN_PSIMD_VAR_SSHAR(psshar_ws, uint64_t, int32_t, int64_t,
+                    EXTRACT32, INSERT32, ELEMS_W, 32, SAT_MIN_W, SAT_MAX_W,
+                    signed_saturate_w)
+
+GEN_PSIMD_VAR_SSHA(ssha, uint32_t, int32_t, int64_t,
+                   EXTRACT32, INSERT32, ELEMS_W, 32, signed_saturate_w)
+
+/**
+ * SSHAR - 32-bit scalar variable shift with rounding and saturation
+ */
+uint32_t HELPER(sshar)(CPURISCVState *env, uint32_t rs1, uint32_t rs2)
+{
+    int32_t a = (int32_t)rs1;
+    int8_t shamt = (int8_t)(rs2 & 0xFF);
+    int sat = 0;
+    int32_t res;
+
+    if (shamt >= 0) {
+        int64_t shifted = (int64_t)a << shamt;
+        res = signed_saturate_w(shifted, &sat);
+    } else {
+        int right = -shamt;
+        if (right >= 32) {
+            res = (a < 0) ? -1 : 0;
+        } else {
+            int64_t rounded = ((a >> (right - 1)) + 1) >> 1;
+            res = (int32_t)rounded;
+        }
+    }
+
+    if (sat) {
+        env->vxsat = 1;
+    }
+    return (uint32_t)res;
+}
+
+GEN_PSIMD_VAR_SRA64(sha, PSIMD_DO_SRA64)
+GEN_PSIMD_VAR_SRA64(shar, PSIMD_DO_RNDSRA64)
+
+GEN_PSIMD_VAR_USHL(psshl_hs, target_ulong, uint16_t, uint32_t,
+                   EXTRACT16, INSERT16, ELEMS_H, 16, unsigned_saturate_h)
+
+GEN_PSIMD_VAR_USHLR(psshlr_hs, target_ulong, uint16_t, uint32_t,
+                    EXTRACT16, INSERT16, ELEMS_H, 16, unsigned_saturate_h)
+
+GEN_PSIMD_VAR_USHL(psshl_ws, uint64_t, uint32_t, uint64_t,
+                   EXTRACT32, INSERT32, ELEMS_W, 32, unsigned_saturate_w)
+
+GEN_PSIMD_VAR_USHLR(psshlr_ws, uint64_t, uint32_t, uint64_t,
+                    EXTRACT32, INSERT32, ELEMS_W, 32, unsigned_saturate_w)
+
+GEN_PSIMD_VAR_USHL(sshl, uint32_t, uint32_t, uint64_t,
+                   EXTRACT32, INSERT32, ELEMS_W, 32, unsigned_saturate_w)
+
+GEN_PSIMD_VAR_USHLR(sshlr, uint32_t, uint32_t, uint64_t,
+                    EXTRACT32, INSERT32, ELEMS_W, 32, unsigned_saturate_w)
+
+GEN_PSIMD_VAR_SRL64(shl, PSIMD_DO_SRL64)
+GEN_PSIMD_VAR_SRL64(shlr, PSIMD_DO_RNDSRL64)
+
-- 
2.34.1


Reply via email to