Signed-off-by: Molly Chen <[email protected]>
---
target/riscv/helper.h | 43 ++++++
target/riscv/insn32.decode | 57 ++++++++
target/riscv/tcg/insn_trans/trans_rvp.c.inc | 43 ++++++
target/riscv/tcg/psimd_helper.c | 149 ++++++++++++++++++++
4 files changed, 292 insertions(+)
diff --git a/target/riscv/helper.h b/target/riscv/helper.h
index 1628ca119ac..d72323890f6 100644
--- a/target/riscv/helper.h
+++ b/target/riscv/helper.h
@@ -1455,3 +1455,46 @@ DEF_HELPER_3(pmaxu_w, i64, env, i64, i64)
DEF_HELPER_3(mseq, i32, env, i32, i32)
DEF_HELPER_3(mslt, i32, env, i32, i32)
DEF_HELPER_3(msltu, i32, env, i32, i32)
+
+/* Packed SIMD - Shift Operations */
+DEF_HELPER_3(pslli_b, tl, env, tl, tl)
+DEF_HELPER_3(psll_bs, tl, env, tl, tl)
+DEF_HELPER_3(pslli_h, tl, env, tl, tl)
+DEF_HELPER_3(psll_hs, tl, env, tl, tl)
+DEF_HELPER_3(pslli_w, i64, env, i64, i64)
+DEF_HELPER_3(psll_ws, i64, env, i64, i64)
+DEF_HELPER_3(psrli_b, tl, env, tl, tl)
+DEF_HELPER_3(psrl_bs, tl, env, tl, tl)
+DEF_HELPER_3(psrli_h, tl, env, tl, tl)
+DEF_HELPER_3(psrl_hs, tl, env, tl, tl)
+DEF_HELPER_3(psrli_w, i64, env, i64, i64)
+DEF_HELPER_3(psrl_ws, i64, env, i64, i64)
+DEF_HELPER_3(psrai_b, tl, env, tl, tl)
+DEF_HELPER_3(psra_bs, tl, env, tl, tl)
+DEF_HELPER_3(psrai_h, tl, env, tl, tl)
+DEF_HELPER_3(psra_hs, tl, env, tl, tl)
+DEF_HELPER_3(psrai_w, i64, env, i64, i64)
+DEF_HELPER_3(psra_ws, i64, env, i64, i64)
+DEF_HELPER_3(psslai_h, tl, env, tl, tl)
+DEF_HELPER_3(psslai_w, i64, env, i64, i64)
+DEF_HELPER_3(sslai, i32, env, i32, i32)
+DEF_HELPER_3(psrari_h, tl, env, tl, tl)
+DEF_HELPER_3(psrari_w, i64, env, i64, i64)
+DEF_HELPER_3(srari_32, i32, env, i32, i32)
+DEF_HELPER_3(srari_64, i64, env, i64, i64)
+DEF_HELPER_3(pssha_hs, tl, env, tl, tl)
+DEF_HELPER_3(pssha_ws, i64, env, i64, i64)
+DEF_HELPER_3(psshar_hs, tl, env, tl, tl)
+DEF_HELPER_3(psshar_ws, i64, env, i64, i64)
+DEF_HELPER_3(ssha, i32, env, i32, i32)
+DEF_HELPER_3(sshar, i32, env, i32, i32)
+DEF_HELPER_3(sha, i64, env, i64, i64)
+DEF_HELPER_3(shar, i64, env, i64, i64)
+DEF_HELPER_3(psshl_hs, tl, env, tl, tl)
+DEF_HELPER_3(psshlr_hs, tl, env, tl, tl)
+DEF_HELPER_3(psshl_ws, i64, env, i64, i64)
+DEF_HELPER_3(psshlr_ws, i64, env, i64, i64)
+DEF_HELPER_3(sshl, i32, env, i32, i32)
+DEF_HELPER_3(sshlr, i32, env, i32, i32)
+DEF_HELPER_3(shl, i64, env, i64, i64)
+DEF_HELPER_3(shlr, i64, env, i64, i64)
diff --git a/target/riscv/insn32.decode b/target/riscv/insn32.decode
index 0c60f2eac7a..fb029bb512f 100644
--- a/target/riscv/insn32.decode
+++ b/target/riscv/insn32.decode
@@ -40,6 +40,7 @@
%imm_z6 26:1 15:5
%imm_mop5 30:1 26:2 20:2
%imm_mop3 30:1 26:2
+%imm_p_ui8 20:3
%imm_p_ui16 20:4
%imm_p_ui32 20:5
%imm_p_ui64 20:6
@@ -109,6 +110,7 @@
@mop5 . . .. .. .... .. ..... ... ..... ....... &mop5 imm=%imm_mop5 %rd %rs1
@mop3 . . .. .. . ..... ..... ... ..... ....... &mop3 imm=%imm_mop3 %rd %rs1
%rs2
+@p_ui8 ..... .... ... ..... ... ..... ....... &i imm=%imm_p_ui8 %rs1 %rd
@p_ui16 ..... .... ... ..... ... ..... ....... &i imm=%imm_p_ui16 %rs1 %rd
@p_ui32 ..... .... ... ..... ... ..... ....... &i imm=%imm_p_ui32 %rs1 %rd
@p_ui64 ..... .... ... ..... ... ..... ....... &i imm=%imm_p_ui64 %rs1 %rd
@@ -1217,3 +1219,58 @@ pmin_w 1110001 ..... ..... 110 ..... 0111011 @r
pminu_w 1110101 ..... ..... 110 ..... 0111011 @r
pmax_w 1111001 ..... ..... 110 ..... 0111011 @r
pmaxu_w 1111101 ..... ..... 110 ..... 0111011 @r
+
+# Packed SIMD - Shift Operations
+pslli_b 10000 0001... ..... 010 ..... 0011011 @p_ui8
+psll_bs 1000110 ..... ..... 010 ..... 0011011 @r
+pslli_h 10000 001.... ..... 010 ..... 0011011 @p_ui16
+psll_hs 1000100 ..... ..... 010 ..... 0011011 @r
+pslli_w 10000 01..... ..... 010 ..... 0011011 @p_ui32
+psll_ws 1000101 ..... ..... 010 ..... 0011011 @r
+psrli_b 10000 0001... ..... 100 ..... 0011011 @p_ui8
+psrl_bs 1000110 ..... ..... 100 ..... 0011011 @r
+psrli_h 10000 001.... ..... 100 ..... 0011011 @p_ui16
+psrl_hs 1000100 ..... ..... 100 ..... 0011011 @r
+psrli_w 10000 01..... ..... 100 ..... 0011011 @p_ui32
+psrl_ws 1000101 ..... ..... 100 ..... 0011011 @r
+psrai_b 11000 0001... ..... 100 ..... 0011011 @p_ui8
+psra_bs 1100110 ..... ..... 100 ..... 0011011 @r
+psrai_h 11000 001.... ..... 100 ..... 0011011 @p_ui16
+psra_hs 1100100 ..... ..... 100 ..... 0011011 @r
+psrai_w 11000 01..... ..... 100 ..... 0011011 @p_ui32
+psra_ws 1100101 ..... ..... 100 ..... 0011011 @r
+psslai_h 11010 001.... ..... 010 ..... 0011011 @p_ui16
+{
+ sslai 11010 01..... ..... 010 ..... 0011011 @p_ui32
+ psslai_w 11010 01..... ..... 010 ..... 0011011 @p_ui32
+}
+psrari_h 11010 001.... ..... 100 ..... 0011011 @p_ui16
+{
+ srari_32 11010 01..... ..... 100 ..... 0011011 @p_ui32
+ psrari_w 11010 01..... ..... 100 ..... 0011011 @p_ui32
+}
+srari_64 110101 ...... ..... 100 ..... 0011011 @p_ui64
+pssha_hs 1110100 ..... ..... 010 ..... 0011011 @r
+{
+ ssha 1110101 ..... ..... 010 ..... 0011011 @r
+ pssha_ws 1110101 ..... ..... 010 ..... 0011011 @r
+}
+psshar_hs 1111100 ..... ..... 010 ..... 0011011 @r
+{
+ sshar 1111101 ..... ..... 010 ..... 0011011 @r
+ psshar_ws 1111101 ..... ..... 010 ..... 0011011 @r
+}
+sha 1110111 ..... ..... 010 ..... 0011011 @r
+shar 1111111 ..... ..... 010 ..... 0011011 @r
+psshl_hs 1010100 ..... ..... 010 ..... 0011011 @r
+psshlr_hs 1011100 ..... ..... 010 ..... 0011011 @r
+{
+ psshl_ws 1010101 ..... ..... 010 ..... 0011011 @r
+ sshl 1010101 ..... ..... 010 ..... 0011011 @r
+}
+{
+ psshlr_ws 1011101 ..... ..... 010 ..... 0011011 @r
+ sshlr 1011101 ..... ..... 010 ..... 0011011 @r
+}
+shl 1010111 ..... ..... 010 ..... 0011011 @r
+shlr 1011111 ..... ..... 010 ..... 0011011 @r
diff --git a/target/riscv/tcg/insn_trans/trans_rvp.c.inc
b/target/riscv/tcg/insn_trans/trans_rvp.c.inc
index 2ad8e818ba5..fb82760a60c 100644
--- a/target/riscv/tcg/insn_trans/trans_rvp.c.inc
+++ b/target/riscv/tcg/insn_trans/trans_rvp.c.inc
@@ -601,3 +601,46 @@ GEN_SIMD_TRANS_32(mseq)
GEN_SIMD_TRANS_32(mslt)
GEN_SIMD_TRANS_32(msltu)
+/* Packed SIMD - Shift Operations */
+GEN_SIMD_TRANS_IMM(pslli_b)
+GEN_SIMD_TRANS(psll_bs)
+GEN_SIMD_TRANS_IMM(pslli_h)
+GEN_SIMD_TRANS(psll_hs)
+GEN_SIMD_TRANS_IMM_64(pslli_w)
+GEN_SIMD_TRANS_64(psll_ws)
+GEN_SIMD_TRANS_IMM(psrli_b)
+GEN_SIMD_TRANS(psrl_bs)
+GEN_SIMD_TRANS_IMM(psrli_h)
+GEN_SIMD_TRANS(psrl_hs)
+GEN_SIMD_TRANS_IMM_64(psrli_w)
+GEN_SIMD_TRANS_64(psrl_ws)
+GEN_SIMD_TRANS_IMM(psrai_b)
+GEN_SIMD_TRANS(psra_bs)
+GEN_SIMD_TRANS_IMM(psrai_h)
+GEN_SIMD_TRANS(psra_hs)
+GEN_SIMD_TRANS_IMM_64(psrai_w)
+GEN_SIMD_TRANS_64(psra_ws)
+GEN_SIMD_TRANS_IMM_VXSAT(psslai_h)
+GEN_SIMD_TRANS_IMM_64_VXSAT(psslai_w)
+GEN_SIMD_TRANS_IMM_32_VXSAT(sslai)
+GEN_SIMD_TRANS_IMM(psrari_h)
+GEN_SIMD_TRANS_IMM_64(psrari_w)
+GEN_SIMD_TRANS_IMM_32(srari_32)
+GEN_SIMD_TRANS_IMM_64(srari_64)
+GEN_SIMD_TRANS_VXSAT(pssha_hs)
+GEN_SIMD_TRANS_64_VXSAT(pssha_ws)
+GEN_SIMD_TRANS_VXSAT(psshar_hs)
+GEN_SIMD_TRANS_64_VXSAT(psshar_ws)
+GEN_SIMD_TRANS_32_VXSAT(ssha)
+GEN_SIMD_TRANS_32_VXSAT(sshar)
+GEN_SIMD_TRANS_64(sha)
+GEN_SIMD_TRANS_64(shar)
+GEN_SIMD_TRANS_VXSAT(psshl_hs)
+GEN_SIMD_TRANS_VXSAT(psshlr_hs)
+GEN_SIMD_TRANS_64_VXSAT(psshl_ws)
+GEN_SIMD_TRANS_64_VXSAT(psshlr_ws)
+GEN_SIMD_TRANS_32_VXSAT(sshl)
+GEN_SIMD_TRANS_32_VXSAT(sshlr)
+GEN_SIMD_TRANS_64(shl)
+GEN_SIMD_TRANS_64(shlr)
+
diff --git a/target/riscv/tcg/psimd_helper.c b/target/riscv/tcg/psimd_helper.c
index 034afe5a054..8b106a336f5 100644
--- a/target/riscv/tcg/psimd_helper.c
+++ b/target/riscv/tcg/psimd_helper.c
@@ -1903,3 +1903,152 @@ GEN_PSIMD_BINOP(mslt, uint32_t, int32_t, uint32_t,
GEN_PSIMD_BINOP(msltu, uint32_t, uint32_t, uint32_t,
EXTRACT32, INSERT32, ELEMS_W, PSIMD_DO_LT_MASK)
+/* Shift operations (immediate and register) */
+
+GEN_PSIMD_SHIFTOP(pslli_b, target_ulong, uint8_t, uint8_t,
+ EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(psll_bs, target_ulong, uint8_t, uint8_t,
+ EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(pslli_h, target_ulong, uint16_t, uint16_t,
+ EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(psll_hs, target_ulong, uint16_t, uint16_t,
+ EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(pslli_w, uint64_t, uint32_t, uint32_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SLL)
+GEN_PSIMD_SHIFTOP(psll_ws, uint64_t, uint32_t, uint32_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SLL)
+
+GEN_PSIMD_SHIFTOP(psrli_b, target_ulong, uint8_t, uint8_t,
+ EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrl_bs, target_ulong, uint8_t, uint8_t,
+ EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrli_h, target_ulong, uint16_t, uint16_t,
+ EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrl_hs, target_ulong, uint16_t, uint16_t,
+ EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrli_w, uint64_t, uint32_t, uint32_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SRL)
+GEN_PSIMD_SHIFTOP(psrl_ws, uint64_t, uint32_t, uint32_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SRL)
+
+GEN_PSIMD_SHIFTOP(psrai_b, target_ulong, int8_t, uint8_t,
+ EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psra_bs, target_ulong, int8_t, uint8_t,
+ EXTRACT8, INSERT8, ELEMS_B, 0x07, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psrai_h, target_ulong, int16_t, uint16_t,
+ EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psra_hs, target_ulong, int16_t, uint16_t,
+ EXTRACT16, INSERT16, ELEMS_H, 0x0f, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psrai_w, uint64_t, int32_t, uint32_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SRA)
+GEN_PSIMD_SHIFTOP(psra_ws, uint64_t, int32_t, uint32_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f, PSIMD_DO_SRA)
+
+/* Saturating shift operations */
+
+GEN_PSIMD_SAT_SHIFTOP(psslai_h, target_ulong, int16_t, int32_t,
+ EXTRACT16, INSERT16, ELEMS_H, 0x0f,
+ signed_saturate_h)
+GEN_PSIMD_SAT_SHIFTOP(psslai_w, uint64_t, int32_t, int64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f,
+ signed_saturate_w)
+
+GEN_PSIMD_SAT_SHIFTOP(sslai, uint32_t, int32_t, int64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f,
+ signed_saturate_w)
+
+/* Rounding shift operations */
+
+GEN_PSIMD_ROUND_SRAI(psrari_h, target_ulong, int16_t, int32_t,
+ EXTRACT16, INSERT16, ELEMS_H, 0x0f)
+GEN_PSIMD_ROUND_SRAI(psrari_w, uint64_t, int32_t, int64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f)
+
+GEN_PSIMD_ROUND_SRAI(srari_32, uint32_t, int32_t, int64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 0x1f)
+
+/**
+ * SRARI_64 - 64-bit scalar arithmetic shift right with rounding
+ */
+uint64_t HELPER(srari_64)(CPURISCVState *env, uint64_t rs1, uint64_t imm)
+{
+ int64_t a = (int64_t)rs1;
+ uint8_t shamt = imm & 0x3F;
+
+ if (shamt == 0) {
+ return rs1;
+ }
+
+ return (uint64_t)(((a >> (shamt - 1)) + 1) >> 1);
+}
+
+/* Variable shift operations (with saturation and rounding) */
+
+GEN_PSIMD_VAR_SSHA(pssha_hs, target_ulong, int16_t, int32_t,
+ EXTRACT16, INSERT16, ELEMS_H, 16, signed_saturate_h)
+GEN_PSIMD_VAR_SSHA(pssha_ws, uint64_t, int32_t, int64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 32, signed_saturate_w)
+
+GEN_PSIMD_VAR_SSHAR(psshar_hs, target_ulong, int16_t, int32_t,
+ EXTRACT16, INSERT16, ELEMS_H, 16, SAT_MIN_H, SAT_MAX_H,
+ signed_saturate_h)
+GEN_PSIMD_VAR_SSHAR(psshar_ws, uint64_t, int32_t, int64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 32, SAT_MIN_W, SAT_MAX_W,
+ signed_saturate_w)
+
+GEN_PSIMD_VAR_SSHA(ssha, uint32_t, int32_t, int64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 32, signed_saturate_w)
+
+/**
+ * SSHAR - 32-bit scalar variable shift with rounding and saturation
+ */
+uint32_t HELPER(sshar)(CPURISCVState *env, uint32_t rs1, uint32_t rs2)
+{
+ int32_t a = (int32_t)rs1;
+ int8_t shamt = (int8_t)(rs2 & 0xFF);
+ int sat = 0;
+ int32_t res;
+
+ if (shamt >= 0) {
+ int64_t shifted = (int64_t)a << shamt;
+ res = signed_saturate_w(shifted, &sat);
+ } else {
+ int right = -shamt;
+ if (right >= 32) {
+ res = (a < 0) ? -1 : 0;
+ } else {
+ int64_t rounded = ((a >> (right - 1)) + 1) >> 1;
+ res = (int32_t)rounded;
+ }
+ }
+
+ if (sat) {
+ env->vxsat = 1;
+ }
+ return (uint32_t)res;
+}
+
+GEN_PSIMD_VAR_SRA64(sha, PSIMD_DO_SRA64)
+GEN_PSIMD_VAR_SRA64(shar, PSIMD_DO_RNDSRA64)
+
+GEN_PSIMD_VAR_USHL(psshl_hs, target_ulong, uint16_t, uint32_t,
+ EXTRACT16, INSERT16, ELEMS_H, 16, unsigned_saturate_h)
+
+GEN_PSIMD_VAR_USHLR(psshlr_hs, target_ulong, uint16_t, uint32_t,
+ EXTRACT16, INSERT16, ELEMS_H, 16, unsigned_saturate_h)
+
+GEN_PSIMD_VAR_USHL(psshl_ws, uint64_t, uint32_t, uint64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 32, unsigned_saturate_w)
+
+GEN_PSIMD_VAR_USHLR(psshlr_ws, uint64_t, uint32_t, uint64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 32, unsigned_saturate_w)
+
+GEN_PSIMD_VAR_USHL(sshl, uint32_t, uint32_t, uint64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 32, unsigned_saturate_w)
+
+GEN_PSIMD_VAR_USHLR(sshlr, uint32_t, uint32_t, uint64_t,
+ EXTRACT32, INSERT32, ELEMS_W, 32, unsigned_saturate_w)
+
+GEN_PSIMD_VAR_SRL64(shl, PSIMD_DO_SRL64)
+GEN_PSIMD_VAR_SRL64(shlr, PSIMD_DO_RNDSRL64)
+
--
2.34.1