On Tue, Jun 30, 2026 at 8:46 PM <[email protected]> wrote:
>
> From: Frank Chang <[email protected]>
>
> When the Zicclsm extension is not enabled, raise misaligned load/store
> exceptions for misaligned accesses from vector load/store instructions.
>
> We will skip the host fast-path and fall back to the slow TLB-path to
> raise misaligned load/store exceptions for the misaligned accesses when
> Zicclsm extension is disabled.
>
> Signed-off-by: Frank Chang <[email protected]>
> Reviewed-by: Max Chou <[email protected]>

Acked-by: Alistair Francis <[email protected]>

Alistair

> ---
>  target/riscv/insn_trans/trans_rvv.c.inc | 18 +++++--
>  target/riscv/vector_helper.c            | 65 +++++++++++++++++++------
>  2 files changed, 65 insertions(+), 18 deletions(-)
>
> diff --git a/target/riscv/insn_trans/trans_rvv.c.inc 
> b/target/riscv/insn_trans/trans_rvv.c.inc
> index 23262b1d036..a22e2cae6ce 100644
> --- a/target/riscv/insn_trans/trans_rvv.c.inc
> +++ b/target/riscv/insn_trans/trans_rvv.c.inc
> @@ -1191,26 +1191,38 @@ static bool ldst_whole_trans(uint32_t vd, uint32_t 
> rs1, uint32_t nf,
>       * Use the helper function if either:
>       * - vstart is not 0.
>       */
> -
>      bool use_helper_fn = !s->vstart_eq_zero;
>
>      if (!use_helper_fn) {
>          uint32_t size = s->cfg_ptr->vlenb * nf;
>          TCGv_i64 t8 = tcg_temp_new_i64();
>          MemOp atomicity = MO_ATOM_NONE;
> +        MemOp alignment = MO_UNALN;
> +
> +        /*
> +         * If Zicclsm is disabled, require alignment based on element size.
> +         * Use MO_ALIGN_* based on log2_esz (0 = MO_UNALN, 1 = MO_ALIGN_2, 
> etc).
> +         */
> +        if (!s->cfg_ptr->ext_zicclsm) {
> +            alignment = log2_esz << MO_ASHIFT;
> +        }
> +
>          if (log2_esz == 0) {
>              atomicity = MO_ATOM_NONE;
>          } else {
>              atomicity = MO_ATOM_IFALIGN_PAIR;
>          }
> +
>          for (int i = 0; i < size; i += 8) {
>              TCGv addr = get_address(s, rs1, i);
>              if (is_load) {
> -                tcg_gen_qemu_ld_i64(t8, addr, s->mem_idx, MO_LEUQ | 
> atomicity);
> +                tcg_gen_qemu_ld_i64(t8, addr, s->mem_idx,
> +                                    MO_LEUQ | atomicity | alignment);
>                  tcg_gen_st_i64(t8, tcg_env, vreg_ofs(s, vd) + i);
>              } else {
>                  tcg_gen_ld_i64(t8, tcg_env, vreg_ofs(s, vd) + i);
> -                tcg_gen_qemu_st_i64(t8, addr, s->mem_idx, MO_LEUQ | 
> atomicity);
> +                tcg_gen_qemu_st_i64(t8, addr, s->mem_idx,
> +                                    MO_LEUQ | atomicity | alignment);
>              }
>              if (i == size - 8) {
>                  tcg_gen_movi_i32(cpu_vstart, 0);
> diff --git a/target/riscv/vector_helper.c b/target/riscv/vector_helper.c
> index e321ca26161..e28d8a3d9fd 100644
> --- a/target/riscv/vector_helper.c
> +++ b/target/riscv/vector_helper.c
> @@ -199,20 +199,34 @@ static inline void vext_set_elem_mask(void *v0, int 
> index,
>      ((uint64_t *)v0)[idx] = deposit64(old, pos, 1, value);
>  }
>
> +static inline MemOpIdx vext_make_memop_idx(CPURISCVState *env, size_t size)
> +{
> +    int mmu_idx = riscv_env_mmu_index(env, false);
> +    MemOp memop = size_memop(size) | mo_endian_env(env);
> +
> +    if (!riscv_cpu_cfg(env)->ext_zicclsm) {
> +        memop |= MO_ALIGN;
> +    }
> +
> +    return make_memop_idx(memop, mmu_idx);
> +}
> +
>  /* elements operations for load and store */
>  typedef void vext_ldst_elem_fn_tlb(CPURISCVState *env, abi_ptr addr,
>                                     uint32_t idx, void *vd, uintptr_t 
> retaddr);
>  typedef void vext_ldst_elem_fn_host(void *vd, uint32_t idx, void *host);
>
> -#define GEN_VEXT_LD_ELEM(NAME, ETYPE, H, LDSUF)             \
> +#define GEN_VEXT_TLB_LD_ELEM(NAME, ETYPE, H, LDSUF)         \
>  static inline QEMU_ALWAYS_INLINE                            \
>  void NAME##_tlb(CPURISCVState *env, abi_ptr addr,           \
>                  uint32_t idx, void *vd, uintptr_t retaddr)  \
>  {                                                           \
>      ETYPE *cur = ((ETYPE *)vd + H(idx));                    \
> -    *cur = cpu_##LDSUF##_data_ra(env, addr, retaddr);       \
> +    MemOpIdx oi = vext_make_memop_idx(env, sizeof(ETYPE));  \
> +    *cur = cpu_##LDSUF##_mmu(env, addr, oi, retaddr);       \
>  }                                                           \
> -                                                            \
> +
> +#define GEN_VEXT_HOST_LD_ELEM(NAME, ETYPE, H, LDSUF)        \
>  static inline QEMU_ALWAYS_INLINE                            \
>  void NAME##_host(void *vd, uint32_t idx, void *host)        \
>  {                                                           \
> @@ -220,20 +234,27 @@ void NAME##_host(void *vd, uint32_t idx, void *host)    
>     \
>      *cur = (ETYPE)LDSUF##_p(host);                          \
>  }
>
> -GEN_VEXT_LD_ELEM(lde_b, uint8_t,  H1, ldub)
> -GEN_VEXT_LD_ELEM(lde_h, uint16_t, H2, lduw_le)
> -GEN_VEXT_LD_ELEM(lde_w, uint32_t, H4, ldl_le)
> -GEN_VEXT_LD_ELEM(lde_d, uint64_t, H8, ldq_le)
> +GEN_VEXT_TLB_LD_ELEM(lde_b, uint8_t,  H1, ldb)
> +GEN_VEXT_TLB_LD_ELEM(lde_h, uint16_t, H2, ldw)
> +GEN_VEXT_TLB_LD_ELEM(lde_w, uint32_t, H4, ldl)
> +GEN_VEXT_TLB_LD_ELEM(lde_d, uint64_t, H8, ldq)
>
> -#define GEN_VEXT_ST_ELEM(NAME, ETYPE, H, STSUF)             \
> +GEN_VEXT_HOST_LD_ELEM(lde_b, uint8_t,  H1, ldub)
> +GEN_VEXT_HOST_LD_ELEM(lde_h, uint16_t, H2, lduw_le)
> +GEN_VEXT_HOST_LD_ELEM(lde_w, uint32_t, H4, ldl_le)
> +GEN_VEXT_HOST_LD_ELEM(lde_d, uint64_t, H8, ldq_le)
> +
> +#define GEN_VEXT_TLB_ST_ELEM(NAME, ETYPE, H, STSUF)         \
>  static inline QEMU_ALWAYS_INLINE                            \
>  void NAME##_tlb(CPURISCVState *env, abi_ptr addr,           \
>                  uint32_t idx, void *vd, uintptr_t retaddr)  \
>  {                                                           \
>      ETYPE data = *((ETYPE *)vd + H(idx));                   \
> -    cpu_##STSUF##_data_ra(env, addr, data, retaddr);        \
> +    MemOpIdx oi = vext_make_memop_idx(env, sizeof(ETYPE));  \
> +    cpu_##STSUF##_mmu(env, addr, data, oi, retaddr);        \
>  }                                                           \
> -                                                            \
> +
> +#define GEN_VEXT_HOST_ST_ELEM(NAME, ETYPE, H, STSUF)        \
>  static inline QEMU_ALWAYS_INLINE                            \
>  void NAME##_host(void *vd, uint32_t idx, void *host)        \
>  {                                                           \
> @@ -241,10 +262,15 @@ void NAME##_host(void *vd, uint32_t idx, void *host)    
>     \
>      STSUF##_p(host, data);                                  \
>  }
>
> -GEN_VEXT_ST_ELEM(ste_b, uint8_t,  H1, stb)
> -GEN_VEXT_ST_ELEM(ste_h, uint16_t, H2, stw_le)
> -GEN_VEXT_ST_ELEM(ste_w, uint32_t, H4, stl_le)
> -GEN_VEXT_ST_ELEM(ste_d, uint64_t, H8, stq_le)
> +GEN_VEXT_TLB_ST_ELEM(ste_b, uint8_t,  H1, stb)
> +GEN_VEXT_TLB_ST_ELEM(ste_h, uint16_t, H2, stw)
> +GEN_VEXT_TLB_ST_ELEM(ste_w, uint32_t, H4, stl)
> +GEN_VEXT_TLB_ST_ELEM(ste_d, uint64_t, H8, stq)
> +
> +GEN_VEXT_HOST_ST_ELEM(ste_b, uint8_t,  H1, stb)
> +GEN_VEXT_HOST_ST_ELEM(ste_h, uint16_t, H2, stw_le)
> +GEN_VEXT_HOST_ST_ELEM(ste_w, uint32_t, H4, stl_le)
> +GEN_VEXT_HOST_ST_ELEM(ste_d, uint64_t, H8, stq_le)
>
>  static inline QEMU_ALWAYS_INLINE void
>  vext_continuous_ldst_tlb(CPURISCVState *env, vext_ldst_elem_fn_tlb *ldst_tlb,
> @@ -398,7 +424,16 @@ vext_page_ldst_us(CPURISCVState *env, void *vd, 
> target_ulong addr,
>      probe_pages(env, addr, size, ra, access_type, mmu_index, &host, &flags,
>                  true);
>
> -    if (flags == 0) {
> +    bool misaligned = addr & (esz - 1);
> +
> +    /*
> +     * Allow the host fast-pash when:
> +     *   1. Page permission/pmp/watchpoint are checked and we have a 
> contigous
> +     *      host mapping.
> +     *   2. Zicclsm is enabled or load/store is not a misaligned access.
> +     * Otherwise, we will fall back to the slow TLB-path.
> +     */
> +    if (flags == 0 && (riscv_cpu_cfg(env)->ext_zicclsm || !misaligned)) {
>          if (nf == 1) {
>              vext_continuous_ldst_host(env, ldst_host, vd, evl, env->vstart,
>                                        host, esz, is_load);
> --
> 2.43.0
>
>

Reply via email to