On Tue, Jun 30, 2026 at 8:46 PM <[email protected]> wrote: > > From: Frank Chang <[email protected]> > > When the Zicclsm extension is not enabled, raise misaligned load/store > exceptions for misaligned accesses from vector load/store instructions. > > We will skip the host fast-path and fall back to the slow TLB-path to > raise misaligned load/store exceptions for the misaligned accesses when > Zicclsm extension is disabled. > > Signed-off-by: Frank Chang <[email protected]> > Reviewed-by: Max Chou <[email protected]>
Acked-by: Alistair Francis <[email protected]> Alistair > --- > target/riscv/insn_trans/trans_rvv.c.inc | 18 +++++-- > target/riscv/vector_helper.c | 65 +++++++++++++++++++------ > 2 files changed, 65 insertions(+), 18 deletions(-) > > diff --git a/target/riscv/insn_trans/trans_rvv.c.inc > b/target/riscv/insn_trans/trans_rvv.c.inc > index 23262b1d036..a22e2cae6ce 100644 > --- a/target/riscv/insn_trans/trans_rvv.c.inc > +++ b/target/riscv/insn_trans/trans_rvv.c.inc > @@ -1191,26 +1191,38 @@ static bool ldst_whole_trans(uint32_t vd, uint32_t > rs1, uint32_t nf, > * Use the helper function if either: > * - vstart is not 0. > */ > - > bool use_helper_fn = !s->vstart_eq_zero; > > if (!use_helper_fn) { > uint32_t size = s->cfg_ptr->vlenb * nf; > TCGv_i64 t8 = tcg_temp_new_i64(); > MemOp atomicity = MO_ATOM_NONE; > + MemOp alignment = MO_UNALN; > + > + /* > + * If Zicclsm is disabled, require alignment based on element size. > + * Use MO_ALIGN_* based on log2_esz (0 = MO_UNALN, 1 = MO_ALIGN_2, > etc). > + */ > + if (!s->cfg_ptr->ext_zicclsm) { > + alignment = log2_esz << MO_ASHIFT; > + } > + > if (log2_esz == 0) { > atomicity = MO_ATOM_NONE; > } else { > atomicity = MO_ATOM_IFALIGN_PAIR; > } > + > for (int i = 0; i < size; i += 8) { > TCGv addr = get_address(s, rs1, i); > if (is_load) { > - tcg_gen_qemu_ld_i64(t8, addr, s->mem_idx, MO_LEUQ | > atomicity); > + tcg_gen_qemu_ld_i64(t8, addr, s->mem_idx, > + MO_LEUQ | atomicity | alignment); > tcg_gen_st_i64(t8, tcg_env, vreg_ofs(s, vd) + i); > } else { > tcg_gen_ld_i64(t8, tcg_env, vreg_ofs(s, vd) + i); > - tcg_gen_qemu_st_i64(t8, addr, s->mem_idx, MO_LEUQ | > atomicity); > + tcg_gen_qemu_st_i64(t8, addr, s->mem_idx, > + MO_LEUQ | atomicity | alignment); > } > if (i == size - 8) { > tcg_gen_movi_i32(cpu_vstart, 0); > diff --git a/target/riscv/vector_helper.c b/target/riscv/vector_helper.c > index e321ca26161..e28d8a3d9fd 100644 > --- a/target/riscv/vector_helper.c > +++ b/target/riscv/vector_helper.c > @@ -199,20 +199,34 @@ static inline void vext_set_elem_mask(void *v0, int > index, > ((uint64_t *)v0)[idx] = deposit64(old, pos, 1, value); > } > > +static inline MemOpIdx vext_make_memop_idx(CPURISCVState *env, size_t size) > +{ > + int mmu_idx = riscv_env_mmu_index(env, false); > + MemOp memop = size_memop(size) | mo_endian_env(env); > + > + if (!riscv_cpu_cfg(env)->ext_zicclsm) { > + memop |= MO_ALIGN; > + } > + > + return make_memop_idx(memop, mmu_idx); > +} > + > /* elements operations for load and store */ > typedef void vext_ldst_elem_fn_tlb(CPURISCVState *env, abi_ptr addr, > uint32_t idx, void *vd, uintptr_t > retaddr); > typedef void vext_ldst_elem_fn_host(void *vd, uint32_t idx, void *host); > > -#define GEN_VEXT_LD_ELEM(NAME, ETYPE, H, LDSUF) \ > +#define GEN_VEXT_TLB_LD_ELEM(NAME, ETYPE, H, LDSUF) \ > static inline QEMU_ALWAYS_INLINE \ > void NAME##_tlb(CPURISCVState *env, abi_ptr addr, \ > uint32_t idx, void *vd, uintptr_t retaddr) \ > { \ > ETYPE *cur = ((ETYPE *)vd + H(idx)); \ > - *cur = cpu_##LDSUF##_data_ra(env, addr, retaddr); \ > + MemOpIdx oi = vext_make_memop_idx(env, sizeof(ETYPE)); \ > + *cur = cpu_##LDSUF##_mmu(env, addr, oi, retaddr); \ > } \ > - \ > + > +#define GEN_VEXT_HOST_LD_ELEM(NAME, ETYPE, H, LDSUF) \ > static inline QEMU_ALWAYS_INLINE \ > void NAME##_host(void *vd, uint32_t idx, void *host) \ > { \ > @@ -220,20 +234,27 @@ void NAME##_host(void *vd, uint32_t idx, void *host) > \ > *cur = (ETYPE)LDSUF##_p(host); \ > } > > -GEN_VEXT_LD_ELEM(lde_b, uint8_t, H1, ldub) > -GEN_VEXT_LD_ELEM(lde_h, uint16_t, H2, lduw_le) > -GEN_VEXT_LD_ELEM(lde_w, uint32_t, H4, ldl_le) > -GEN_VEXT_LD_ELEM(lde_d, uint64_t, H8, ldq_le) > +GEN_VEXT_TLB_LD_ELEM(lde_b, uint8_t, H1, ldb) > +GEN_VEXT_TLB_LD_ELEM(lde_h, uint16_t, H2, ldw) > +GEN_VEXT_TLB_LD_ELEM(lde_w, uint32_t, H4, ldl) > +GEN_VEXT_TLB_LD_ELEM(lde_d, uint64_t, H8, ldq) > > -#define GEN_VEXT_ST_ELEM(NAME, ETYPE, H, STSUF) \ > +GEN_VEXT_HOST_LD_ELEM(lde_b, uint8_t, H1, ldub) > +GEN_VEXT_HOST_LD_ELEM(lde_h, uint16_t, H2, lduw_le) > +GEN_VEXT_HOST_LD_ELEM(lde_w, uint32_t, H4, ldl_le) > +GEN_VEXT_HOST_LD_ELEM(lde_d, uint64_t, H8, ldq_le) > + > +#define GEN_VEXT_TLB_ST_ELEM(NAME, ETYPE, H, STSUF) \ > static inline QEMU_ALWAYS_INLINE \ > void NAME##_tlb(CPURISCVState *env, abi_ptr addr, \ > uint32_t idx, void *vd, uintptr_t retaddr) \ > { \ > ETYPE data = *((ETYPE *)vd + H(idx)); \ > - cpu_##STSUF##_data_ra(env, addr, data, retaddr); \ > + MemOpIdx oi = vext_make_memop_idx(env, sizeof(ETYPE)); \ > + cpu_##STSUF##_mmu(env, addr, data, oi, retaddr); \ > } \ > - \ > + > +#define GEN_VEXT_HOST_ST_ELEM(NAME, ETYPE, H, STSUF) \ > static inline QEMU_ALWAYS_INLINE \ > void NAME##_host(void *vd, uint32_t idx, void *host) \ > { \ > @@ -241,10 +262,15 @@ void NAME##_host(void *vd, uint32_t idx, void *host) > \ > STSUF##_p(host, data); \ > } > > -GEN_VEXT_ST_ELEM(ste_b, uint8_t, H1, stb) > -GEN_VEXT_ST_ELEM(ste_h, uint16_t, H2, stw_le) > -GEN_VEXT_ST_ELEM(ste_w, uint32_t, H4, stl_le) > -GEN_VEXT_ST_ELEM(ste_d, uint64_t, H8, stq_le) > +GEN_VEXT_TLB_ST_ELEM(ste_b, uint8_t, H1, stb) > +GEN_VEXT_TLB_ST_ELEM(ste_h, uint16_t, H2, stw) > +GEN_VEXT_TLB_ST_ELEM(ste_w, uint32_t, H4, stl) > +GEN_VEXT_TLB_ST_ELEM(ste_d, uint64_t, H8, stq) > + > +GEN_VEXT_HOST_ST_ELEM(ste_b, uint8_t, H1, stb) > +GEN_VEXT_HOST_ST_ELEM(ste_h, uint16_t, H2, stw_le) > +GEN_VEXT_HOST_ST_ELEM(ste_w, uint32_t, H4, stl_le) > +GEN_VEXT_HOST_ST_ELEM(ste_d, uint64_t, H8, stq_le) > > static inline QEMU_ALWAYS_INLINE void > vext_continuous_ldst_tlb(CPURISCVState *env, vext_ldst_elem_fn_tlb *ldst_tlb, > @@ -398,7 +424,16 @@ vext_page_ldst_us(CPURISCVState *env, void *vd, > target_ulong addr, > probe_pages(env, addr, size, ra, access_type, mmu_index, &host, &flags, > true); > > - if (flags == 0) { > + bool misaligned = addr & (esz - 1); > + > + /* > + * Allow the host fast-pash when: > + * 1. Page permission/pmp/watchpoint are checked and we have a > contigous > + * host mapping. > + * 2. Zicclsm is enabled or load/store is not a misaligned access. > + * Otherwise, we will fall back to the slow TLB-path. > + */ > + if (flags == 0 && (riscv_cpu_cfg(env)->ext_zicclsm || !misaligned)) { > if (nf == 1) { > vext_continuous_ldst_host(env, ldst_host, vd, evl, env->vstart, > host, esz, is_load); > -- > 2.43.0 > >
