[PATCH v7 6/8] target/riscv: Support raising misaligned exceptions for vector loads/stores
[email protected] Thu, 6 Aug 2026 00:43:43 +0800
Newsgroups
org.nongnu.qemu-riscv,org.nongnu.qemu-devel
Message-ID
<[email protected] >
From: Frank Chang <[email protected] >
When the Zicclsm extension is not enabled, raise misaligned load/store
exceptions for misaligned accesses from vector load/store instructions.
We will skip the host fast-path and fall back to the slow TLB-path to
raise misaligned load/store exceptions for the misaligned accesses when
Zicclsm extension is disabled.
Signed-off-by: Frank Chang <[email protected] >
Reviewed-by: Max Chou <[email protected] >
Acked-by: Alistair Francis <[email protected] >
---
target/riscv/tcg/insn_trans/trans_rvv.c.inc | 18 +++++-
target/riscv/tcg/vector_helper.c | 65 ++++++++++++++++-----
2 files changed, 65 insertions(+), 18 deletions(-)
diff --git a/target/riscv/tcg/insn_trans/trans_rvv.c.inc b/target/riscv/tcg/insn_trans/trans_rvv.c.inc
index 23262b1d036..a22e2cae6ce 100644
--- a/target/riscv/tcg/insn_trans/trans_rvv.c.inc
+++ b/target/riscv/tcg/insn_trans/trans_rvv.c.inc
@@ -1191,26 +1191,38 @@ static bool ldst_whole_trans(uint32_t vd, uint32_t rs1, uint32_t nf,
* Use the helper function if either:
* - vstart is not 0.
*/
-
bool use_helper_fn = !s->vstart_eq_zero;
if (!use_helper_fn) {
uint32_t size = s->cfg_ptr->vlenb * nf;
TCGv_i64 t8 = tcg_temp_new_i64();
MemOp atomicity = MO_ATOM_NONE;
+ MemOp alignment = MO_UNALN;
+
+ /*
+ * If Zicclsm is disabled, require alignment based on element size.
+ * Use MO_ALIGN_* based on log2_esz (0 = MO_UNALN, 1 = MO_ALIGN_2, etc).
+ */
+ if (!s->cfg_ptr->ext_zicclsm) {
+ alignment = log2_esz << MO_ASHIFT;
+ }
+
if (log2_esz == 0) {
atomicity = MO_ATOM_NONE;
} else {
atomicity = MO_ATOM_IFALIGN_PAIR;
}
+
for (int i = 0; i < size; i += 8) {
TCGv addr = get_address(s, rs1, i);
if (is_load) {
- tcg_gen_qemu_ld_i64(t8, addr, s->mem_idx, MO_LEUQ | atomicity);
+ tcg_gen_qemu_ld_i64(t8, addr, s->mem_idx,
+ MO_LEUQ | atomicity | alignment);
tcg_gen_st_i64(t8, tcg_env, vreg_ofs(s, vd) + i);
} else {
tcg_gen_ld_i64(t8, tcg_env, vreg_ofs(s, vd) + i);
- tcg_gen_qemu_st_i64(t8, addr, s->mem_idx, MO_LEUQ | atomicity);
+ tcg_gen_qemu_st_i64(t8, addr, s->mem_idx,
+ MO_LEUQ | atomicity | alignment);
}
if (i == size - 8) {
tcg_gen_movi_i32(cpu_vstart, 0);
diff --git a/target/riscv/tcg/vector_helper.c b/target/riscv/tcg/vector_helper.c
index e321ca26161..e28d8a3d9fd 100644
--- a/target/riscv/tcg/vector_helper.c
+++ b/target/riscv/tcg/vector_helper.c
@@ -199,20 +199,34 @@ static inline void vext_set_elem_mask(void *v0, int index,
((uint64_t *)v0)[idx] = deposit64(old, pos, 1, value);
}
+static inline MemOpIdx vext_make_memop_idx(CPURISCVState *env, size_t size)
+{
+ int mmu_idx = riscv_env_mmu_index(env, false);
+ MemOp memop = size_memop(size) | mo_endian_env(env);
+
+ if (!riscv_cpu_cfg(env)->ext_zicclsm) {
+ memop |= MO_ALIGN;
+ }
+
+ return make_memop_idx(memop, mmu_idx);
+}
+
/* elements operations for load and store */
typedef void vext_ldst_elem_fn_tlb(CPURISCVState *env, abi_ptr addr,
uint32_t idx, void *vd, uintptr_t retaddr);
typedef void vext_ldst_elem_fn_host(void *vd, uint32_t idx, void *host);
-#define GEN_VEXT_LD_ELEM(NAME, ETYPE, H, LDSUF) \
+#define GEN_VEXT_TLB_LD_ELEM(NAME, ETYPE, H, LDSUF) \
static inline QEMU_ALWAYS_INLINE \
void NAME##_tlb(CPURISCVState *env, abi_ptr addr, \
uint32_t idx, void *vd, uintptr_t retaddr) \
{ \
ETYPE *cur = ((ETYPE *)vd + H(idx)); \
- *cur = cpu_##LDSUF##_data_ra(env, addr, retaddr); \
+ MemOpIdx oi = vext_make_memop_idx(env, sizeof(ETYPE)); \
+ *cur = cpu_##LDSUF##_mmu(env, addr, oi, retaddr); \
} \
- \
+
+#define GEN_VEXT_HOST_LD_ELEM(NAME, ETYPE, H, LDSUF) \
static inline QEMU_ALWAYS_INLINE \
void NAME##_host(void *vd, uint32_t idx, void *host) \
{ \
@@ -220,20 +234,27 @@ void NAME##_host(void *vd, uint32_t idx, void *host) \
*cur = (ETYPE)LDSUF##_p(host); \
}
-GEN_VEXT_LD_ELEM(lde_b, uint8_t, H1, ldub)
-GEN_VEXT_LD_ELEM(lde_h, uint16_t, H2, lduw_le)
-GEN_VEXT_LD_ELEM(lde_w, uint32_t, H4, ldl_le)
-GEN_VEXT_LD_ELEM(lde_d, uint64_t, H8, ldq_le)
+GEN_VEXT_TLB_LD_ELEM(lde_b, uint8_t, H1, ldb)
+GEN_VEXT_TLB_LD_ELEM(lde_h, uint16_t, H2, ldw)
+GEN_VEXT_TLB_LD_ELEM(lde_w, uint32_t, H4, ldl)
+GEN_VEXT_TLB_LD_ELEM(lde_d, uint64_t, H8, ldq)
-#define GEN_VEXT_ST_ELEM(NAME, ETYPE, H, STSUF) \
+GEN_VEXT_HOST_LD_ELEM(lde_b, uint8_t, H1, ldub)
+GEN_VEXT_HOST_LD_ELEM(lde_h, uint16_t, H2, lduw_le)
+GEN_VEXT_HOST_LD_ELEM(lde_w, uint32_t, H4, ldl_le)
+GEN_VEXT_HOST_LD_ELEM(lde_d, uint64_t, H8, ldq_le)
+
+#define GEN_VEXT_TLB_ST_ELEM(NAME, ETYPE, H, STSUF) \
static inline QEMU_ALWAYS_INLINE \
void NAME##_tlb(CPURISCVState *env, abi_ptr addr, \
uint32_t idx, void *vd, uintptr_t retaddr) \
{ \
ETYPE data = *((ETYPE *)vd + H(idx)); \
- cpu_##STSUF##_data_ra(env, addr, data, retaddr); \
+ MemOpIdx oi = vext_make_memop_idx(env, sizeof(ETYPE)); \
+ cpu_##STSUF##_mmu(env, addr, data, oi, retaddr); \
} \
- \
+
+#define GEN_VEXT_HOST_ST_ELEM(NAME, ETYPE, H, STSUF) \
static inline QEMU_ALWAYS_INLINE \
void NAME##_host(void *vd, uint32_t idx, void *host) \
{ \
@@ -241,10 +262,15 @@ void NAME##_host(void *vd, uint32_t idx, void *host) \
STSUF##_p(host, data); \
}
-GEN_VEXT_ST_ELEM(ste_b, uint8_t, H1, stb)
-GEN_VEXT_ST_ELEM(ste_h, uint16_t, H2, stw_le)
-GEN_VEXT_ST_ELEM(ste_w, uint32_t, H4, stl_le)
-GEN_VEXT_ST_ELEM(ste_d, uint64_t, H8, stq_le)
+GEN_VEXT_TLB_ST_ELEM(ste_b, uint8_t, H1, stb)
+GEN_VEXT_TLB_ST_ELEM(ste_h, uint16_t, H2, stw)
+GEN_VEXT_TLB_ST_ELEM(ste_w, uint32_t, H4, stl)
+GEN_VEXT_TLB_ST_ELEM(ste_d, uint64_t, H8, stq)
+
+GEN_VEXT_HOST_ST_ELEM(ste_b, uint8_t, H1, stb)
+GEN_VEXT_HOST_ST_ELEM(ste_h, uint16_t, H2, stw_le)
+GEN_VEXT_HOST_ST_ELEM(ste_w, uint32_t, H4, stl_le)
+GEN_VEXT_HOST_ST_ELEM(ste_d, uint64_t, H8, stq_le)
static inline QEMU_ALWAYS_INLINE void
vext_continuous_ldst_tlb(CPURISCVState *env, vext_ldst_elem_fn_tlb *ldst_tlb,
@@ -398,7 +424,16 @@ vext_page_ldst_us(CPURISCVState *env, void *vd, target_ulong addr,
probe_pages(env, addr, size, ra, access_type, mmu_index, &host, &flags,
true);
- if (flags == 0) {
+ bool misaligned = addr & (esz - 1);
+
+ /*
+ * Allow the host fast-pash when:
+ * 1. Page permission/pmp/watchpoint are checked and we have a contigous
+ * host mapping.
+ * 2. Zicclsm is enabled or load/store is not a misaligned access.
+ * Otherwise, we will fall back to the slow TLB-path.
+ */
+ if (flags == 0 && (riscv_cpu_cfg(env)->ext_zicclsm || !misaligned)) {
if (nf == 1) {
vext_continuous_ldst_host(env, ldst_host, vd, evl, env->vstart,
host, esz, is_load);
--
2.43.0