https://gcc.gnu.org/g:e485ab3fc8b509f1860a2e8456e516d140e23a34
commit r17-3838-ge485ab3fc8b509f1860a2e8456e516d140e23a34 Author: Pan Li <[email protected]> Date: Tue Sep 1 10:25:32 2026 +0800 RISC-V: Add test cases for vwmaccus.vx reg overlap Add test cases for register group overlap, please note it is not overlap as much as possible. gcc/testsuite/ChangeLog: * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m1.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m2.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m4.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-mf2.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-mf4.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m1.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m2.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m4.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-mf2.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m1.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m2.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m4.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf2.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf4.c: New test. * gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf8.c: New test. Signed-off-by: Pan Li <[email protected]> Diff: --- .../rvv/autovec/group_overlap/vwmaccus_vx-i16-m1.c | 58 ++++++++++++++++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i16-m2.c | 50 +++++++++++++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i16-m4.c | 46 +++++++++++++++++ .../autovec/group_overlap/vwmaccus_vx-i16-mf2.c | 21 ++++++++ .../autovec/group_overlap/vwmaccus_vx-i16-mf4.c | 21 ++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i32-m1.c | 58 ++++++++++++++++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i32-m2.c | 50 +++++++++++++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i32-m4.c | 46 +++++++++++++++++ .../autovec/group_overlap/vwmaccus_vx-i32-mf2.c | 21 ++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i8-m1.c | 58 ++++++++++++++++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i8-m2.c | 50 +++++++++++++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i8-m4.c | 46 +++++++++++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i8-mf2.c | 21 ++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i8-mf4.c | 21 ++++++++ .../rvv/autovec/group_overlap/vwmaccus_vx-i8-mf8.c | 21 ++++++++ 15 files changed, 588 insertions(+) diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m1.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m1.c new file mode 100644 index 000000000000..f3c542f7862e --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m1.c @@ -0,0 +1,58 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e16m1, + vint16m1_t, + vint32m2_t, + uint16_t, + __riscv_vle16_v_i16m1, + __riscv_vle32_v_i32m2, + __riscv_vwmaccus_vx_i32m2, + __riscv_vse32_v_i32m2, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e16m1, + vint16m1_t, + vint32m2_t, + vint16m2_t, + uint16_t, + __riscv_vle32_v_i32m2, + __riscv_vreinterpret_v_i32m2_i16m2, + __riscv_vget_v_i16m2_i16m1, + __riscv_vwmaccus_vx_i32m2, + __riscv_vse32_v_i32m2, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v6,s0,v27([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,s0,v26([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v14,s0,v31([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v12,s0,v30([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v2,s0,v25([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v26,s0,v24([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v10,s0,v29([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v30,s0,v23([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,s0,v22([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,s0,v28([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v26,s0,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,s0,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v22,s0,v21([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,s0,v18([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v20,s0,v19([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v18,s0,v1([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v5([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v2,a2,v3([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m2.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m2.c new file mode 100644 index 000000000000..76be9321813a --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m2.c @@ -0,0 +1,50 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e16m2, + vint16m2_t, + vint32m4_t, + uint16_t, + __riscv_vle16_v_i16m2, + __riscv_vle32_v_i32m4, + __riscv_vwmaccus_vx_i32m4, + __riscv_vse32_v_i32m4, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X8) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e16m2, + vint16m2_t, + vint32m4_t, + vint16m4_t, + uint16_t, + __riscv_vle32_v_i32m4, + __riscv_vreinterpret_v_i32m4_i16m4, + __riscv_vget_v_i16m4_i16m2, + __riscv_vwmaccus_vx_i32m4, + __riscv_vse32_v_i32m4, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v12,a2,v30([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v28([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v26([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,a2,v24([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,a2,v22([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,a2,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v20,a2,v2([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v10([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v6([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m4.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m4.c new file mode 100644 index 000000000000..0598c58ac504 --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-m4.c @@ -0,0 +1,46 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e16m4, + vint16m4_t, + vint32m8_t, + uint16_t, + __riscv_vle16_v_i16m4, + __riscv_vle32_v_i32m8, + __riscv_vwmaccus_vx_i32m8, + __riscv_vse32_v_i32m8, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X4) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e16m4, + vint16m4_t, + vint32m8_t, + vint16m8_t, + uint16_t, + __riscv_vle32_v_i32m8, + __riscv_vreinterpret_v_i32m8_i16m8, + __riscv_vget_v_i16m8_i16m4, + __riscv_vwmaccus_vx_i32m8, + __riscv_vse32_v_i32m8, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,a2,v12([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v8([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v4([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v12([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-mf2.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-mf2.c new file mode 100644 index 000000000000..8432b660cb83 --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-mf2.c @@ -0,0 +1,21 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e16m1, + vint16mf2_t, + vint32m1_t, + uint16_t, + __riscv_vle16_v_i16mf2, + __riscv_vle32_v_i32m1, + __riscv_vwmaccus_vx_i32m1, + __riscv_vse32_v_i32m1, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +/* The fractional LMUL source has EMUL < 1, thus the widened destination + register group must not overlap the source at all. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx} 16 } } */ +/* { dg-final { scan-assembler-not {vwmaccus\.vx\s+(v[0-9]+),[a-z0-9]+,\1([^0-9]|$)} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-mf4.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-mf4.c new file mode 100644 index 000000000000..8f306e4fc894 --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i16-mf4.c @@ -0,0 +1,21 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e16m1, + vint16mf4_t, + vint32mf2_t, + uint16_t, + __riscv_vle16_v_i16mf4, + __riscv_vle32_v_i32mf2, + __riscv_vwmaccus_vx_i32mf2, + __riscv_vse32_v_i32mf2, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +/* The fractional LMUL source has EMUL < 1, thus the widened destination + register group must not overlap the source at all. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx} 16 } } */ +/* { dg-final { scan-assembler-not {vwmaccus\.vx\s+(v[0-9]+),[a-z0-9]+,\1([^0-9]|$)} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m1.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m1.c new file mode 100644 index 000000000000..0202db0e622a --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m1.c @@ -0,0 +1,58 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e32m1, + vint32m1_t, + vint64m2_t, + uint32_t, + __riscv_vle32_v_i32m1, + __riscv_vle64_v_i64m2, + __riscv_vwmaccus_vx_i64m2, + __riscv_vse64_v_i64m2, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e32m1, + vint32m1_t, + vint64m2_t, + vint32m2_t, + uint32_t, + __riscv_vle64_v_i64m2, + __riscv_vreinterpret_v_i64m2_i32m2, + __riscv_vget_v_i32m2_i32m1, + __riscv_vwmaccus_vx_i64m2, + __riscv_vse64_v_i64m2, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v6,s0,v27([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,s0,v26([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v14,s0,v31([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v12,s0,v30([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v2,s0,v25([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v26,s0,v24([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v10,s0,v29([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v30,s0,v23([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,s0,v22([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,s0,v28([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v26,s0,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,s0,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v22,s0,v21([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,s0,v18([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v20,s0,v19([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v18,s0,v1([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v5([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v2,a2,v3([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m2.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m2.c new file mode 100644 index 000000000000..4269b027a034 --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m2.c @@ -0,0 +1,50 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e32m2, + vint32m2_t, + vint64m4_t, + uint32_t, + __riscv_vle32_v_i32m2, + __riscv_vle64_v_i64m4, + __riscv_vwmaccus_vx_i64m4, + __riscv_vse64_v_i64m4, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X8) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e32m2, + vint32m2_t, + vint64m4_t, + vint32m4_t, + uint32_t, + __riscv_vle64_v_i64m4, + __riscv_vreinterpret_v_i64m4_i32m4, + __riscv_vget_v_i32m4_i32m2, + __riscv_vwmaccus_vx_i64m4, + __riscv_vse64_v_i64m4, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v12,a2,v30([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v28([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v26([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,a2,v24([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,a2,v22([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,a2,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v20,a2,v2([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v10([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v6([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m4.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m4.c new file mode 100644 index 000000000000..3452089ab08d --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-m4.c @@ -0,0 +1,46 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e32m4, + vint32m4_t, + vint64m8_t, + uint32_t, + __riscv_vle32_v_i32m4, + __riscv_vle64_v_i64m8, + __riscv_vwmaccus_vx_i64m8, + __riscv_vse64_v_i64m8, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X4) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e32m4, + vint32m4_t, + vint64m8_t, + vint32m8_t, + uint32_t, + __riscv_vle64_v_i64m8, + __riscv_vreinterpret_v_i64m8_i32m8, + __riscv_vget_v_i32m8_i32m4, + __riscv_vwmaccus_vx_i64m8, + __riscv_vse64_v_i64m8, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,a2,v12([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v8([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v4([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v12([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-mf2.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-mf2.c new file mode 100644 index 000000000000..584557c7016c --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i32-mf2.c @@ -0,0 +1,21 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e32m1, + vint32mf2_t, + vint64m1_t, + uint32_t, + __riscv_vle32_v_i32mf2, + __riscv_vle64_v_i64m1, + __riscv_vwmaccus_vx_i64m1, + __riscv_vse64_v_i64m1, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +/* The fractional LMUL source has EMUL < 1, thus the widened destination + register group must not overlap the source at all. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx} 16 } } */ +/* { dg-final { scan-assembler-not {vwmaccus\.vx\s+(v[0-9]+),[a-z0-9]+,\1([^0-9]|$)} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m1.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m1.c new file mode 100644 index 000000000000..bffc1c62bd0a --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m1.c @@ -0,0 +1,58 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e8m1, + vint8m1_t, + vint16m2_t, + uint8_t, + __riscv_vle8_v_i8m1, + __riscv_vle16_v_i16m2, + __riscv_vwmaccus_vx_i16m2, + __riscv_vse16_v_i16m2, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e8m1, + vint8m1_t, + vint16m2_t, + vint8m2_t, + uint8_t, + __riscv_vle16_v_i16m2, + __riscv_vreinterpret_v_i16m2_i8m2, + __riscv_vget_v_i8m2_i8m1, + __riscv_vwmaccus_vx_i16m2, + __riscv_vse16_v_i16m2, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v6,s0,v27([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,s0,v26([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v14,s0,v31([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v12,s0,v30([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v2,s0,v25([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v26,s0,v24([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v10,s0,v29([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v30,s0,v23([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,s0,v22([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,s0,v28([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v26,s0,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,s0,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v22,s0,v21([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,s0,v18([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v20,s0,v19([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v18,s0,v1([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v5([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v2,a2,v3([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m2.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m2.c new file mode 100644 index 000000000000..e9d1dc07fa1f --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m2.c @@ -0,0 +1,50 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e8m2, + vint8m2_t, + vint16m4_t, + uint8_t, + __riscv_vle8_v_i8m2, + __riscv_vle16_v_i16m4, + __riscv_vwmaccus_vx_i16m4, + __riscv_vse16_v_i16m4, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X8) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e8m2, + vint8m2_t, + vint16m4_t, + vint8m4_t, + uint8_t, + __riscv_vle16_v_i16m4, + __riscv_vreinterpret_v_i16m4_i8m4, + __riscv_vget_v_i8m4_i8m2, + __riscv_vwmaccus_vx_i16m4, + __riscv_vse16_v_i16m4, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v12,a2,v30([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v28([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v26([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,a2,v24([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,a2,v22([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v28,a2,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v20,a2,v2([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v10([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v4,a2,v6([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m4.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m4.c new file mode 100644 index 000000000000..c72725db0b84 --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-m4.c @@ -0,0 +1,46 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e8m4, + vint8m4_t, + vint16m8_t, + uint8_t, + __riscv_vle8_v_i8m4, + __riscv_vle16_v_i16m8, + __riscv_vwmaccus_vx_i16m8, + __riscv_vse16_v_i16m8, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X4) + +DEF_GROUP_OVERLAP_TERNARY_5( + __riscv_vsetvlmax_e8m4, + vint8m4_t, + vint16m8_t, + vint8m8_t, + uint8_t, + __riscv_vle16_v_i16m8, + __riscv_vreinterpret_v_i16m8_i8m8, + __riscv_vget_v_i8m8_i8m4, + __riscv_vwmaccus_vx_i16m8, + __riscv_vse16_v_i16m8, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_OVERLAP_X2) + +/* ternary_4: the accumulator occupies the whole destination register group and + is live when the narrowed source is read, so the source cannot be allocated + inside the destination register group. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v24,a2,v12([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v0([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v8([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v4([^0-9]|$)} 1 } } */ + +/* ternary_5: the narrowed source is the highest-numbered half of its own + accumulator, thus it overlaps the highest-numbered part of the destination + register group. Without the group overlap the source would have to be + copied out to a disjoint register group first. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v16,a2,v20([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx\s+v8,a2,v12([^0-9]|$)} 1 } } */ +/* { dg-final { scan-assembler-not {vmv[0-9]+r\.v} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf2.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf2.c new file mode 100644 index 000000000000..72982bc6f723 --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf2.c @@ -0,0 +1,21 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e8m1, + vint8mf2_t, + vint16m1_t, + uint8_t, + __riscv_vle8_v_i8mf2, + __riscv_vle16_v_i16m1, + __riscv_vwmaccus_vx_i16m1, + __riscv_vse16_v_i16m1, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +/* The fractional LMUL source has EMUL < 1, thus the widened destination + register group must not overlap the source at all. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx} 16 } } */ +/* { dg-final { scan-assembler-not {vwmaccus\.vx\s+(v[0-9]+),[a-z0-9]+,\1([^0-9]|$)} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf4.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf4.c new file mode 100644 index 000000000000..bc3aeffdf9f4 --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf4.c @@ -0,0 +1,21 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e8m1, + vint8mf4_t, + vint16mf2_t, + uint8_t, + __riscv_vle8_v_i8mf4, + __riscv_vle16_v_i16mf2, + __riscv_vwmaccus_vx_i16mf2, + __riscv_vse16_v_i16mf2, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +/* The fractional LMUL source has EMUL < 1, thus the widened destination + register group must not overlap the source at all. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx} 16 } } */ +/* { dg-final { scan-assembler-not {vwmaccus\.vx\s+(v[0-9]+),[a-z0-9]+,\1([^0-9]|$)} } } */ diff --git a/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf8.c b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf8.c new file mode 100644 index 000000000000..a4dfc322f018 --- /dev/null +++ b/gcc/testsuite/gcc.target/riscv/rvv/autovec/group_overlap/vwmaccus_vx-i8-mf8.c @@ -0,0 +1,21 @@ +/* { dg-do compile } */ +/* { dg-options "-march=rv64gcv -mabi=lp64d" } */ + +#include "group_overlap.h" + +DEF_GROUP_OVERLAP_TERNARY_4( + __riscv_vsetvlmax_e8m1, + vint8mf8_t, + vint16mf4_t, + uint8_t, + __riscv_vle8_v_i8mf8, + __riscv_vle16_v_i16mf4, + __riscv_vwmaccus_vx_i16mf4, + __riscv_vse16_v_i16mf4, + vwmaccus_vx, + LOOP_DUAL_WIDEN_TERNARY_VX_BODY_X16) + +/* The fractional LMUL source has EMUL < 1, thus the widened destination + register group must not overlap the source at all. */ +/* { dg-final { scan-assembler-times {vwmaccus\.vx} 16 } } */ +/* { dg-final { scan-assembler-not {vwmaccus\.vx\s+(v[0-9]+),[a-z0-9]+,\1([^0-9]|$)} } } */
