https://github.com/wangpc-pp updated https://github.com/llvm/llvm-project/pull/209420
>From bf5ad3b324ff511ecc67ecf14c25cddabf9c7aa1 Mon Sep 17 00:00:00 2001 From: Pengcheng Wang <[email protected]> Date: Tue, 14 Jul 2026 17:56:49 +0800 Subject: [PATCH 1/4] clang-format Created using spr 1.3.6-beta.1 --- llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp index 9208762e8d1fe..0e69533f56a50 100644 --- a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp @@ -3736,8 +3736,8 @@ static bool isRegRegScaleLoadOrStore(SDNode *User, SDValue Add, return false; EVT VT = cast<MemSDNode>(User)->getMemoryVT(); if (!(VT.isScalarInteger() && - (Subtarget.hasStdExtZilx() || - Subtarget.hasVendorXTHeadMemIdx() || Subtarget.hasVendorXqcisls())) && + (Subtarget.hasStdExtZilx() || Subtarget.hasVendorXTHeadMemIdx() || + Subtarget.hasVendorXqcisls())) && !((VT == MVT::f32 || VT == MVT::f64) && Subtarget.hasVendorXTHeadFMemIdx())) return false; >From ddc4fe1cd521bff676fcf361ab67acc72ad1671f Mon Sep 17 00:00:00 2001 From: Pengcheng Wang <[email protected]> Date: Wed, 15 Jul 2026 14:49:45 +0800 Subject: [PATCH 2/4] Increase Complexity Created using spr 1.3.6-beta.1 --- llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td b/llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td index c7247efed0f4c..7c12cd63f99d0 100644 --- a/llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td +++ b/llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td @@ -86,10 +86,11 @@ def LXD_S_UW : IndexedLoad<0b101, 0b11, 0, "lxd.s.uw">, } // Predicates = [HasStdExtZilx, IsRV64] class AddrRegReg<int N> - : ComplexPattern<iPTR, 2, "SelectAddrRegRegFixedScale<"#N#">">; + : ComplexPattern<iPTR, 2, "SelectAddrRegRegFixedScale<"#N#">", + [], [], !add(10, N)>; class AddrRegZextReg<int N> : ComplexPattern<i64, 2, "SelectAddrRegZextRegFixedScale<"#N#", 32>", - [], [], 10>; + [], [], !add(20, N)>; def AddrRegReg0 : AddrRegReg<0>; def AddrRegReg1 : AddrRegReg<1>; >From 210e1a40980749a2f028c26f6fa863391f1e62be Mon Sep 17 00:00:00 2001 From: Pengcheng Wang <[email protected]> Date: Wed, 15 Jul 2026 14:58:14 +0800 Subject: [PATCH 3/4] Add more tests Created using spr 1.3.6-beta.1 --- llvm/test/CodeGen/RISCV/zilx.ll | 279 ++++++++++++++++++++++++++++++++ 1 file changed, 279 insertions(+) diff --git a/llvm/test/CodeGen/RISCV/zilx.ll b/llvm/test/CodeGen/RISCV/zilx.ll index 3a27d801230a7..5ec14ca429a95 100644 --- a/llvm/test/CodeGen/RISCV/zilx.ll +++ b/llvm/test/CodeGen/RISCV/zilx.ll @@ -968,3 +968,282 @@ define i64 @lxd_s_uw(ptr %a, i32 %b) { %3 = load i64, ptr %2, align 8 ret i64 %3 } + +;------------------------------------------------------------------------------ +; Load with a shift amount that doesn't match the access size +; +; The scaled forms require the index shift amount to match log2 of the access +; size. When the shift is smaller or larger than the access size the scaled +; forms can't be used: the shift is materialized separately, but the base+index +; add can still fold into the unscaled index load. +;------------------------------------------------------------------------------ + +; i8 access with the index scaled by 2 (larger than the access size). +define zeroext i8 @lxbu_shl_too_large(ptr %a, iXLen %b) { +; RV32-LABEL: lxbu_shl_too_large: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 2 +; RV32-NEXT: add a0, a0, a1 +; RV32-NEXT: lbu a0, 0(a0) +; RV32-NEXT: ret +; +; RV32-ZILX-LABEL: lxbu_shl_too_large: +; RV32-ZILX: # %bb.0: +; RV32-ZILX-NEXT: slli a1, a1, 2 +; RV32-ZILX-NEXT: lxbu a0, (a0), a1 +; RV32-ZILX-NEXT: ret +; +; RV64-LABEL: lxbu_shl_too_large: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 2 +; RV64-NEXT: add a0, a0, a1 +; RV64-NEXT: lbu a0, 0(a0) +; RV64-NEXT: ret +; +; RV64-ZILX-LABEL: lxbu_shl_too_large: +; RV64-ZILX: # %bb.0: +; RV64-ZILX-NEXT: slli a1, a1, 2 +; RV64-ZILX-NEXT: lxbu a0, (a0), a1 +; RV64-ZILX-NEXT: ret + %1 = getelementptr i32, ptr %a, iXLen %b + %2 = load i8, ptr %1, align 1 + ret i8 %2 +} + +; i16 access with the index scaled by 2 (larger than the access size). +define signext i16 @lxh_s_shl_too_large(ptr %a, iXLen %b) { +; RV32-LABEL: lxh_s_shl_too_large: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 2 +; RV32-NEXT: add a0, a0, a1 +; RV32-NEXT: lh a0, 0(a0) +; RV32-NEXT: ret +; +; RV32-ZILX-LABEL: lxh_s_shl_too_large: +; RV32-ZILX: # %bb.0: +; RV32-ZILX-NEXT: slli a1, a1, 2 +; RV32-ZILX-NEXT: lxh a0, (a0), a1 +; RV32-ZILX-NEXT: ret +; +; RV64-LABEL: lxh_s_shl_too_large: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 2 +; RV64-NEXT: add a0, a0, a1 +; RV64-NEXT: lh a0, 0(a0) +; RV64-NEXT: ret +; +; RV64-ZILX-LABEL: lxh_s_shl_too_large: +; RV64-ZILX: # %bb.0: +; RV64-ZILX-NEXT: slli a1, a1, 2 +; RV64-ZILX-NEXT: lxh a0, (a0), a1 +; RV64-ZILX-NEXT: ret + %1 = getelementptr i32, ptr %a, iXLen %b + %2 = load i16, ptr %1, align 2 + ret i16 %2 +} + +; i32 access with the index scaled by 1 (smaller than the access size). +define i32 @lxw_s_shl_too_small(ptr %a, iXLen %b) { +; RV32-LABEL: lxw_s_shl_too_small: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 1 +; RV32-NEXT: add a0, a0, a1 +; RV32-NEXT: lw a0, 0(a0) +; RV32-NEXT: ret +; +; RV32-ZILX-LABEL: lxw_s_shl_too_small: +; RV32-ZILX: # %bb.0: +; RV32-ZILX-NEXT: slli a1, a1, 1 +; RV32-ZILX-NEXT: lxw a0, (a0), a1 +; RV32-ZILX-NEXT: ret +; +; RV64-LABEL: lxw_s_shl_too_small: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 1 +; RV64-NEXT: add a0, a0, a1 +; RV64-NEXT: lw a0, 0(a0) +; RV64-NEXT: ret +; +; RV64-ZILX-LABEL: lxw_s_shl_too_small: +; RV64-ZILX: # %bb.0: +; RV64-ZILX-NEXT: slli a1, a1, 1 +; RV64-ZILX-NEXT: lxw a0, (a0), a1 +; RV64-ZILX-NEXT: ret + %1 = getelementptr i16, ptr %a, iXLen %b + %2 = load i32, ptr %1, align 4 + ret i32 %2 +} + +; i32 access with the index scaled by 3 (larger than the access size). +define i32 @lxw_s_shl_too_large(ptr %a, iXLen %b) { +; RV32-LABEL: lxw_s_shl_too_large: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 3 +; RV32-NEXT: add a0, a0, a1 +; RV32-NEXT: lw a0, 0(a0) +; RV32-NEXT: ret +; +; RV32-ZILX-LABEL: lxw_s_shl_too_large: +; RV32-ZILX: # %bb.0: +; RV32-ZILX-NEXT: slli a1, a1, 3 +; RV32-ZILX-NEXT: lxw a0, (a0), a1 +; RV32-ZILX-NEXT: ret +; +; RV64-LABEL: lxw_s_shl_too_large: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 3 +; RV64-NEXT: add a0, a0, a1 +; RV64-NEXT: lw a0, 0(a0) +; RV64-NEXT: ret +; +; RV64-ZILX-LABEL: lxw_s_shl_too_large: +; RV64-ZILX: # %bb.0: +; RV64-ZILX-NEXT: slli a1, a1, 3 +; RV64-ZILX-NEXT: lxw a0, (a0), a1 +; RV64-ZILX-NEXT: ret + %1 = getelementptr i64, ptr %a, iXLen %b + %2 = load i32, ptr %1, align 4 + ret i32 %2 +} + +; i64 access with the index scaled by 2 (smaller than the access size). +define i64 @lxd_s_shl_too_small(ptr %a, iXLen %b) { +; RV32-LABEL: lxd_s_shl_too_small: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 2 +; RV32-NEXT: add a1, a0, a1 +; RV32-NEXT: lw a0, 0(a1) +; RV32-NEXT: lw a1, 4(a1) +; RV32-NEXT: ret +; +; RV32-ZILX-LABEL: lxd_s_shl_too_small: +; RV32-ZILX: # %bb.0: +; RV32-ZILX-NEXT: addi a2, a0, 4 +; RV32-ZILX-NEXT: lxw.s a0, (a0), a1 +; RV32-ZILX-NEXT: lxw.s a1, (a2), a1 +; RV32-ZILX-NEXT: ret +; +; RV64-LABEL: lxd_s_shl_too_small: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 2 +; RV64-NEXT: add a0, a0, a1 +; RV64-NEXT: ld a0, 0(a0) +; RV64-NEXT: ret +; +; RV64-ZILX-LABEL: lxd_s_shl_too_small: +; RV64-ZILX: # %bb.0: +; RV64-ZILX-NEXT: slli a1, a1, 2 +; RV64-ZILX-NEXT: lxd a0, (a0), a1 +; RV64-ZILX-NEXT: ret + %1 = getelementptr i32, ptr %a, iXLen %b + %2 = load i64, ptr %1, align 8 + ret i64 %2 +} + +; i16 access with the zero-extended index scaled by 2 (larger than the access +; size). +define signext i16 @lxh_s_uw_shl_too_large(ptr %a, i32 %b) { +; RV32-LABEL: lxh_s_uw_shl_too_large: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 2 +; RV32-NEXT: add a0, a0, a1 +; RV32-NEXT: lh a0, 0(a0) +; RV32-NEXT: ret +; +; RV32-ZILX-LABEL: lxh_s_uw_shl_too_large: +; RV32-ZILX: # %bb.0: +; RV32-ZILX-NEXT: slli a1, a1, 2 +; RV32-ZILX-NEXT: lxh a0, (a0), a1 +; RV32-ZILX-NEXT: ret +; +; RV64-LABEL: lxh_s_uw_shl_too_large: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 32 +; RV64-NEXT: srli a1, a1, 30 +; RV64-NEXT: add a0, a0, a1 +; RV64-NEXT: lh a0, 0(a0) +; RV64-NEXT: ret +; +; RV64-ZILX-LABEL: lxh_s_uw_shl_too_large: +; RV64-ZILX: # %bb.0: +; RV64-ZILX-NEXT: slli a1, a1, 32 +; RV64-ZILX-NEXT: srli a1, a1, 30 +; RV64-ZILX-NEXT: lxh a0, (a0), a1 +; RV64-ZILX-NEXT: ret + %1 = zext i32 %b to i64 + %2 = getelementptr i32, ptr %a, i64 %1 + %3 = load i16, ptr %2, align 2 + ret i16 %3 +} + +; i32 access with the zero-extended index scaled by 1 (smaller than the access +; size). +define i32 @lxw_s_uw_shl_too_small(ptr %a, i32 %b) { +; RV32-LABEL: lxw_s_uw_shl_too_small: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 1 +; RV32-NEXT: add a0, a0, a1 +; RV32-NEXT: lw a0, 0(a0) +; RV32-NEXT: ret +; +; RV32-ZILX-LABEL: lxw_s_uw_shl_too_small: +; RV32-ZILX: # %bb.0: +; RV32-ZILX-NEXT: slli a1, a1, 1 +; RV32-ZILX-NEXT: lxw a0, (a0), a1 +; RV32-ZILX-NEXT: ret +; +; RV64-LABEL: lxw_s_uw_shl_too_small: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 32 +; RV64-NEXT: srli a1, a1, 31 +; RV64-NEXT: add a0, a0, a1 +; RV64-NEXT: lw a0, 0(a0) +; RV64-NEXT: ret +; +; RV64-ZILX-LABEL: lxw_s_uw_shl_too_small: +; RV64-ZILX: # %bb.0: +; RV64-ZILX-NEXT: slli a1, a1, 32 +; RV64-ZILX-NEXT: srli a1, a1, 31 +; RV64-ZILX-NEXT: lxw a0, (a0), a1 +; RV64-ZILX-NEXT: ret + %1 = zext i32 %b to i64 + %2 = getelementptr i16, ptr %a, i64 %1 + %3 = load i32, ptr %2, align 4 + ret i32 %3 +} + +; i32 access with the zero-extended index scaled by 3 (larger than the access +; size). +define i32 @lxw_s_uw_shl_too_large(ptr %a, i32 %b) { +; RV32-LABEL: lxw_s_uw_shl_too_large: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 3 +; RV32-NEXT: add a0, a0, a1 +; RV32-NEXT: lw a0, 0(a0) +; RV32-NEXT: ret +; +; RV32-ZILX-LABEL: lxw_s_uw_shl_too_large: +; RV32-ZILX: # %bb.0: +; RV32-ZILX-NEXT: slli a1, a1, 3 +; RV32-ZILX-NEXT: lxw a0, (a0), a1 +; RV32-ZILX-NEXT: ret +; +; RV64-LABEL: lxw_s_uw_shl_too_large: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 32 +; RV64-NEXT: srli a1, a1, 29 +; RV64-NEXT: add a0, a0, a1 +; RV64-NEXT: lw a0, 0(a0) +; RV64-NEXT: ret +; +; RV64-ZILX-LABEL: lxw_s_uw_shl_too_large: +; RV64-ZILX: # %bb.0: +; RV64-ZILX-NEXT: slli a1, a1, 32 +; RV64-ZILX-NEXT: srli a1, a1, 29 +; RV64-ZILX-NEXT: lxw a0, (a0), a1 +; RV64-ZILX-NEXT: ret + %1 = zext i32 %b to i64 + %2 = getelementptr i64, ptr %a, i64 %1 + %3 = load i32, ptr %2, align 4 + ret i32 %3 +} >From f092d03628360072a4000f1ab5e561d6e5697c3b Mon Sep 17 00:00:00 2001 From: Pengcheng Wang <[email protected]> Date: Thu, 16 Jul 2026 11:51:00 +0800 Subject: [PATCH 4/4] Add comments and change the calculation of complexities Created using spr 1.3.6-beta.1 --- llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td b/llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td index 900971d282b1b..eb9f124dcc5cd 100644 --- a/llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td +++ b/llvm/lib/Target/RISCV/RISCVInstrInfoZilx.td @@ -85,12 +85,25 @@ def LXD_S_UW : IndexedLoad<0b101, 0b11, 0, "lxd.s.uw">, Sched<[WriteLDD, ReadMemBase, ReadMemBase]>; } // Predicates = [HasStdExtZilx, IsRV64] +// The complexities below are chosen to give the indexed loads a strict +// priority over the plain reg+imm loads and to encode which form is the more +// specific match. Without an explicit complexity these patterns tie with the +// base L{B,H,W,D}[U] patterns (which also match a "reg + reg" address as +// base+0) and, being listed later, would never be selected. The required +// ordering is: +// base L* < scale-0 (unscaled) Zilx < scaled Zilx +// For a given load only the scale-0 form (which leaves the shift as a separate +// slli) and the exactly-matching scaled form can compete, so a scaled load +// only needs to outrank its own unscaled form; distinct non-zero scales never +// compete for the same node. class AddrRegReg<int N> : ComplexPattern<iPTR, 2, "SelectAddrRegRegFixedScale<"#N#">", - [], [], !add(10, N)>; + [], [], !if(!eq(N, 0), 10, 11)>; +// The zero-extended-index forms additionally fold the zext, so they are the +// most specific match and must outrank the plain AddrRegReg forms above. class AddrRegZextReg<int N> : ComplexPattern<i64, 2, "SelectAddrRegZextRegFixedScale<"#N#", 32>", - [], [], !add(20, N)>; + [], [], !if(!eq(N, 0), 20, 21)>; def AddrRegReg0 : AddrRegReg<0>; def AddrRegReg1 : AddrRegReg<1>; _______________________________________________ llvm-branch-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/llvm-branch-commits
