[clang] [llvm] [RISCV] Add a command line option to disable register overlap for vector indexed load. (PR #224739)

Craig Topper via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 15:58:06 PDT 2026


https://github.com/topperc updated https://github.com/llvm/llvm-project/pull/224739

>From e7d9098204fd16c1399152915147d8e3b08bad4c Mon Sep 17 00:00:00 2001
From: Craig Topper <craig.topper at sifive.com>
Date: Fri, 18 Sep 2026 13:22:10 -0700
Subject: [PATCH 1/4] [RISCV] Add a command line option to disable register
 overlap for vector index load.

If a trap occurs on an element other than 0, some SiFive cores will write a garbage value to that element and all active elements after it in the destination register before running the trap handler. This is normally allowed by the spec if restarting the instruction would overwrite the elements with the correct results.

For a vector indexed load with overlapping source and destination, writing the later elements with garbage corrupts some of the source indices. This prevents the instruction from restarting correctly once the fault has been handled.

This option allows users to opt out of overlapping registers on
these instructions.

This is an alternative to #220079.

I have not enabled this by default for -mcpu yet.
---
 clang/include/clang/Options/Options.td          |  4 ++++
 clang/lib/Driver/ToolChains/Arch/RISCV.cpp      |  8 ++++++++
 clang/test/Driver/riscv-features.c              |  5 +++++
 llvm/lib/Target/RISCV/RISCVFeatures.td          |  4 ++++
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp     | 14 ++++++++++++++
 llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td |  2 ++
 llvm/test/CodeGen/RISCV/features-info.ll        |  1 +
 7 files changed, 38 insertions(+)

diff --git a/clang/include/clang/Options/Options.td b/clang/include/clang/Options/Options.td
index c4c49df73c15a..4a6e77e204412 100644
--- a/clang/include/clang/Options/Options.td
+++ b/clang/include/clang/Options/Options.td
@@ -5855,6 +5855,10 @@ def mzilsd_word_align : Flag<["-"], "mzilsd-word-align">, Group<m_Group>,
   HelpText<"Allow Zilsd/Zclsd memory accesses to be 4-byte aligned (RISC-V only)">;
 def mzilsd_strict_align : Flag<["-"], "mzilsd-strict-align">, Group<m_Group>,
   HelpText<"Force all Zilsd/Zclsd memory accesses to be 8-byte aligned (RISC-V only)">;
+def mvector_index_load_overlap : Flag<["-"], "mvector-index-load-overlap">, Group<m_Group>,
+  HelpText<"Allow vector index load to have overlapping source and destination register groups (RISC-V only)">;
+def mno_vector_index_load_overlap : Flag<["-"], "mno-vector-index-load-overlap">, Group<m_Group>,
+  HelpText<"Force vector index load to have non-overlapping source and destination register groups (RISC-V only)">;
 def mno_thumb : Flag<["-"], "mno-thumb">, Group<m_arm_Features_Group>;
 def mrestrict_it: Flag<["-"], "mrestrict-it">, Group<m_arm_Features_Group>,
   HelpText<"Disallow generation of complex IT blocks. It is off by default.">;
diff --git a/clang/lib/Driver/ToolChains/Arch/RISCV.cpp b/clang/lib/Driver/ToolChains/Arch/RISCV.cpp
index 8e650ddf92dfc..2a55754c1aac3 100644
--- a/clang/lib/Driver/ToolChains/Arch/RISCV.cpp
+++ b/clang/lib/Driver/ToolChains/Arch/RISCV.cpp
@@ -178,6 +178,14 @@ void riscv::getRISCVTargetFeatures(const Driver &D, const llvm::Triple &Triple,
     Features.push_back("+unaligned-vector-mem");
   }
 
+  if (const Arg *A =
+          Args.getLastArg(options::OPT_mvector_index_load_overlap,
+                          options::OPT_mno_vector_index_load_overlap)) {
+    if (A->getOption().matches(options::OPT_mno_vector_index_load_overlap))
+      Features.push_back("+no-vector-index-load-overlap");
+    else
+      Features.push_back("-no-vector-index-load-overlap");
+  }
   if (Triple.isRISCV32()) {
     // Handle `-mzilsd-word-align` and `-mzilsd-strict-align` on rv32. These
     // interact with the scalar alignment options - if unaligned scalar memory
diff --git a/clang/test/Driver/riscv-features.c b/clang/test/Driver/riscv-features.c
index 97736ff81c799..287077c7cd18d 100644
--- a/clang/test/Driver/riscv-features.c
+++ b/clang/test/Driver/riscv-features.c
@@ -48,6 +48,11 @@
 // FAST-VECTOR-UNALIGNED-ACCESS: "-target-feature" "+unaligned-vector-mem"
 // NO-FAST-VECTOR-UNALIGNED-ACCESS: "-target-feature" "-unaligned-vector-mem"
 
+// RUN: %clang --target=riscv32-unknown-elf -### %s -mvector-index-load-overlap 2>&1 | FileCheck %s -check-prefix=VECTOR-INDEX-LOAD-OVERLAP
+// RUN: %clang --target=riscv32-unknown-elf -### %s -mno-vector-index-load-overlap 2>&1 | FileCheck %s -check-prefix=NO-VECTOR-INDEX-LOAD-OVERLAP
+// VECTOR-INDEX-LOAD-OVERLAP: "-target-feature" "-no-vector-index-load-overlap"
+// NO-VECTOR-INDEX-LOAD-OVERLAP: "-target-feature" "+no-vector-index-load-overlap"
+
 // RUN: %clang --target=riscv32-unknown-elf -### %s 2>&1 | FileCheck %s -check-prefix=NOUWTABLE
 // RUN: %clang --target=riscv32-unknown-elf -fasynchronous-unwind-tables -### %s 2>&1 | FileCheck %s -check-prefix=UWTABLE
 // RUN: %clang --target=riscv64-unknown-elf -### %s 2>&1 | FileCheck %s -check-prefix=NOUWTABLE
diff --git a/llvm/lib/Target/RISCV/RISCVFeatures.td b/llvm/lib/Target/RISCV/RISCVFeatures.td
index ca68611339ee1..4376721086bf8 100644
--- a/llvm/lib/Target/RISCV/RISCVFeatures.td
+++ b/llvm/lib/Target/RISCV/RISCVFeatures.td
@@ -2057,6 +2057,10 @@ def FeatureTaggedGlobals : SubtargetFeature<"tagged-globals",
     "true", "Use an instruction sequence for taking the address of a global "
     "that allows a memory tag in the upper address bits">;
 
+def FeatureNoVectorIndexLoadOverlap : SubtargetFeature<"no-vector-index-load-overlap",
+    "AllowVectorIndexLoadOverlap", "false",
+    "Disable overlapping source and destination for vector index loads">;
+
 //===----------------------------------------------------------------------===//
 // Tuning features
 //===----------------------------------------------------------------------===//
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index e749279168ab9..7d25baea79fd5 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -27003,6 +27003,20 @@ RISCVTargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
 
 void RISCVTargetLowering::AdjustInstrPostInstrSelection(MachineInstr &MI,
                                                         SDNode *Node) const {
+  if (!Subtarget.allowVectorIndexLoadOverlap()) {
+    const RISCVVPseudosTable::PseudoInfo *RVV =
+        RISCVVPseudosTable::getPseudoInfo(MI.getOpcode());
+    if (RVV && (RVV->BaseInstr == RISCV::VLUXEI8_V ||
+                RVV->BaseInstr == RISCV::VLUXEI16_V ||
+                RVV->BaseInstr == RISCV::VLUXEI32_V ||
+                RVV->BaseInstr == RISCV::VLUXEI64_V ||
+                RVV->BaseInstr == RISCV::VLOXEI8_V ||
+                RVV->BaseInstr == RISCV::VLOXEI16_V ||
+                RVV->BaseInstr == RISCV::VLOXEI32_V ||
+                RVV->BaseInstr == RISCV::VLOXEI64_V))
+      MI.getOperand(0).setIsEarlyClobber(true);
+  }
+
   // If instruction defines FRM operand, conservatively set it as non-dead to
   // express data dependency with FRM users and prevent incorrect instruction
   // reordering.
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td b/llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td
index c6241f582de67..e2fc58234e208 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td
@@ -936,6 +936,7 @@ class VPseudoILoadNoMask<VReg RetClass,
   let HasVecPolicyOp = 1;
   let Constraints = !if(!eq(EarlyClobber, 1), "@earlyclobber $rd, $rd = $passthru", "$rd = $passthru");
   let TargetOverlapConstraintType = TargetConstraintType;
+  let hasPostISelHook = 1;
 }
 
 class VPseudoILoadMask<VReg RetClass,
@@ -960,6 +961,7 @@ class VPseudoILoadMask<VReg RetClass,
   let HasVecPolicyOp = 1;
   let UsesMaskPolicy = 1;
   let IncludeInInversePseudoTable = 0;
+  let hasPostISelHook = 1;
 }
 
 class VPseudoUSStoreNoMask<VReg StClass,
diff --git a/llvm/test/CodeGen/RISCV/features-info.ll b/llvm/test/CodeGen/RISCV/features-info.ll
index 6d4a437595be8..7b33fc78cc605 100644
--- a/llvm/test/CodeGen/RISCV/features-info.ll
+++ b/llvm/test/CodeGen/RISCV/features-info.ll
@@ -102,6 +102,7 @@
 ; CHECK-NEXT:   no-default-unroll                - Disable default unroll preference.
 ; CHECK-NEXT:   no-sink-splat-operands           - Disable sink splat operands to enable .vx, .vf,.wx, and .wf instructions.
 ; CHECK-NEXT:   no-trailing-seq-cst-fence        - Disable trailing fence for seq-cst store.
+; CHECK-NEXT:   no-vector-index-load-overlap     - Disable overlapping source and destination for vector index loads.
 ; CHECK-NEXT:   optimized-nf2-segment-load-store - vlseg2eN.v and vsseg2eN.v are implemented as a wide memory op and shuffle.
 ; CHECK-NEXT:   optimized-nf3-segment-load-store - vlseg3eN.v and vsseg3eN.v are implemented as a wide memory op and shuffle.
 ; CHECK-NEXT:   optimized-nf4-segment-load-store - vlseg4eN.v and vsseg4eN.v are implemented as a wide memory op and shuffle.

>From a534d1d20fdb2da671f0ff4e7b83a1413bd19efd Mon Sep 17 00:00:00 2001
From: Craig Topper <craig.topper at sifive.com>
Date: Fri, 18 Sep 2026 15:10:51 -0700
Subject: [PATCH 2/4] fixup! Add tests

---
 .../RISCV/rvv/vluxei-vloxei-overlap.ll        | 58 +++++++++++++++++++
 1 file changed, 58 insertions(+)
 create mode 100644 llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll

diff --git a/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll b/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll
new file mode 100644
index 0000000000000..f8d945d1d5eaa
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll
@@ -0,0 +1,58 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: sed 's/iXLen/i32/g' %s | llc -mtriple=riscv32 -mattr=+v,+zvfhmin,+zvfbfmin \
+; RUN:   -verify-machineinstrs -target-abi=ilp32d | FileCheck %s --check-prefixes=OVERLAP
+; RUN: sed 's/iXLen/i64/g' %s | llc -mtriple=riscv64 -mattr=+v,+zvfhmin,+zvfbfmin \
+; RUN:   -verify-machineinstrs -target-abi=lp64d | FileCheck %s --check-prefixes=OVERLAP
+; RUN: sed 's/iXLen/i32/g' %s | llc -mtriple=riscv32 -mattr=+v,+zvfhmin,+zvfbfmin,+no-vector-index-load-overlap \
+; RUN:   -verify-machineinstrs -target-abi=ilp32d | FileCheck %s --check-prefixes=NOOVERLAP
+; RUN: sed 's/iXLen/i64/g' %s | llc -mtriple=riscv64 -mattr=+v,+zvfhmin,+zvfbfmin,+no-vector-index-load-overlap \
+; RUN:   -verify-machineinstrs -target-abi=lp64d | FileCheck %s --check-prefixes=NOOVERLAP
+
+define <vscale x 1 x i8> @intrinsic_vluxei_v_nxv1i8_nxv1i8_nxv1i32(ptr %0, <vscale x 1 x i32> %1, iXLen %2) nounwind {
+; OVERLAP-LABEL: intrinsic_vluxei_v_nxv1i8_nxv1i8_nxv1i32:
+; OVERLAP:       # %bb.0: # %entry
+; OVERLAP-NEXT:    vsetvli zero, a1, e8, mf8, ta, ma
+; OVERLAP-NEXT:    vluxei32.v v8, (a0), v8
+; OVERLAP-NEXT:    ret
+;
+; NOOVERLAP-LABEL: intrinsic_vluxei_v_nxv1i8_nxv1i8_nxv1i32:
+; NOOVERLAP:       # %bb.0: # %entry
+; NOOVERLAP-NEXT:    vsetvli zero, a1, e8, mf8, ta, ma
+; NOOVERLAP-NEXT:    vluxei32.v v9, (a0), v8
+; NOOVERLAP-NEXT:    vmv1r.v v8, v9
+; NOOVERLAP-NEXT:    ret
+entry:
+  %a = call <vscale x 1 x i8> @llvm.riscv.vluxei.nxv1i8.nxv1i32(
+    <vscale x 1 x i8> poison,
+    ptr %0,
+    <vscale x 1 x i32> %1,
+    iXLen %2)
+
+  ret <vscale x 1 x i8> %a
+}
+
+define <vscale x 1 x i8> @intrinsic_vloxei_v_nxv1i8_nxv1i8_nxv1i32(ptr %0, <vscale x 1 x i32> %1, iXLen %2) nounwind {
+; OVERLAP-LABEL: intrinsic_vloxei_v_nxv1i8_nxv1i8_nxv1i32:
+; OVERLAP:       # %bb.0: # %entry
+; OVERLAP-NEXT:    vsetvli zero, a1, e8, mf8, ta, ma
+; OVERLAP-NEXT:    vloxei32.v v8, (a0), v8
+; OVERLAP-NEXT:    ret
+;
+; NOOVERLAP-LABEL: intrinsic_vloxei_v_nxv1i8_nxv1i8_nxv1i32:
+; NOOVERLAP:       # %bb.0: # %entry
+; NOOVERLAP-NEXT:    vsetvli zero, a1, e8, mf8, ta, ma
+; NOOVERLAP-NEXT:    vloxei32.v v9, (a0), v8
+; NOOVERLAP-NEXT:    vmv1r.v v8, v9
+; NOOVERLAP-NEXT:    ret
+entry:
+  %a = call <vscale x 1 x i8> @llvm.riscv.vloxei.nxv1i8.nxv1i32(
+    <vscale x 1 x i8> poison,
+    ptr %0,
+    <vscale x 1 x i32> %1,
+    iXLen %2)
+
+  ret <vscale x 1 x i8> %a
+}
+
+declare <vscale x 1 x i8> @llvm.riscv.vluxei.nxv1i8.nxv1i32(<vscale x 1 x i8>, ptr, <vscale x 1 x i32>, iXLen)
+declare <vscale x 1 x i8> @llvm.riscv.vloxei.nxv1i8.nxv1i32(<vscale x 1 x i8>, ptr, <vscale x 1 x i32>, iXLen)

>From 59090ba91bc5989fa82069b677a460b937742a91 Mon Sep 17 00:00:00 2001
From: Craig Topper <craig.topper at sifive.com>
Date: Mon, 21 Sep 2026 12:32:24 -0700
Subject: [PATCH 3/4] fixup! Drop intrinsic declarations

---
 llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll | 3 ---
 1 file changed, 3 deletions(-)

diff --git a/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll b/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll
index f8d945d1d5eaa..dfe500f050bc6 100644
--- a/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll
@@ -53,6 +53,3 @@ entry:
 
   ret <vscale x 1 x i8> %a
 }
-
-declare <vscale x 1 x i8> @llvm.riscv.vluxei.nxv1i8.nxv1i32(<vscale x 1 x i8>, ptr, <vscale x 1 x i32>, iXLen)
-declare <vscale x 1 x i8> @llvm.riscv.vloxei.nxv1i8.nxv1i32(<vscale x 1 x i8>, ptr, <vscale x 1 x i32>, iXLen)

>From 3985e40bcf8fa5d572af197427757ece8a7d2658 Mon Sep 17 00:00:00 2001
From: Craig Topper <craig.topper at sifive.com>
Date: Mon, 21 Sep 2026 15:57:38 -0700
Subject: [PATCH 4/4] fixup! Use BoolOptionWithoutMarshalling.

---
 clang/include/clang/Options/Options.td | 9 +++++----
 1 file changed, 5 insertions(+), 4 deletions(-)

diff --git a/clang/include/clang/Options/Options.td b/clang/include/clang/Options/Options.td
index 4a6e77e204412..6e8c5f4226bbf 100644
--- a/clang/include/clang/Options/Options.td
+++ b/clang/include/clang/Options/Options.td
@@ -5855,10 +5855,11 @@ def mzilsd_word_align : Flag<["-"], "mzilsd-word-align">, Group<m_Group>,
   HelpText<"Allow Zilsd/Zclsd memory accesses to be 4-byte aligned (RISC-V only)">;
 def mzilsd_strict_align : Flag<["-"], "mzilsd-strict-align">, Group<m_Group>,
   HelpText<"Force all Zilsd/Zclsd memory accesses to be 8-byte aligned (RISC-V only)">;
-def mvector_index_load_overlap : Flag<["-"], "mvector-index-load-overlap">, Group<m_Group>,
-  HelpText<"Allow vector index load to have overlapping source and destination register groups (RISC-V only)">;
-def mno_vector_index_load_overlap : Flag<["-"], "mno-vector-index-load-overlap">, Group<m_Group>,
-  HelpText<"Force vector index load to have non-overlapping source and destination register groups (RISC-V only)">;
+defm vector_index_load_overlap : BoolOptionWithoutMarshalling<"m", "vector-index-load-overlap",
+  PosFlag<SetTrue, [], [], "Allow vector index load to have ">,
+  NegFlag<SetFalse, [], [], "Force vector index load to have non-">,
+  BothFlags<[], [], "overlapping source and destination register groups (RISC-V only)">>,
+  Group<m_Group>;
 def mno_thumb : Flag<["-"], "mno-thumb">, Group<m_arm_Features_Group>;
 def mrestrict_it: Flag<["-"], "mrestrict-it">, Group<m_arm_Features_Group>,
   HelpText<"Disallow generation of complex IT blocks. It is off by default.">;



More information about the llvm-commits mailing list