[llvm] [ARM] Protect against multi use increment in gather scatter combine (PR #227247)
David Green via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 02:50:18 PDT 2026
https://github.com/davemgreen created https://github.com/llvm/llvm-project/pull/227247
None
>From d05a42259fd6999fafccd527d2c57c83264c0f87 Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Tue, 29 Sep 2026 09:09:06 +0100
Subject: [PATCH 1/2] [ARM] Add mve test for misbehaving gather combine
---
.../CodeGen/Thumb2/mve-gather-increment.ll | 109 ++++++++++++++----
1 file changed, 89 insertions(+), 20 deletions(-)
diff --git a/llvm/test/CodeGen/Thumb2/mve-gather-increment.ll b/llvm/test/CodeGen/Thumb2/mve-gather-increment.ll
index c43f76ea10444..73d74df8af24e 100644
--- a/llvm/test/CodeGen/Thumb2/mve-gather-increment.ll
+++ b/llvm/test/CodeGen/Thumb2/mve-gather-increment.ll
@@ -1590,24 +1590,93 @@ for.cond.cleanup: ; preds = %vector.body, %entry
ret void
}
+define void @_Z6gatherv() {
+; CHECK-LABEL: _Z6gatherv:
+; CHECK: @ %bb.0: @ %entry
+; CHECK-NEXT: .save {r7, lr}
+; CHECK-NEXT: push {r7, lr}
+; CHECK-NEXT: .pad #576
+; CHECK-NEXT: sub.w sp, sp, #576
+; CHECK-NEXT: adr r0, .LCPI20_0
+; CHECK-NEXT: mov.w lr, #30
+; CHECK-NEXT: vldrw.u32 q0, [r0]
+; CHECK-NEXT: add r0, sp, #96
+; CHECK-NEXT: movs r1, #4
+; CHECK-NEXT: movs r2, #1
+; CHECK-NEXT: mov r3, r0
+; CHECK-NEXT: .LBB20_1: @ %vector.body
+; CHECK-NEXT: @ =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: vadd.i32 q1, q0, r1
+; CHECK-NEXT: vadd.i32 q0, q0, r2
+; CHECK-NEXT: vcvt.f32.u32 q0, q0
+; CHECK-NEXT: vstrb.8 q0, [r3], #16
+; CHECK-NEXT: vmov q0, q1
+; CHECK-NEXT: le lr, .LBB20_1
+; CHECK-NEXT: @ %bb.2: @ %for.cond.cleanup
+; CHECK-NEXT: adr r2, .LCPI20_1
+; CHECK-NEXT: mov r1, sp
+; CHECK-NEXT: vldrw.u32 q0, [r2]
+; CHECK-NEXT: movs r2, #0
+; CHECK-NEXT: vadd.i32 q0, q0, r0
+; CHECK-NEXT: movs r0, #4
+; CHECK-NEXT: .LBB20_3: @ %for.cond6.preheader
+; CHECK-NEXT: @ =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: vdup.32 q1, r2
+; CHECK-NEXT: add r2, r0
+; CHECK-NEXT: vshl.i32 q1, q1, #2
+; CHECK-NEXT: vadd.i32 q1, q0, q1
+; CHECK-NEXT: vldrw.u32 q2, [q1]
+; CHECK-NEXT: vstrb.8 q2, [r1], #16
+; CHECK-NEXT: b .LBB20_3
+; CHECK-NEXT: .p2align 4
+; CHECK-NEXT: @ %bb.4:
+; CHECK-NEXT: .LCPI20_0:
+; CHECK-NEXT: .long 0 @ 0x0
+; CHECK-NEXT: .long 1 @ 0x1
+; CHECK-NEXT: .long 2 @ 0x2
+; CHECK-NEXT: .long 3 @ 0x3
+; CHECK-NEXT: .LCPI20_1:
+; CHECK-NEXT: .long 12 @ 0xc
+; CHECK-NEXT: .long 4 @ 0x4
+; CHECK-NEXT: .long 0 @ 0x0
+; CHECK-NEXT: .long 8 @ 0x8
+entry:
+ %input = alloca [120 x float], align 4
+ %out = alloca [24 x float], align 4
+ call void @llvm.lifetime.start.p0(ptr nonnull %input) #4
+ br label %vector.body
+
+vector.body: ; preds = %vector.body, %entry
+ %index = phi i32 [ 0, %entry ], [ %index.next, %vector.body ]
+ %vec.ind = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %entry ], [ %vec.ind.next, %vector.body ]
+ %0 = add nuw nsw <4 x i32> %vec.ind, splat (i32 1)
+ %1 = uitofp nneg <4 x i32> %0 to <4 x float>
+ %2 = getelementptr inbounds nuw [4 x i8], ptr %input, i32 %index
+ store <4 x float> %1, ptr %2, align 4
+ %index.next = add nuw i32 %index, 4
+ %vec.ind.next = add nuw nsw <4 x i32> %vec.ind, splat (i32 4)
+ %3 = icmp eq i32 %index.next, 120
+ br i1 %3, label %for.cond.cleanup, label %vector.body
-declare <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr>, i32, <2 x i1>, <2 x i32>)
-declare <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr>, i32, <4 x i1>, <4 x i32>)
-declare <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr>, i32, <8 x i1>, <8 x i32>)
-declare <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr>, i32, <16 x i1>, <16 x i32>)
-declare <2 x float> @llvm.masked.gather.v2f32.v2p0(<2 x ptr>, i32, <2 x i1>, <2 x float>)
-declare <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr>, i32, <4 x i1>, <4 x float>)
-declare <8 x float> @llvm.masked.gather.v8f32.v8p0(<8 x ptr>, i32, <8 x i1>, <8 x float>)
-declare <2 x i16> @llvm.masked.gather.v2i16.v2p0(<2 x ptr>, i32, <2 x i1>, <2 x i16>)
-declare <4 x i16> @llvm.masked.gather.v4i16.v4p0(<4 x ptr>, i32, <4 x i1>, <4 x i16>)
-declare <8 x i16> @llvm.masked.gather.v8i16.v8p0(<8 x ptr>, i32, <8 x i1>, <8 x i16>)
-declare <16 x i16> @llvm.masked.gather.v16i16.v16p0(<16 x ptr>, i32, <16 x i1>, <16 x i16>)
-declare <4 x half> @llvm.masked.gather.v4f16.v4p0(<4 x ptr>, i32, <4 x i1>, <4 x half>)
-declare <8 x half> @llvm.masked.gather.v8f16.v8p0(<8 x ptr>, i32, <8 x i1>, <8 x half>)
-declare <16 x half> @llvm.masked.gather.v16f16.v16p0(<16 x ptr>, i32, <16 x i1>, <16 x half>)
-declare <4 x i8> @llvm.masked.gather.v4i8.v4p0(<4 x ptr>, i32, <4 x i1>, <4 x i8>)
-declare <8 x i8> @llvm.masked.gather.v8i8.v8p0(<8 x ptr>, i32, <8 x i1>, <8 x i8>)
-declare <16 x i8> @llvm.masked.gather.v16i8.v16p0(<16 x ptr>, i32, <16 x i1>, <16 x i8>)
-declare <32 x i8> @llvm.masked.gather.v32i8.v32p0(<32 x ptr>, i32, <32 x i1>, <32 x i8>)
-declare void @llvm.masked.store.v4i32.p0(<4 x i32>, ptr, i32, <4 x i1>)
-declare <4 x i1> @llvm.get.active.lane.mask.v4i1.i32(i32, i32)
+for.cond.cleanup: ; preds = %vector.body
+ call void @llvm.lifetime.start.p0(ptr nonnull %out) #4
+ %invariant.gep = getelementptr [4 x i8], ptr %input, <4 x i32> <i32 3, i32 1, i32 0, i32 2>
+ br label %for.cond6.preheader
+
+for.cond6.preheader: ; preds = %for.cond.cleanup, %for.cond6.preheader
+ %outer.033 = phi i32 [ 0, %for.cond.cleanup ], [ %inc20, %for.cond6.preheader ]
+ %mul = shl nuw nsw i32 %outer.033, 2
+ %4 = getelementptr inbounds nuw [4 x i8], ptr %out, i32 %mul
+ %gep = getelementptr [4 x i8], <4 x ptr> %invariant.gep, i32 %mul
+ %wide.masked.gather = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 %gep, <4 x i1> splat (i1 true), <4 x float> poison)
+ store <4 x float> %wide.masked.gather, ptr %4, align 4
+ %inc20 = add nuw nsw i32 %outer.033, 1
+ %exitcond35.not = icmp eq i32 %inc20, 6
+ br i1 %exitcond35.not, label %for.cond.cleanup3, label %for.cond6.preheader
+
+for.cond.cleanup3: ; preds = %for.cond6.preheader
+ store i32 24, ptr %out
+ call void @llvm.lifetime.end.p0(ptr nonnull %out) #4
+ call void @llvm.lifetime.end.p0(ptr nonnull %input) #4
+ ret void
+}
>From 34c0ca1a947a8e0e8f848cde87167925d8209ce5 Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Tue, 29 Sep 2026 10:27:32 +0100
Subject: [PATCH 2/2] [ARM] Protect against multi use increment in gather
scatter combine
---
.../Target/ARM/MVEGatherScatterLowering.cpp | 10 +------
.../CodeGen/Thumb2/mve-gather-increment.ll | 28 +++++++++++--------
2 files changed, 17 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Target/ARM/MVEGatherScatterLowering.cpp b/llvm/lib/Target/ARM/MVEGatherScatterLowering.cpp
index ed95cba487e77..869ec4f2f2fba 100644
--- a/llvm/lib/Target/ARM/MVEGatherScatterLowering.cpp
+++ b/llvm/lib/Target/ARM/MVEGatherScatterLowering.cpp
@@ -1051,17 +1051,9 @@ bool MVEGatherScatterLowering::optimiseOffsets(Value *Offsets, BasicBlock *BB,
// If the phi is not used by anything else, we can just adapt it when
// replacing the instruction; if it is, we'll have to duplicate it
PHINode *NewPhi;
- if (Phi->hasNUses(2)) {
+ if (Phi->hasNUses(2) && IncInstruction->hasOneUse()) {
// No other users -> reuse existing phi (One user is the instruction
// we're looking at, the other is the phi increment)
- if (!IncInstruction->hasOneUse()) {
- // If the incrementing instruction does have more users than
- // our phi, we need to copy it
- IncInstruction = BinaryOperator::Create(
- Instruction::BinaryOps(IncInstruction->getOpcode()), Phi,
- IncrementPerRound, "LoopIncrement", IncInstruction->getIterator());
- Phi->setIncomingValue(IncrementingBlock, IncInstruction);
- }
NewPhi = Phi;
} else {
// There are other users -> create a new phi
diff --git a/llvm/test/CodeGen/Thumb2/mve-gather-increment.ll b/llvm/test/CodeGen/Thumb2/mve-gather-increment.ll
index 73d74df8af24e..2ba416a7dd113 100644
--- a/llvm/test/CodeGen/Thumb2/mve-gather-increment.ll
+++ b/llvm/test/CodeGen/Thumb2/mve-gather-increment.ll
@@ -1597,16 +1597,16 @@ define void @_Z6gatherv() {
; CHECK-NEXT: push {r7, lr}
; CHECK-NEXT: .pad #576
; CHECK-NEXT: sub.w sp, sp, #576
-; CHECK-NEXT: adr r0, .LCPI20_0
; CHECK-NEXT: mov.w lr, #30
+; CHECK-NEXT: adr r0, .LCPI20_0
+; CHECK-NEXT: add r1, sp, #96
; CHECK-NEXT: vldrw.u32 q0, [r0]
-; CHECK-NEXT: add r0, sp, #96
-; CHECK-NEXT: movs r1, #4
+; CHECK-NEXT: movs r0, #4
; CHECK-NEXT: movs r2, #1
-; CHECK-NEXT: mov r3, r0
+; CHECK-NEXT: mov r3, r1
; CHECK-NEXT: .LBB20_1: @ %vector.body
; CHECK-NEXT: @ =>This Inner Loop Header: Depth=1
-; CHECK-NEXT: vadd.i32 q1, q0, r1
+; CHECK-NEXT: vadd.i32 q1, q0, r0
; CHECK-NEXT: vadd.i32 q0, q0, r2
; CHECK-NEXT: vcvt.f32.u32 q0, q0
; CHECK-NEXT: vstrb.8 q0, [r3], #16
@@ -1614,22 +1614,26 @@ define void @_Z6gatherv() {
; CHECK-NEXT: le lr, .LBB20_1
; CHECK-NEXT: @ %bb.2: @ %for.cond.cleanup
; CHECK-NEXT: adr r2, .LCPI20_1
-; CHECK-NEXT: mov r1, sp
+; CHECK-NEXT: mov.w lr, #6
; CHECK-NEXT: vldrw.u32 q0, [r2]
+; CHECK-NEXT: mov r0, sp
; CHECK-NEXT: movs r2, #0
-; CHECK-NEXT: vadd.i32 q0, q0, r0
-; CHECK-NEXT: movs r0, #4
+; CHECK-NEXT: vadd.i32 q0, q0, r1
+; CHECK-NEXT: movs r1, #4
; CHECK-NEXT: .LBB20_3: @ %for.cond6.preheader
; CHECK-NEXT: @ =>This Inner Loop Header: Depth=1
; CHECK-NEXT: vdup.32 q1, r2
-; CHECK-NEXT: add r2, r0
+; CHECK-NEXT: add r2, r1
; CHECK-NEXT: vshl.i32 q1, q1, #2
; CHECK-NEXT: vadd.i32 q1, q0, q1
; CHECK-NEXT: vldrw.u32 q2, [q1]
-; CHECK-NEXT: vstrb.8 q2, [r1], #16
-; CHECK-NEXT: b .LBB20_3
+; CHECK-NEXT: vstrb.8 q2, [r0], #16
+; CHECK-NEXT: le lr, .LBB20_3
+; CHECK-NEXT: @ %bb.4: @ %for.cond.cleanup3
+; CHECK-NEXT: add.w sp, sp, #576
+; CHECK-NEXT: pop {r7, pc}
; CHECK-NEXT: .p2align 4
-; CHECK-NEXT: @ %bb.4:
+; CHECK-NEXT: @ %bb.5:
; CHECK-NEXT: .LCPI20_0:
; CHECK-NEXT: .long 0 @ 0x0
; CHECK-NEXT: .long 1 @ 0x1
More information about the llvm-commits
mailing list