[llvm-branch-commits] [llvm] release/23.x: [AArch64][LV] Adjust costs for low-VF interleaved access (#209441) (PR #210696)

via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Mon Jul 20 05:02:11 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-aarch64

Author: llvmbot

<details>
<summary>Changes</summary>

Backport 79e05f4aff2e5b9f0513c0539d146e44cc80193c

Requested by: @<!-- -->MacDue

---
Full diff: https://github.com/llvm/llvm-project/pull/210696.diff


2 Files Affected:

- (modified) llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp (+16-7) 
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/sve-interleave-low-vf-cost.ll (+70-9) 


``````````diff
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 9e1a0960617a5..828bf2e3e8d0a 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5460,13 +5460,22 @@ InstructionCost AArch64TTIImpl::getInterleavedMemoryOpCost(
       }
 
       // llvm.vector.deinterleaveN is lowered as a binary tree of deinterleave2
-      // operations. A binary tree producing Factor leaf vectors has
-      // (Factor -1) inner deinterleave2 nodes. Each deinterleave2 on a pair of
-      // SVE registers emits one uzp1 + one uzp2.
-      // Total shuffle cost: (Factor - 1) deinterleave2 operations, each
-      // processing LT.first legal vector parts,with one uzp shuffle per part.
-      auto LT = getTypeLegalizationCost(VecTy);
-      return MemCost + (Factor - 1) * LT.first;
+      // operations. The tree has Log2(Factor) levels, with Factor UZP/ZIP
+      // operations at each level, giving a total shuffle cost of
+      // Factor * Log2(Factor).
+      auto SubVecCost = getTypeLegalizationCost(SubVecTy);
+      auto ResultCost = getTypeLegalizationCost(VecTy);
+      llvm::InstructionCost LegalizationCost = SubVecCost.first;
+
+      // FIXME: A temporary increase to the cost in cases where the input
+      // element type is 4x the output type. Otherwise it produces an SVE tail
+      // loop which is significantly larger than the NEON equivalent.
+      if (Opcode == Instruction::Store && Factor == 4 &&
+          SubVecCost.second.getScalarSizeInBits() ==
+              (4 * ResultCost.second.getScalarSizeInBits()))
+        LegalizationCost *= 4;
+
+      return MemCost + (Factor * LegalizationCost) + (Factor * Log2_64(Factor));
     }
   }
 
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-interleave-low-vf-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-interleave-low-vf-cost.ll
index 9e99bedb9ed0c..2bd9a42404a5b 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-interleave-low-vf-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-interleave-low-vf-cost.ll
@@ -5,17 +5,18 @@ target triple = "aarch64"
 
 ; Cost-model test for the path where the interleave factor is > the VF.
 ; The cost is modelled as a contiguous load of the wide vector plus a
-; vector.deinterleave4 lowered as a binary tree of (Factor - 1) deinterleave2 shuffles:
+; vector.deinterleave4 lowered as a binary tree of Log2(Factor) levels,
+; each with Factor operations.
 ;
-;   cost = MemCost + (Factor - 1) * LT.first
-;  * VF vscale x 2: wide type <vscale x 8 x i16>, LT.first = 1
-;       => 1 (load) + 3 * 1 (shuffles) = 4
-;  * VF vscale x 4: wide type <vscale x 16 x i16> LT.first = 2
-;       => 2 (loads) + 3 * 2 (shuffles) = 8
+;   cost = MemCost + (Factor * LegalizationCost) + (Factor * Log2(Factor))
+;  * VF vscale x 2: wide type <vscale x 8 x i16>, MemCost = 1, Subvector Legalization Cost = 1
+;       => 1 + (4 * 1) + (4 * 2) = 13
+;  * VF vscale x 4: wide type <vscale x 16 x i16>, MemCost = 2, Subvector Legalization Cost = 1
+;       => 2 + (4 * 1) + (4 * 2) = 14
 
 ; CHECK-LABEL: LV: Checking a loop in 'deinterleave4_nxv2i16_load'
-; CHECK: Cost of 4 for VF vscale x 2: INTERLEAVE-GROUP with factor 4, ir<%ptr.b>
-; CHECK: Cost of 8 for VF vscale x 4: INTERLEAVE-GROUP with factor 4, ir<%ptr.b>
+; CHECK: Cost of 13 for VF vscale x 2: INTERLEAVE-GROUP with factor 4, ir<%ptr.b>
+; CHECK: Cost of 14 for VF vscale x 4: INTERLEAVE-GROUP with factor 4, ir<%ptr.b>
 ; CHECK: LV: Selecting VF: vscale x 2
 define void @deinterleave4_nxv2i16_load(ptr noalias readonly %src, ptr noalias %out, i64 %n) #0 {
 entry:
@@ -69,4 +70,64 @@ exit:
   ret void
 }
 
-attributes #0 = { "target-features"="+sve" }
\ No newline at end of file
+; Check that the increased low-VF interleaved-store cost prevents selection of
+; an SVE epilogue.
+
+; For VF vscale x 4:
+;   load cost  = 2 + (4 * 1) + (4 * 2) = 14
+;   store cost = 1 + (4 * 4) + (4 * 2) = 25
+;
+; This makes the fixed VF 8 epilogue preferable to VF vscale x 4.
+;
+; CHECK-LABEL: LV: Checking a loop in 'deinterleave4_nxv4i16_load_interleave4_nxv4i8_store'
+; CHECK: Cost of 14 for VF vscale x 4: INTERLEAVE-GROUP with factor 4
+; CHECK: Cost of 25 for VF vscale x 4: INTERLEAVE-GROUP with factor 4
+; CHECK: LV: Selecting VF: vscale x 16
+; CHECK: LEV: Vectorizing epilogue loop with VF = 8
+define void @deinterleave4_nxv4i16_load_interleave4_nxv4i8_store(
+    ptr readonly %src, ptr writeonly %out, i32 %n) #0 {
+entry:
+  %empty = icmp eq i32 %n, 0
+  br i1 %empty, label %exit, label %loop
+
+loop:
+  %src.iv = phi ptr [ %src.next, %loop ], [ %src, %entry ]
+  %out.iv = phi ptr [ %out.next, %loop ], [ %out, %entry ]
+  %iv = phi i32 [ %iv.next, %loop ], [ %n, %entry ]
+
+  %ptr.g = getelementptr inbounds i16, ptr %src.iv, i64 1
+  %ptr.r = getelementptr inbounds i16, ptr %src.iv, i64 2
+  %ptr.a = getelementptr inbounds i16, ptr %src.iv, i64 3
+  %load.b = load i16, ptr %src.iv, align 2
+  %load.g = load i16, ptr %ptr.g, align 2
+  %load.r = load i16, ptr %ptr.r, align 2
+  %load.a = load i16, ptr %ptr.a, align 2
+
+  %shift.b = lshr i16 %load.b, 8
+  %shift.g = lshr i16 %load.g, 8
+  %shift.r = lshr i16 %load.r, 8
+  %shift.a = lshr i16 %load.a, 8
+  %trunc.b = trunc nuw i16 %shift.b to i8
+  %trunc.g = trunc nuw i16 %shift.g to i8
+  %trunc.r = trunc nuw i16 %shift.r to i8
+  %trunc.a = trunc nuw i16 %shift.a to i8
+
+  %out.g = getelementptr inbounds i8, ptr %out.iv, i64 1
+  %out.r = getelementptr inbounds i8, ptr %out.iv, i64 2
+  %out.a = getelementptr inbounds i8, ptr %out.iv, i64 3
+  store i8 %trunc.b, ptr %out.iv, align 1
+  store i8 %trunc.g, ptr %out.g, align 1
+  store i8 %trunc.r, ptr %out.r, align 1
+  store i8 %trunc.a, ptr %out.a, align 1
+
+  %src.next = getelementptr inbounds i16, ptr %src.iv, i64 4
+  %out.next = getelementptr inbounds i8, ptr %out.iv, i64 4
+  %iv.next = add nsw i32 %iv, -1
+  %done = icmp eq i32 %iv.next, 0
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+attributes #0 = { "target-features"="+sve" }

``````````

</details>


https://github.com/llvm/llvm-project/pull/210696


More information about the llvm-branch-commits mailing list