[llvm] [X86] Prevent legalization cycles in lane-permute shuffle lowering (PR #224265)

via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 17 05:07:54 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-x86

Author: azwolski

<details>
<summary>Changes</summary>

When a cross-lane shuffle reproduces the original mask, trying a finer
sublane size can produce a decomposition that legalization transforms
back into the original shuffle. This causes legalization to cycle
indefinitely.

Stop trying finer sublane sizes when this condition is detected, and add
a regression test for the shuffle that previously caused `llc` to hang.

Fixes #<!-- -->224266

---
Full diff: https://github.com/llvm/llvm-project/pull/224265.diff


2 Files Affected:

- (modified) llvm/lib/Target/X86/X86ISelLowering.cpp (+24-22) 
- (added) llvm/test/CodeGen/X86/shuffle-lane-permute-cycle.ll (+17) 


``````````diff
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 64a4a191e5881..0acca62893752 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -15762,9 +15762,7 @@ static SDValue lowerShuffleAsLanePermuteAndPermute(
   /// Attempts to find a sublane permute with the given size
   /// that gets all elements into their target lanes.
   ///
-  /// If successful, fills CrossLaneMask and InLaneMask and returns true.
-  /// If unsuccessful, returns false and may overwrite InLaneMask.
-  auto getSublanePermute = [&](int NumSublanes) -> SDValue {
+  auto getSublanePermute = [&](int NumSublanes, bool &SameMask) -> SDValue {
     int NumSublanesPerLane = NumSublanes / NumLanes;
     int NumEltsPerSublane = NumElts / NumSublanes;
 
@@ -15833,7 +15831,13 @@ static SDValue lowerShuffleAsLanePermuteAndPermute(
     // Avoid returning the same shuffle operation. For example,
     // t7: v16i16 = vector_shuffle<8,9,10,11,4,5,6,7,0,1,2,3,12,13,14,15> t5,
     //                             undef:v16i16
-    if (CrossLaneMask == Mask || InLaneMask == Mask)
+    if (CrossLaneMask == Mask) {
+      // Trying another sublane size could bring us back to this shuffle,
+      // causing an infinite loop during legalization.
+      SameMask = true;
+      return SDValue();
+    }
+    if (InLaneMask == Mask)
       return SDValue();
 
     SDValue CrossLane = DAG.getVectorShuffle(VT, DL, V1, V2, CrossLaneMask);
@@ -15841,24 +15845,22 @@ static SDValue lowerShuffleAsLanePermuteAndPermute(
                                 InLaneMask);
   };
 
-  // First attempt a solution with full lanes.
-  if (SDValue V = getSublanePermute(/*NumSublanes=*/NumLanes))
-    return V;
-
-  // The rest of the solutions use sublanes.
-  if (!CanUseSublanes)
-    return SDValue();
-
-  // Then attempt a solution with 64-bit sublanes (vpermq).
-  if (SDValue V = getSublanePermute(/*NumSublanes=*/NumLanes * 2))
-    return V;
-
-  // If that doesn't work and we have fast variable cross-lane shuffle,
-  // attempt 32-bit sublanes (vpermd).
-  if (!Subtarget.hasFastVariableCrossLaneShuffle())
-    return SDValue();
-
-  return getSublanePermute(/*NumSublanes=*/NumLanes * 4);
+  // Try 128-bit lanes first. If that fails, try 64-bit and then 32-bit
+  // sublanes, if supported.
+  int MaxSublaneScale = 1;
+  if (CanUseSublanes)
+    MaxSublaneScale = Subtarget.hasFastVariableCrossLaneShuffle() ? 4 : 2;
+
+  for (int SublaneScale = 1; SublaneScale <= MaxSublaneScale;
+       SublaneScale *= 2) {
+    bool SameMask = false;
+    if (SDValue V = getSublanePermute(
+            /*NumSublanes=*/NumLanes * SublaneScale, SameMask))
+      return V;
+    if (SameMask)
+      return SDValue();
+  }
+  return SDValue();
 }
 
 /// Helper to get compute inlane shuffle mask for a complete shuffle mask.
diff --git a/llvm/test/CodeGen/X86/shuffle-lane-permute-cycle.ll b/llvm/test/CodeGen/X86/shuffle-lane-permute-cycle.ll
new file mode 100644
index 0000000000000..fb34f1edeeabc
--- /dev/null
+++ b/llvm/test/CodeGen/X86/shuffle-lane-permute-cycle.ll
@@ -0,0 +1,17 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mcpu=skylake-avx512 | FileCheck %s
+
+; This shuffle used to bounce between two masks, causing an infinite loop
+; during legalization.
+define <64 x i8> @shuffle_v64i8_lane_permute_cycle(<48 x i8> %a) {
+; CHECK-LABEL: shuffle_v64i8_lane_permute_cycle:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vmovdqa64 {{.*#+}} zmm1 = [0,1,1,2,3,4,4,5]
+; CHECK-NEXT:    vpermq %zmm0, %zmm1, %zmm0
+; CHECK-NEXT:    vpshufb {{.*#+}} zmm0 = zmm0[0,1,2,u,3,4,5,u,6,7,8,u,9,10,11,u,20,21,22,u,23,24,25,u,26,27,28,u,29,30,31,u,32,33,34,u,35,36,37,u,38,39,40,u,41,42,43,u,52,53,54,u,55,56,57,u,58,59,60,u,61,62,63,u]
+; CHECK-NEXT:    vpord {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm0, %zmm0
+; CHECK-NEXT:    retq
+  %or = or <48 x i8> %a, splat (i8 1)
+  %res = shufflevector <48 x i8> %or, <48 x i8> poison, <64 x i32> <i32 0, i32 1, i32 2, i32 48, i32 3, i32 4, i32 5, i32 51, i32 6, i32 7, i32 8, i32 54, i32 9, i32 10, i32 11, i32 57, i32 12, i32 13, i32 14, i32 60, i32 15, i32 16, i32 17, i32 63, i32 18, i32 19, i32 20, i32 66, i32 21, i32 22, i32 23, i32 69, i32 24, i32 25, i32 26, i32 72, i32 27, i32 28, i32 29, i32 75, i32 30, i32 31, i32 32, i32 78, i32 33, i32 34, i32 35, i32 81, i32 36, i32 37, i32 38, i32 84, i32 39, i32 40, i32 41, i32 87, i32 42, i32 43, i32 44, i32 90, i32 45, i32 46, i32 47, i32 93>
+  ret <64 x i8> %res
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/224265


More information about the llvm-commits mailing list