[llvm] [X86] Prevent legalization cycles in lane-permute shuffle lowering (PR #224265)
Evgenii Kudriashov via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 17 06:25:29 PDT 2026
================
@@ -15833,32 +15831,36 @@ static SDValue lowerShuffleAsLanePermuteAndPermute(
// Avoid returning the same shuffle operation. For example,
// t7: v16i16 = vector_shuffle<8,9,10,11,4,5,6,7,0,1,2,3,12,13,14,15> t5,
// undef:v16i16
- if (CrossLaneMask == Mask || InLaneMask == Mask)
+ if (CrossLaneMask == Mask) {
+ // Trying another sublane size could bring us back to this shuffle,
+ // causing an infinite loop during legalization.
+ SameMask = true;
+ return SDValue();
+ }
+ if (InLaneMask == Mask)
return SDValue();
SDValue CrossLane = DAG.getVectorShuffle(VT, DL, V1, V2, CrossLaneMask);
return DAG.getVectorShuffle(VT, DL, CrossLane, DAG.getUNDEF(VT),
InLaneMask);
};
- // First attempt a solution with full lanes.
- if (SDValue V = getSublanePermute(/*NumSublanes=*/NumLanes))
- return V;
-
- // The rest of the solutions use sublanes.
- if (!CanUseSublanes)
- return SDValue();
-
- // Then attempt a solution with 64-bit sublanes (vpermq).
- if (SDValue V = getSublanePermute(/*NumSublanes=*/NumLanes * 2))
- return V;
-
- // If that doesn't work and we have fast variable cross-lane shuffle,
- // attempt 32-bit sublanes (vpermd).
- if (!Subtarget.hasFastVariableCrossLaneShuffle())
- return SDValue();
-
- return getSublanePermute(/*NumSublanes=*/NumLanes * 4);
+ // Try 128-bit lanes first. If that fails, try 64-bit and then 32-bit
+ // sublanes, if supported.
+ int MaxSublaneScale = 1;
+ if (CanUseSublanes)
+ MaxSublaneScale = Subtarget.hasFastVariableCrossLaneShuffle() ? 4 : 2;
+
+ for (int SublaneScale = 1; SublaneScale <= MaxSublaneScale;
+ SublaneScale *= 2) {
+ bool SameMask = false;
+ if (SDValue V = getSublanePermute(
+ /*NumSublanes=*/NumLanes * SublaneScale, SameMask))
+ return V;
+ if (SameMask)
+ return SDValue();
+ }
----------------
e-kud wrote:
```suggestion
bool SameMask = false;
for (int SublaneScale = 1; SublaneScale <= MaxSublaneScale && !SameMask;
SublaneScale *= 2) {
if (SDValue V = getSublanePermute(NumLanes * SublaneScale, SameMask))
return V;
}
```
I'm not sure if it is much better, just as an option.
https://github.com/llvm/llvm-project/pull/224265
More information about the llvm-commits
mailing list