[llvm] [AArch64] Add SVE shuffle optimization pass (PR #193951)
via llvm-commits
llvm-commits at lists.llvm.org
Thu May 7 03:47:27 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-aarch64
Author: Graham Hunter (huntergr-arm)
<details>
<summary>Changes</summary>
Add a pass to perform VLA shuffle optimizations for SVE.
First up is using tbl to replace deinterleave4+uunpk+zext/uitofp.
---
Patch is 45.67 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/193951.diff
6 Files Affected:
- (modified) llvm/lib/Target/AArch64/AArch64.h (+2)
- (modified) llvm/lib/Target/AArch64/AArch64TargetMachine.cpp (+9)
- (modified) llvm/lib/Target/AArch64/CMakeLists.txt (+1)
- (added) llvm/lib/Target/AArch64/SVEShuffleOpts.cpp (+300)
- (modified) llvm/test/CodeGen/AArch64/O3-pipeline.ll (+2)
- (added) llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll (+602)
``````````diff
diff --git a/llvm/lib/Target/AArch64/AArch64.h b/llvm/lib/Target/AArch64/AArch64.h
index cb64d2d2b968a..9f645b252d2f6 100644
--- a/llvm/lib/Target/AArch64/AArch64.h
+++ b/llvm/lib/Target/AArch64/AArch64.h
@@ -74,6 +74,7 @@ FunctionPass *createSMEPeepholeOptPass();
FunctionPass *createMachineSMEABIPass(CodeGenOptLevel);
FunctionPass *createAArch64SRLTDefineSuperRegsPass();
ModulePass *createSVEIntrinsicOptsPass();
+FunctionPass *createSVEShuffleOptsPass();
InstructionSelector *
createAArch64InstructionSelector(const AArch64TargetMachine &,
const AArch64Subtarget &,
@@ -178,6 +179,7 @@ void initializeSMEPeepholeOptPass(PassRegistry &);
void initializeMachineSMEABIPass(PassRegistry &);
void initializeAArch64SRLTDefineSuperRegsPass(PassRegistry &);
void initializeSVEIntrinsicOptsPass(PassRegistry &);
+void initializeSVEShuffleOptsPass(PassRegistry &);
void initializeAArch64Arm64ECCallLoweringPass(PassRegistry &);
class AArch64StackTaggingPreRAPass
diff --git a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
index 226fc380a9244..f35b25201c592 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
@@ -229,6 +229,12 @@ static cl::opt<bool> EnableSRLTSubregToRegMitigation(
"super-regs when using Subreg Liveness Tracking"),
cl::init(true), cl::Hidden);
+static cl::opt<bool> EnableSVETblOpt(
+ "aarch64-sve-tbl-opt",
+ cl::desc("Enable the use of SVE tbls instructions to replace other kinds"
+ "shuffles, particularly combining them into one operation."),
+ cl::init(true), cl::Hidden);
+
extern "C" LLVM_ABI LLVM_EXTERNAL_VISIBILITY void
LLVMInitializeAArch64Target() {
// Register the target.
@@ -689,6 +695,9 @@ void AArch64PassConfig::addIRPasses() {
addPass(createAArch64StackTaggingPass(
/*IsOptNone=*/TM->getOptLevel() == CodeGenOptLevel::None));
+ if (getOptLevel() == CodeGenOptLevel::Aggressive && EnableSVETblOpt)
+ addPass(createSVEShuffleOptsPass());
+
// Match complex arithmetic patterns
if (TM->getOptLevel() >= CodeGenOptLevel::Default)
addPass(createComplexDeinterleavingPass(TM));
diff --git a/llvm/lib/Target/AArch64/CMakeLists.txt b/llvm/lib/Target/AArch64/CMakeLists.txt
index 80848845c2c24..e05b8aebd7132 100644
--- a/llvm/lib/Target/AArch64/CMakeLists.txt
+++ b/llvm/lib/Target/AArch64/CMakeLists.txt
@@ -90,6 +90,7 @@ add_llvm_target(AArch64CodeGen
AArch64TargetTransformInfo.cpp
SMEPeepholeOpt.cpp
SVEIntrinsicOpts.cpp
+ SVEShuffleOpts.cpp
MachineSMEABIPass.cpp
AArch64SRLTDefineSuperRegs.cpp
AArch64SIMDInstrOpt.cpp
diff --git a/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
new file mode 100644
index 0000000000000..ab90c6eebb3d8
--- /dev/null
+++ b/llvm/lib/Target/AArch64/SVEShuffleOpts.cpp
@@ -0,0 +1,300 @@
+//===------- SVEShuffleOpts - SVE Shuffle Optimization --------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// Tries to pattern match and combine scalable vector shuffles that could
+// be more efficiently performed by tbl instructions.
+//
+// An example would be a loop with 4 multiply-accumulate reductions, where the
+// new data in each vector iterations comes from a 4-way deinterleaving of
+// smaller datatypes loaded from memory, which are then extended and multiplied
+// by a common term loaded in reverse order from memory before being added to
+// the accumulator.
+//
+// If the initial load is a legal vector rather than 4x the size (generating a
+// structured ld4 instead), we would see multiple uunpkhi/lo instructions for
+// the extensions, followed by uzp1/2 instructions for the deinterleave, and rev
+// instructions for the common terms. Instead, we can replace all of those with
+// 4 tbl instructions. The tradeoff, of course, is that we now have 4 mask
+// values to maintain which increases register pressure.
+//
+// We should also be able to introduce new shuffles in order to balance out
+// SVE's bottom/top instruction pairs, which act on even/odd lanes instead of
+// the high or low half of a register.
+//
+//===----------------------------------------------------------------------===//
+
+#include "AArch64.h"
+#include "AArch64Subtarget.h"
+#include "AArch64TargetMachine.h"
+#include "Utils/AArch64BaseInfo.h"
+#include "llvm/ADT/SetVector.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/TargetTransformInfo.h"
+#include "llvm/CodeGen/TargetPassConfig.h"
+#include "llvm/CodeGen/TargetSubtargetInfo.h"
+#include "llvm/IR/Constants.h"
+#include "llvm/IR/IRBuilder.h"
+#include "llvm/IR/Instructions.h"
+#include "llvm/IR/IntrinsicInst.h"
+#include "llvm/IR/IntrinsicsAArch64.h"
+#include "llvm/IR/LLVMContext.h"
+#include "llvm/IR/Module.h"
+#include "llvm/IR/PassManager.h"
+#include "llvm/IR/PatternMatch.h"
+#include "llvm/InitializePasses.h"
+#include <optional>
+
+using namespace llvm;
+using namespace llvm::PatternMatch;
+
+#define DEBUG_TYPE "aarch64-sve-shuffle-opts"
+
+namespace {
+
+class SVEShuffleImpl {
+ const AArch64TargetMachine *TM = nullptr;
+ const LoopInfo *LI = nullptr;
+
+public:
+ SVEShuffleImpl() {};
+ SVEShuffleImpl(const AArch64TargetMachine *TM) : TM(TM) {};
+
+ PreservedAnalyses run(Function &F, FunctionAnalysisManager &FAM);
+ bool runOnFunction(Function &F, Pass &P);
+
+private:
+ bool processLoop(Loop &L);
+};
+
+struct SVEShuffleOpts : public FunctionPass {
+ SVEShuffleImpl Impl;
+ static char ID; // Pass identification, replacement for typeid
+ SVEShuffleOpts() : FunctionPass(ID) {}
+
+ bool runOnFunction(Function &F) override {
+ if (skipFunction(F))
+ return false;
+
+ return Impl.runOnFunction(F, *this);
+ }
+ void getAnalysisUsage(AnalysisUsage &AU) const override;
+
+ StringRef getPassName() const override { return "SVE Tbl Folding Opts"; }
+
+private:
+};
+} // end anonymous namespace
+
+/// A mapping between a vector_deinterleaveN intrinsic and extending cast
+/// instructions used on the resulting subvectors.
+using DeinterleaveMap =
+ SmallDenseMap<CallInst *, SmallVector<std::pair<CastInst *, unsigned>, 4>>;
+
+static void evaluateDeinterleave(IntrinsicInst *I, DeinterleaveMap &Candidates,
+ Loop &L) {
+ // TODO: 'Legalize' if the input is wider than nx128b but not wide enough
+ // to match a structured load?
+ if (I->getOperand(0)
+ ->getType()
+ ->getPrimitiveSizeInBits()
+ .getKnownMinValue() != AArch64::SVEBitsPerBlock)
+ return;
+
+ unsigned IntId = I->getIntrinsicID();
+ assert(IntId == Intrinsic::vector_deinterleave4 &&
+ "Only deinterleave4 supported currently");
+ SmallVector<std::pair<CastInst *, unsigned>, 4> Extends;
+ unsigned Opcode = 0;
+ Type *DestTy = nullptr;
+ for (User *U : I->users()) {
+ auto *Extract = dyn_cast<ExtractValueInst>(U);
+ // We expect only a single cast instruction as a user.
+ if (!Extract || Extract->getNumIndices() != 1)
+ return;
+
+ auto *Extend = dyn_cast<CastInst>(Extract->getUniqueUndroppableUser());
+ if (!Extend || (!isa<ZExtInst>(Extend) && !isa<UIToFPInst>(Extend)))
+ return;
+
+ // We're only interested if the uses are in the loop. This is almost
+ // certainly the case.
+ if (!L.contains(Extract) || !L.contains(Extend))
+ return;
+
+ Opcode = Extend->getOpcode();
+ DestTy = Extend->getDestTy();
+ Type *SrcTy = Extend->getSrcTy();
+
+ // For now, we only want to handle scalable vectors here.
+ if (!DestTy->isScalableTy())
+ return;
+
+ unsigned SrcBits = SrcTy->getScalarSizeInBits();
+ unsigned DestBits = DestTy->getScalarSizeInBits();
+
+ // Looking to match the deinterleave factor.
+ if (DestBits / SrcBits != 4)
+ return;
+
+ // We can't abuse the invalid index trick for tbls of bytes, since the
+ // largest possible SVE vector (2048b) would have 256 bytes, leaving no
+ // way of zeroing.
+ // TODO: If we know vscale is 8 or less, then we could use tbls for bytes.
+ if (SrcBits <= 8)
+ return;
+
+ Extends.push_back({Extend, Extract->getIndices()[0]});
+ }
+
+ // Check that all extracted values are being extended the same way, and that
+ // we have the expected number of extensions.
+ if (Extends.size() != 4 ||
+ !all_of(Extends, [&](std::pair<CastInst *, unsigned> Ext) {
+ CastInst *CI = Ext.first;
+ return CI->getDestTy() == DestTy && CI->getOpcode() == Opcode;
+ }))
+ return;
+
+ Candidates.insert({I, Extends});
+}
+
+// Optimize zext and uitofp from a 4-way deinterleaved load.
+static void optimizeSVEDeinterleavedExtends(DeinterleaveMap Deinterleaves) {
+ // TODO: Cache tbl patterns and reuse, and abandon transforms for a particular
+ // deinterleave if it would introduce too many. We probably want a
+ // hardcoded number of tbls to start with, but if we can estimate
+ // register pressure then we could make better decisions.
+ for (auto &[Deinterleave, Extends] : Deinterleaves) {
+ VectorType *DestTy = cast<VectorType>(Extends[0].first->getDestTy());
+ VectorType *SrcTy = cast<VectorType>(Extends[0].first->getSrcTy());
+ unsigned DstBits = DestTy->getScalarSizeInBits();
+ unsigned SrcBits = SrcTy->getScalarSizeInBits();
+ bool IsUIToFP = isa<UIToFPInst>(Extends[0].first);
+ VectorType *StepVecTy = VectorType::getInteger(DestTy);
+ Type *StepTy = StepVecTy->getScalarType();
+ Value *Input = Deinterleave->getOperand(0);
+ Type *InputTy = Input->getType();
+
+ APInt Invalid = APInt::getAllOnes(DstBits);
+ for (auto &[Extend, Idx] : Extends) {
+ // Build mask
+ APInt StartIdx = Invalid << SrcBits;
+ StartIdx += Idx;
+ IRBuilder<> Builder(Extend);
+ Value *StepVector = Builder.CreateStepVector(StepVecTy);
+ Value *ScaledSteps = Builder.CreateMul(
+ StepVector, Builder.CreateVectorSplat(StepVecTy->getElementCount(),
+ ConstantInt::get(StepTy, 4)));
+ Value *Start = ConstantInt::get(StepTy, StartIdx);
+ Value *ZextTbl = Builder.CreateAdd(
+ ScaledSteps,
+ Builder.CreateVectorSplat(StepVecTy->getElementCount(), Start));
+ Value *FinalMask = Builder.CreateBitCast(ZextTbl, InputTy);
+
+ // Replace the deinterleave, extractvalue, and extension chain with
+ // a tbl directly on the input value.
+ Value *Tbl = Builder.CreateIntrinsic(Intrinsic::aarch64_sve_tbl,
+ {InputTy}, {Input, FinalMask});
+ Value *Widen = Builder.CreateBitCast(Tbl, StepVecTy);
+ if (IsUIToFP)
+ Widen = Builder.CreateUIToFP(Widen, DestTy);
+ LLVM_DEBUG(dbgs() << "SVETBLOPT: Replaced " << *Extend << " with "
+ << *Widen << "\n");
+ Extend->replaceAllUsesWith(Widen);
+ Extend->eraseFromParent();
+ }
+ }
+}
+
+bool SVEShuffleImpl::processLoop(Loop &L) {
+ // TODO: Pull other shuffles into the tbl where possible.
+ // TODO: Add more advanced cases, such as introducing shuffles so that
+ // the SVE odd/even BT narrowing instructions can be used.
+ // TODO: Support other deinterleaves.
+ DeinterleaveMap Candidates;
+ for (auto *BB : L.blocks())
+ for (auto &I : *BB)
+ if (match(&I, m_Intrinsic<Intrinsic::vector_deinterleave4>(m_Value())))
+ evaluateDeinterleave(cast<IntrinsicInst>(&I), Candidates, L);
+
+ if (Candidates.empty())
+ return false;
+
+ optimizeSVEDeinterleavedExtends(Candidates);
+ return true;
+}
+
+void SVEShuffleOpts::getAnalysisUsage(AnalysisUsage &AU) const {
+ AU.addRequired<LoopInfoWrapperPass>();
+ AU.addRequired<TargetPassConfig>();
+ AU.setPreservesCFG();
+}
+
+char SVEShuffleOpts::ID = 0;
+static const char *name = "SVE VLA shuffle optimizations";
+INITIALIZE_PASS_BEGIN(SVEShuffleOpts, DEBUG_TYPE, name, false, false)
+INITIALIZE_PASS_DEPENDENCY(LoopInfoWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(TargetPassConfig)
+INITIALIZE_PASS_END(SVEShuffleOpts, DEBUG_TYPE, name, false, false)
+
+FunctionPass *llvm::createSVEShuffleOptsPass() { return new SVEShuffleOpts(); }
+
+namespace llvm {
+class SVEShuffleOptsPass : public PassInfoMixin<SVEShuffleOptsPass> {
+ const AArch64TargetMachine *TM;
+
+public:
+ explicit SVEShuffleOptsPass(const AArch64TargetMachine &TM) : TM(&TM) {}
+ PreservedAnalyses run(Function &F, FunctionAnalysisManager &FAM) {
+ SVEShuffleImpl Impl(TM);
+ return Impl.run(F, FAM);
+ }
+};
+} // end namespace llvm
+
+bool SVEShuffleImpl::runOnFunction(Function &F, Pass &P) {
+ // Make sure we can use SVE
+ TargetPassConfig &TPC = P.getAnalysis<TargetPassConfig>();
+ TM = &TPC.getTM<AArch64TargetMachine>();
+ const AArch64Subtarget *ST = TM->getSubtargetImpl(F);
+ if (!ST->isSVEorStreamingSVEAvailable())
+ return false;
+
+ LI = &P.getAnalysis<LoopInfoWrapperPass>().getLoopInfo();
+
+ bool Changed = false;
+ // Only looking to tranform innermost loops, given the increase in
+ // register usage.
+ for (Loop *L : LI->getLoopsInPreorder()) {
+ if (L->isInnermost())
+ Changed |= processLoop(*L);
+ }
+
+ return Changed;
+}
+
+PreservedAnalyses SVEShuffleImpl::run(Function &F,
+ FunctionAnalysisManager &FAM) {
+ const AArch64Subtarget *ST = TM->getSubtargetImpl(F);
+ if (!ST->isSVEorStreamingSVEAvailable())
+ return PreservedAnalyses::all();
+
+ LI = &FAM.getResult<LoopAnalysis>(F);
+
+ bool Changed = false;
+ // Only looking to tranform innermost loops, given the increase in
+ // register usage.
+ for (Loop *L : LI->getLoopsInPreorder()) {
+ if (L->isInnermost())
+ Changed |= processLoop(*L);
+ }
+
+ // Can we do better than 'none'?
+ // We're not actually using the new pass manager though.
+ return Changed ? PreservedAnalyses::none() : PreservedAnalyses::all();
+}
diff --git a/llvm/test/CodeGen/AArch64/O3-pipeline.ll b/llvm/test/CodeGen/AArch64/O3-pipeline.ll
index 451b79bfa42eb..85b55ae2e17e9 100644
--- a/llvm/test/CodeGen/AArch64/O3-pipeline.ll
+++ b/llvm/test/CodeGen/AArch64/O3-pipeline.ll
@@ -84,6 +84,8 @@
; CHECK-NEXT: Basic Alias Analysis (stateless AA impl)
; CHECK-NEXT: Function Alias Analysis Results
; CHECK-NEXT: AArch64 Stack Tagging
+; CHECK-NEXT: Natural Loop Information
+; CHECK-NEXT: SVE Tbl Folding Opts
; CHECK-NEXT: Complex Deinterleaving Pass
; CHECK-NEXT: Function Alias Analysis Results
; CHECK-NEXT: Memory SSA
diff --git a/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
new file mode 100644
index 0000000000000..e9de3f0b3e4d9
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-tbl-folding-opts.ll
@@ -0,0 +1,602 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -O3 -mtriple=aarch64-linux-gnu -mattr=+sve < %s | FileCheck %s
+
+define void @zext_nxv8i16_to_nxv8i64_deinterleave_in_loop(ptr %src, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: zext_nxv8i16_to_nxv8i64_deinterleave_in_loop:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: index z7.d, #0, #4
+; CHECK-NEXT: mov z4.d, #0xffffffffffff0000
+; CHECK-NEXT: mov z5.d, #0xffffffffffff0001
+; CHECK-NEXT: mov x8, #-65534 // =0xffffffffffff0002
+; CHECK-NEXT: mov z24.d, #0xffffffffffff0003
+; CHECK-NEXT: movi v0.2d, #0000000000000000
+; CHECK-NEXT: mov z6.d, x8
+; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: mov w8, #2048 // =0x800
+; CHECK-NEXT: movi v2.2d, #0000000000000000
+; CHECK-NEXT: movi v3.2d, #0000000000000000
+; CHECK-NEXT: cntd x9
+; CHECK-NEXT: add z4.d, z7.d, z4.d
+; CHECK-NEXT: add z5.d, z7.d, z5.d
+; CHECK-NEXT: rdvl x10, #1
+; CHECK-NEXT: add z6.d, z7.d, z6.d
+; CHECK-NEXT: add z7.d, z7.d, z24.d
+; CHECK-NEXT: .LBB0_1: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: ld1h { z24.h }, p0/z, [x0]
+; CHECK-NEXT: subs x8, x8, x9
+; CHECK-NEXT: add x0, x0, x10
+; CHECK-NEXT: tbl z25.h, { z24.h }, z4.h
+; CHECK-NEXT: tbl z26.h, { z24.h }, z5.h
+; CHECK-NEXT: tbl z27.h, { z24.h }, z6.h
+; CHECK-NEXT: tbl z24.h, { z24.h }, z7.h
+; CHECK-NEXT: add z0.d, z0.d, z25.d
+; CHECK-NEXT: add z1.d, z1.d, z26.d
+; CHECK-NEXT: add z2.d, z2.d, z27.d
+; CHECK-NEXT: add z3.d, z3.d, z24.d
+; CHECK-NEXT: b.ne .LBB0_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: str z0, [x1]
+; CHECK-NEXT: str z1, [x1, #1, mul vl]
+; CHECK-NEXT: str z2, [x1, #2, mul vl]
+; CHECK-NEXT: str z3, [x1, #3, mul vl]
+; CHECK-NEXT: ret
+entry:
+ %vscale = tail call i64 @llvm.vscale.i64()
+ %stride = shl nuw nsw i64 %vscale, 1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc.b.i64 = phi <vscale x 2 x i64> [ splat(i64 0), %entry ], [ %add.b.i64, %loop ]
+ %acc.g.i64 = phi <vscale x 2 x i64> [ splat(i64 0), %entry ], [ %add.g.i64, %loop ]
+ %acc.r.i64 = phi <vscale x 2 x i64> [ splat(i64 0), %entry ], [ %add.r.i64, %loop ]
+ %acc.a.i64 = phi <vscale x 2 x i64> [ splat(i64 0), %entry ], [ %add.a.i64, %loop ]
+ %src.gep = getelementptr inbounds nuw [4 x i16], ptr %src, i64 %iv
+ %bgra = call <vscale x 8 x i16> @llvm.masked.load(ptr %src.gep, <vscale x 8 x i1> %mask, <vscale x 8 x i16> zeroinitializer)
+ %deinterleave = tail call { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } @llvm.vector.deinterleave4(<vscale x 8 x i16> %bgra)
+ %b.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 0
+ %g.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 1
+ %r.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 2
+ %a.i16 = extractvalue { <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16>, <vscale x 2 x i16> } %deinterleave, 3
+ %b.i64 = zext <vscale x 2 x i16> %b.i16 to <vscale x 2 x i64>
+ %g.i64 = zext <vscale x 2 x i16> %g.i16 to <vscale x 2 x i64>
+ %r.i64 = zext <vscale x 2 x i16> %r.i16 to <vscale x 2 x i64>
+ %a.i64 = zext <vscale x 2 x i16> %a.i16 to <vscale x 2 x i64>
+ %add.b.i64 = add <vscale x 2 x i64> %acc.b.i64, %b.i64
+ %add.g.i64 = add <vscale x 2 x i64> %acc.g.i64, %g.i64
+ %add.r.i64 = add <vscale x 2 x i64> %acc.r.i64, %r.i64
+ %add.a.i64 = add <vscale x 2 x i64> %acc.a.i64, %a.i64
+ %iv.next = add nuw i64 %iv, %stride
+ %ec = icmp eq i64 %iv.next, 2048
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ store <vscale x 2 x i64> %add.b.i64, ptr %dst
+ %g.i64.gep = getelementptr <vscale x 2 x i64>, ptr %dst, i64 1
+ store <vscale x 2 x i64> %add.g.i64, ptr %g.i64.gep
+ %r.i64.gep = getelementptr <vscale x 2 x i64>, ptr %dst, i64 2
+ store <vscale x 2 x i64> %add.r.i64, ptr %r.i64.gep
+ %a.i64.gep = getelementptr <vscale x 2 x i64>, ptr %dst, i64 3
+ store <vscale x 2 x i64> %add.a.i64, ptr %a.i64.gep
+ ret void
+}
+
+;; TODO: Do we want to perform the sext equivalent? Requires a splat of the
+;; sign bits into another register (using asr) and a more complex tbl
+;; mask to choose; more instructions, but may still be worthwhile if
+;; we find cases in real code.
+define void @sext_nxv8i16_to_nxv8i64_deinterleave_in_loop(ptr %src, ptr %dst, <vscale x 8 x i1> %mask) #0 {
+; CHECK-LABEL: sext_nxv8i16_to_nxv8i64_deinterleave_in_loop:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: movi v0.2d, #0000000000000000
+; CHECK-NEXT: movi v1.2d, #0000000000000000
+; CHECK-NEXT: mov w8, #2048 // =0x800
+; CHECK-NEXT: movi v2.2d, #0000000000000000
+; CHECK-NEXT: movi v3.2d, #0000000000000000
+; CHECK-NEXT: cntd x9
+; CHECK-NEXT: ptrue p1.d
+; CHECK-NEXT: rdvl x10, #1
+; CHECK-NEXT: .LBB1_1: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: ld1h { z4.h }, p0/z, [x0]
+; CHECK-NEXT: subs x8, x8, x9
+; CHECK-NEXT: add x0, x0, x10
+; CHECK-NEXT: uunpkhi z5.s, z4...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/193951
More information about the llvm-commits
mailing list