[llvm] [SPIRV] Simplify 64 bit popcount for Vulkan (PR #219605)

Justin Bogner via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 1 22:05:58 PDT 2026


https://github.com/bogner updated https://github.com/llvm/llvm-project/pull/219605

>From 6ca83288ed3f37f8ff75311d9b45f49d04dd147b Mon Sep 17 00:00:00 2001
From: Justin Bogner <mail at justinbogner.com>
Date: Fri, 28 Aug 2026 16:23:27 -0700
Subject: [PATCH 1/3] [SPIRV] Simplify 64 bit popcount for Vulkan

Instead of converting types using a series of vector shuffles, just use
trunc and right shift. This is simpler to implement and makes the
resulting logic easier to follow.
---
 .../Target/SPIRV/SPIRVInstructionSelector.cpp | 124 ++++--------------
 .../CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll |  80 +++++------
 2 files changed, 60 insertions(+), 144 deletions(-)

diff --git a/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp b/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
index 891a8d9da12cf..498bbb31bba68 100644
--- a/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
+++ b/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
@@ -61,13 +61,6 @@ struct ImageOperands {
   std::optional<Register> Compare;
 };
 
-struct SplitParts {
-  SPIRVTypeInst Type = nullptr;
-  Register High;
-  Register Low;
-  bool IsScalar = false;
-};
-
 llvm::SPIRV::SelectionControl::SelectionControl
 getSelectionOperandForImm(int Imm) {
   if (Imm == 2)
@@ -544,11 +537,6 @@ class SPIRVInstructionSelector : public InstructionSelector {
   void decorateUsesAsNonUniform(Register &NonUniformReg) const;
   bool errorIfInstrOutsideShader(MachineInstr &I) const;
 
-  std::optional<SplitParts> splitEvenOddLanes(Register PopCountReg,
-                                              unsigned ComponentCount,
-                                              MachineInstr &I,
-                                              SPIRVTypeInst I32Type) const;
-
   bool
   handle64BitOverflow(Register ResVReg, SPIRVTypeInst ResType, MachineInstr &I,
                       Register SrcReg, unsigned int Opcode,
@@ -1794,66 +1782,6 @@ bool SPIRVInstructionSelector::selectOpWithSrcs(Register ResVReg,
   return true;
 }
 
-std::optional<SplitParts> SPIRVInstructionSelector::splitEvenOddLanes(
-    Register PopCountReg, unsigned ComponentCount, MachineInstr &I,
-    SPIRVTypeInst I32Type) const {
-  SplitParts Parts;
-
-  if (ComponentCount == 1) {
-    // ---- Scalar path: extract element 1 (high word) and element 0 (low word)
-    // ----
-    Parts.IsScalar = true;
-    Parts.Type = I32Type;
-    Parts.High = MRI->createVirtualRegister(GR.getRegClass(I32Type));
-    Parts.Low = MRI->createVirtualRegister(GR.getRegClass(I32Type));
-
-    bool ZeroAsNull = !STI.isShader();
-    Register IdxZero = GR.getOrCreateConstInt(0, I, I32Type, TII, ZeroAsNull);
-    Register IdxOne = GR.getOrCreateConstInt(1, I, I32Type, TII, ZeroAsNull);
-
-    if (!selectOpWithSrcs(Parts.High, I32Type, I, {PopCountReg, IdxOne},
-                          SPIRV::OpVectorExtractDynamic))
-      return std::nullopt;
-
-    if (!selectOpWithSrcs(Parts.Low, I32Type, I, {PopCountReg, IdxZero},
-                          SPIRV::OpVectorExtractDynamic))
-      return std::nullopt;
-
-  } else {
-    // ---- Vector path: shuffle odd lanes → High, even lanes → Low ----
-    MachineIRBuilder MIRBuilder(I);
-    Parts.IsScalar = false;
-    Parts.Type = GR.getOrCreateSPIRVVectorType(I32Type, ComponentCount,
-                                               MIRBuilder, /*IsSigned=*/false);
-    Parts.High = MRI->createVirtualRegister(GR.getRegClass(Parts.Type));
-    Parts.Low = MRI->createVirtualRegister(GR.getRegClass(Parts.Type));
-
-    // High = odd-indexed elements (1, 3, 5, …) — the upper 32-bit halves.
-    auto MIB = BuildMI(*I.getParent(), I, I.getDebugLoc(),
-                       TII.get(SPIRV::OpVectorShuffle))
-                   .addDef(Parts.High)
-                   .addUse(GR.getSPIRVTypeID(Parts.Type))
-                   .addUse(PopCountReg)
-                   .addUse(PopCountReg);
-    for (unsigned J = 1; J < ComponentCount * 2; J += 2)
-      MIB.addImm(J);
-    MIB.constrainAllUses(TII, TRI, RBI);
-
-    // Low = even-indexed elements (0, 2, 4, …) — the lower 32-bit halves.
-    MIB = BuildMI(*I.getParent(), I, I.getDebugLoc(),
-                  TII.get(SPIRV::OpVectorShuffle))
-              .addDef(Parts.Low)
-              .addUse(GR.getSPIRVTypeID(Parts.Type))
-              .addUse(PopCountReg)
-              .addUse(PopCountReg);
-    for (unsigned J = 0; J < ComponentCount * 2; J += 2)
-      MIB.addImm(J);
-    MIB.constrainAllUses(TII, TRI, RBI);
-  }
-
-  return Parts;
-}
-
 bool SPIRVInstructionSelector::selectPopCount16(Register ResVReg,
                                                 SPIRVTypeInst ResType,
                                                 MachineInstr &I,
@@ -1894,43 +1822,47 @@ bool SPIRVInstructionSelector::selectPopCount64(Register ResVReg,
                                                 MachineInstr &I,
                                                 Register SrcReg,
                                                 unsigned Opcode) const {
-  unsigned ComponentCount = GR.getScalarOrVectorComponentCount(ResType);
-  if (ComponentCount > 2)
-    return handle64BitOverflow(
-        ResVReg, ResType, I, SrcReg, Opcode,
-        [this](Register R, SPIRVTypeInst T, MachineInstr &I, Register S,
-               unsigned O) { return this->selectPopCount64(R, T, I, S, O); });
-
   MachineIRBuilder MIRBuilder(I);
 
-  // ---- Types ----
+  SPIRVTypeInst SrcType = GR.getSPIRVTypeForVReg(SrcReg);
+  unsigned ComponentCount = GR.getScalarOrVectorComponentCount(SrcType);
   SPIRVTypeInst I32Type = GR.getOrCreateSPIRVIntegerType(32, MIRBuilder);
   SPIRVTypeInst VecI32Type = GR.getOrCreateSPIRVVectorType(
-      I32Type, 2 * ComponentCount, MIRBuilder, /*IsSigned=*/false);
+      I32Type, ComponentCount, MIRBuilder, /*IsSigned=*/false);
 
-  // Converts 64 bit into and array of 32 bit, containing 2 elements.
-  Register Vec32 = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
-  if (!selectOpWithSrcs(Vec32, VecI32Type, I, {SrcReg}, SPIRV::OpBitcast))
+  // Truncate and count the low bits.
+  Register Trunc = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
+  if (!selectOpWithSrcs(Trunc, VecI32Type, I, {SrcReg}, SPIRV::OpUConvert))
+    return false;
+
+  Register LowCount = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
+  if (!selectOpWithSrcs(LowCount, VecI32Type, I, {Trunc}, SPIRV::OpBitCount))
+    return false;
+
+  // Shift the high bits over and count them too.
+  Register ShiftAmount =
+      ComponentCount == 1 ? GR.getOrCreateConstInt(32, I, SrcType, TII)
+                          : GR.getOrCreateConstVector(32, I, SrcType, TII);
+  unsigned ShiftOp = ComponentCount == 1 ? SPIRV::OpShiftRightLogicalS
+                                         : SPIRV::OpShiftRightLogicalV;
+  Register Shift = MRI->createVirtualRegister(GR.getRegClass(SrcType));
+  if (!selectOpWithSrcs(Shift, SrcType, I, {SrcReg, ShiftAmount}, ShiftOp))
     return false;
 
-  // Apply popcount on each 32 bit lane
-  Register Pop32 = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
-  if (!selectPopCount32(Pop32, VecI32Type, I, Vec32, Opcode))
+  Trunc = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
+  if (!selectOpWithSrcs(Trunc, VecI32Type, I, {Shift}, SPIRV::OpUConvert))
     return false;
 
-  // Splits result into highbit lane and lowbit lane
-  auto MaybeParts = splitEvenOddLanes(Pop32, ComponentCount, I, I32Type);
-  if (!MaybeParts)
+  Register HighCount = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
+  if (!selectOpWithSrcs(HighCount, VecI32Type, I, {Trunc}, SPIRV::OpBitCount))
     return false;
-  SplitParts &Parts = *MaybeParts;
 
-  // Sum high part and low part
-  unsigned OpAdd = Parts.IsScalar ? SPIRV::OpIAddS : SPIRV::OpIAddV;
-  Register Sum = MRI->createVirtualRegister(GR.getRegClass(Parts.Type));
-  if (!selectOpWithSrcs(Sum, Parts.Type, I, {Parts.High, Parts.Low}, OpAdd))
+  // Add them up and zext or sext back to 64 bit values.
+  Register Sum = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
+  if (!selectOpWithSrcs(Sum, VecI32Type, I, {HighCount, LowCount},
+                        ComponentCount == 1 ? SPIRV::OpIAddS : SPIRV::OpIAddV))
     return false;
 
-  // Convert 32 bit sum into 64 bit scalar
   bool IsSigned = GR.isScalarOrVectorSigned(ResType);
   unsigned ConvOp = IsSigned ? SPIRV::OpSConvert : SPIRV::OpUConvert;
   return selectOpWithSrcs(ResVReg, ResType, I, {Sum}, ConvOp);
diff --git a/llvm/test/CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll b/llvm/test/CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll
index 4b739f7e14790..066d6151ff0d4 100644
--- a/llvm/test/CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll
+++ b/llvm/test/CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll
@@ -14,9 +14,10 @@
 ; CHECK-DAG: [[i64x4_t:%.+]] = OpTypeVector [[i64_t]] 4
 ; CHECK-DAG: [[i16x3_t:%.+]] = OpTypeVector [[i16_t]] 3
 
-; CHECK-DAG: [[zero:%.*]] = OpConstant [[i32_t]] 0
-; CHECK-DAG: [[one:%.*]] = OpConstant [[i32_t]] 1
-; CHECK-DAG: [[two:%.*]] = OpConstant [[i64_t]] 2
+; CHECK-DAG: [[i64_32:%.+]] = OpConstant [[i64_t]] 32
+; CHECK-DAG: [[i64x2_32:%.+]] = OpConstantComposite [[i64x2_t]] [[i64_32]] [[i64_32]]
+; CHECK-DAG: [[i64x3_32:%.+]] = OpConstantComposite [[i64x3_t]] [[i64_32]] [[i64_32]] [[i64_32]]
+; CHECK-DAG: [[i64x4_32:%.+]] = OpConstantComposite [[i64x4_t]] [[i64_32]] [[i64_32]] [[i64_32]] [[i64_32]]
 
 ; CHECK-LABEL:  ; -- Begin function test
 
@@ -44,60 +45,43 @@
 ; CHECK: [[p32_bitcount:%.+]] = OpBitCount [[i32_t]] [[p32]]
 
 ; p64
-; CHECK: [[p64_bitcast:%.+]] = OpBitcast [[i32x2_t]] [[p64]]
-; CHECK: [[p64_bitcount:%.+]] = OpBitCount [[i32x2_t]] [[p64_bitcast]]
-; CHECK: [[index_one:%.+]] = OpVectorExtractDynamic [[i32_t]] [[p64_bitcount]] [[one]]
-; CHECK: [[index_zero:%.+]] = OpVectorExtractDynamic [[i32_t]] [[p64_bitcount]] [[zero]]
-; CHECK: [[add:%.+]] = OpIAdd [[i32_t]] [[index_one]] [[index_zero]]
-; CHECK: [[#]] = OpUConvert [[i64_t]] [[add]]
+; CHECK: [[p64_trunc_low:%.+]] = OpUConvert [[i32_t]] [[p64]]
+; CHECK: [[p64_low:%.+]] = OpBitCount [[i32_t]] [[p64_trunc_low]]
+; CHECK: [[p64_shift_high:%.+]] = OpShiftRightLogical [[i64_t]] [[p64]] [[i64_32]]
+; CHECK: [[p64_trunc_high:%.+]] = OpUConvert [[i32_t]] [[p64_shift_high]]
+; CHECK: [[p64_high:%.+]] = OpBitCount [[i32_t]] [[p64_trunc_high]]
+; CHECK: [[p64_sum:%.+]] = OpIAdd [[i32_t]] [[p64_high]] [[p64_low]]
+; CHECK: %[[#]] = OpUConvert [[i64_t]] [[p64_sum]]
 
 ; p32x2
 ; CHECK: [[#]] = OpBitCount [[i32x2_t]] [[p32x2]]
 
 ; p64x2
-; CHECK: [[p64x2_bitcast:%.+]] = OpBitcast [[i32x4_t]] [[p64x2]]
-; CHECK: [[p64x2_bitcount:%.+]] = OpBitCount [[i32x4_t]] [[p64x2_bitcast]]
-; CHECK: [[odd_indexes:%.+]] = OpVectorShuffle [[i32x2_t]] [[p64x2_bitcount]] [[p64x2_bitcount]] 1 3
-; CHECK: [[even_indexes:%.+]] = OpVectorShuffle [[i32x2_t]] [[p64x2_bitcount]] [[p64x2_bitcount]] 0 2
-; CHECK: [[add:%.+]] = OpIAdd [[i32x2_t]] [[odd_indexes]] [[even_indexes]]
-; CHECK: [[#]] = OpUConvert [[i64x2_t]] [[add]]
+; CHECK: [[p64x2_trunc_low:%.+]] = OpUConvert [[i32x2_t]] [[p64x2]]
+; CHECK: [[p64x2_low:%.+]] = OpBitCount [[i32x2_t]] [[p64x2_trunc_low]]
+; CHECK: [[p64x2_shift_high:%.+]] = OpShiftRightLogical [[i64x2_t]] [[p64x2]] [[i64x2_32]]
+; CHECK: [[p64x2_trunc_high:%.+]] = OpUConvert [[i32x2_t]] [[p64x2_shift_high]]
+; CHECK: [[p64x2_high:%.+]] = OpBitCount [[i32x2_t]] [[p64x2_trunc_high]]
+; CHECK: [[p64x2_sum:%.+]] = OpIAdd [[i32x2_t]] [[p64x2_high]] [[p64x2_low]]
+; CHECK: %[[#]] = OpUConvert [[i64x2_t]] [[p64x2_sum]]
 
 ; p64x3
-; CHECK: [[first_half:%.+]] = OpVectorShuffle [[i64x2_t]] [[p64x3]] [[p64x3]] 0 1
-; CHECK: [[p64x2_bitcast:%.+]] = OpBitcast [[i32x4_t]] [[first_half]]
-; CHECK: [[p64x2_bitcount:%.+]] = OpBitCount [[i32x4_t]] [[p64x2_bitcast]]
-; CHECK: [[odd_indexes:%.+]] = OpVectorShuffle [[i32x2_t]] [[p64x2_bitcount]] [[p64x2_bitcount]] 1 3
-; CHECK: [[even_indexes:%.+]] = OpVectorShuffle [[i32x2_t]] [[p64x2_bitcount]] [[p64x2_bitcount]] 0 2
-; CHECK: [[add:%.+]] = OpIAdd [[i32x2_t]] [[odd_indexes]] [[even_indexes]]
-; CHECK: [[first_half_result:%.+]] = OpUConvert [[i64x2_t]] [[add]]
-
-; CHECK: [[second_half:%.+]] = OpVectorExtractDynamic [[i64_t]] [[p64x3]] [[two]]
-; CHECK: [[p64_bitcast:%.+]] = OpBitcast [[i32x2_t]] [[second_half]]
-; CHECK: [[p64_bitcount:%.+]] = OpBitCount [[i32x2_t]] [[p64_bitcast]]
-; CHECK: [[index_one:%.+]] = OpVectorExtractDynamic [[i32_t]] [[p64_bitcount]] [[one]]
-; CHECK: [[index_zero:%.+]] = OpVectorExtractDynamic [[i32_t]] [[p64_bitcount]] [[zero]]
-; CHECK: [[add:%.+]] = OpIAdd [[i32_t]] [[index_one]] [[index_zero]]
-; CHECK: [[second_half_result:%.+]] = OpUConvert [[i64_t]] [[add]]
-; CHECK: %[[#]] = OpCompositeConstruct [[i64x3_t]] [[first_half_result]] [[second_half_result]]
+; CHECK: [[p64x3_trunc_low:%.+]] = OpUConvert [[i32x3_t]] [[p64x3]]
+; CHECK: [[p64x3_low:%.+]] = OpBitCount [[i32x3_t]] [[p64x3_trunc_low]]
+; CHECK: [[p64x3_shift_high:%.+]] = OpShiftRightLogical [[i64x3_t]] [[p64x3]] [[i64x3_32]]
+; CHECK: [[p64x3_trunc_high:%.+]] = OpUConvert [[i32x3_t]] [[p64x3_shift_high]]
+; CHECK: [[p64x3_high:%.+]] = OpBitCount [[i32x3_t]] [[p64x3_trunc_high]]
+; CHECK: [[p64x3_sum:%.+]] = OpIAdd [[i32x3_t]] [[p64x3_high]] [[p64x3_low]]
+; CHECK: %[[#]] = OpUConvert [[i64x3_t]] [[p64x3_sum]]
 
 ; p64x4
-; CHECK: [[first_half:%.+]] = OpVectorShuffle [[i64x2_t]] [[p64x4]] [[p64x4]] 0 1
-; CHECK: [[p64x2_bitcast:%.+]] = OpBitcast [[i32x4_t]] [[first_half]]
-; CHECK: [[p64x2_bitcount:%.+]] = OpBitCount [[i32x4_t]] [[p64x2_bitcast]]
-; CHECK: [[odd_indexes:%.+]] = OpVectorShuffle [[i32x2_t]] [[p64x2_bitcount]] [[p64x2_bitcount]] 1 3
-; CHECK: [[even_indexes:%.+]] = OpVectorShuffle [[i32x2_t]] [[p64x2_bitcount]] [[p64x2_bitcount]] 0 2
-; CHECK: [[add:%.+]] = OpIAdd [[i32x2_t]] [[odd_indexes]] [[even_indexes]]
-; CHECK: [[first_half_result:%.+]] = OpUConvert [[i64x2_t]] [[add]]
-
-; CHECK: [[second_half:%.+]] = OpVectorShuffle [[i64x2_t]] [[p64x4]] [[p64x4]] 2 3
-; CHECK: [[p64x2_bitcast:%.+]] = OpBitcast [[i32x4_t]] [[second_half]]
-; CHECK: [[p64x2_bitcount:%.+]] = OpBitCount [[i32x4_t]] [[p64x2_bitcast]]
-; CHECK: [[odd_indexes:%.+]] = OpVectorShuffle [[i32x2_t]] [[p64x2_bitcount]] [[p64x2_bitcount]] 1 3
-; CHECK: [[even_indexes:%.+]] = OpVectorShuffle [[i32x2_t]] [[p64x2_bitcount]] [[p64x2_bitcount]] 0 2
-; CHECK: [[add:%.+]] = OpIAdd [[i32x2_t]] [[odd_indexes]] [[even_indexes]]
-; CHECK: [[second_half_result:%.+]] = OpUConvert [[i64x2_t]] [[add]]
-
-; CHECK: %[[#]] = OpCompositeConstruct [[i64x4_t]] [[first_half_result]] [[second_half_result]]
+; CHECK: [[p64x4_trunc_low:%.+]] = OpUConvert [[i32x4_t]] [[p64x4]]
+; CHECK: [[p64x4_low:%.+]] = OpBitCount [[i32x4_t]] [[p64x4_trunc_low]]
+; CHECK: [[p64x4_shift_high:%.+]] = OpShiftRightLogical [[i64x4_t]] [[p64x4]] [[i64x4_32]]
+; CHECK: [[p64x4_trunc_high:%.+]] = OpUConvert [[i32x4_t]] [[p64x4_shift_high]]
+; CHECK: [[p64x4_high:%.+]] = OpBitCount [[i32x4_t]] [[p64x4_trunc_high]]
+; CHECK: [[p64x4_sum:%.+]] = OpIAdd [[i32x4_t]] [[p64x4_high]] [[p64x4_low]]
+; CHECK: %[[#]] = OpUConvert [[i64x4_t]] [[p64x4_sum]]
 
 ; p16x3
 ; CHECK: [[p16_conversion_in:%.+]] = OpUConvert [[i32x3_t]] [[p16x3]]

>From 959d946da87b4169c1b084192b2dd4d7cd083932 Mon Sep 17 00:00:00 2001
From: Justin Bogner <mail at justinbogner.com>
Date: Fri, 28 Aug 2026 17:55:41 -0700
Subject: [PATCH 2/3] fixup: clang-format

---
 llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp b/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
index 498bbb31bba68..c4793ee1e19ba 100644
--- a/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
+++ b/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
@@ -1840,9 +1840,9 @@ bool SPIRVInstructionSelector::selectPopCount64(Register ResVReg,
     return false;
 
   // Shift the high bits over and count them too.
-  Register ShiftAmount =
-      ComponentCount == 1 ? GR.getOrCreateConstInt(32, I, SrcType, TII)
-                          : GR.getOrCreateConstVector(32, I, SrcType, TII);
+  Register ShiftAmount = ComponentCount == 1
+                             ? GR.getOrCreateConstInt(32, I, SrcType, TII)
+                             : GR.getOrCreateConstVector(32, I, SrcType, TII);
   unsigned ShiftOp = ComponentCount == 1 ? SPIRV::OpShiftRightLogicalS
                                          : SPIRV::OpShiftRightLogicalV;
   Register Shift = MRI->createVirtualRegister(GR.getRegClass(SrcType));

>From a69410d3e3bc9eeaca9de1de88be21ba1dbb4b5a Mon Sep 17 00:00:00 2001
From: Justin Bogner <mail at justinbogner.com>
Date: Tue, 1 Sep 2026 22:04:11 -0700
Subject: [PATCH 3/3] fixup: Handle 1-element vectors

---
 .../Target/SPIRV/SPIRVInstructionSelector.cpp |  55 ++--
 .../CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll | 250 +++++++++++-------
 2 files changed, 182 insertions(+), 123 deletions(-)

diff --git a/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp b/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
index c4793ee1e19ba..35242809b76b1 100644
--- a/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
+++ b/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
@@ -1787,16 +1787,19 @@ bool SPIRVInstructionSelector::selectPopCount16(Register ResVReg,
                                                 MachineInstr &I,
                                                 unsigned ExtOpcode,
                                                 unsigned Opcode) const {
-  Register OpReg = I.getOperand(1).getReg();
-  unsigned NumElems = GR.getScalarOrVectorComponentCount(OpReg);
-
   MachineIRBuilder MIRBuilder(I);
+
+  Register OpReg = I.getOperand(1).getReg();
+  SPIRVTypeInst SrcType = GR.getSPIRVTypeForVReg(OpReg);
+  unsigned ComponentCount = GR.getScalarOrVectorComponentCount(SrcType);
+  bool IsScalar = !isVectorType(SrcType);
   SPIRVTypeInst I32Type = GR.getOrCreateSPIRVIntegerType(32, MIRBuilder);
-  SPIRVTypeInst I32VectorType =
-      GR.getOrCreateSPIRVVectorType(I32Type, NumElems, MIRBuilder, false);
 
-  bool IsVector = NumElems > 1;
-  SPIRVTypeInst ExtType = IsVector ? I32VectorType : I32Type;
+  SPIRVTypeInst ExtType =
+      IsScalar ? I32Type
+               : GR.getOrCreateSPIRVVectorType(I32Type, ComponentCount,
+                                               MIRBuilder, /*IsSigned=*/false);
+
   Register ExtReg = MRI->createVirtualRegister(GR.getRegClass(ExtType));
   // Always use OpUConvert to always use a 0 extend
   if (!selectOpWithSrcs(ExtReg, ExtType, I, {OpReg}, SPIRV::OpUConvert))
@@ -1826,41 +1829,47 @@ bool SPIRVInstructionSelector::selectPopCount64(Register ResVReg,
 
   SPIRVTypeInst SrcType = GR.getSPIRVTypeForVReg(SrcReg);
   unsigned ComponentCount = GR.getScalarOrVectorComponentCount(SrcType);
+  bool IsScalar = !isVectorType(SrcType);
   SPIRVTypeInst I32Type = GR.getOrCreateSPIRVIntegerType(32, MIRBuilder);
-  SPIRVTypeInst VecI32Type = GR.getOrCreateSPIRVVectorType(
-      I32Type, ComponentCount, MIRBuilder, /*IsSigned=*/false);
+
+  // We need to work with a type matching the shape of the input but using 32
+  // bit int instead of 64.
+  SPIRVTypeInst WorkingType =
+      IsScalar ? I32Type
+               : GR.getOrCreateSPIRVVectorType(I32Type, ComponentCount,
+                                               MIRBuilder, /*IsSigned=*/false);
 
   // Truncate and count the low bits.
-  Register Trunc = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
-  if (!selectOpWithSrcs(Trunc, VecI32Type, I, {SrcReg}, SPIRV::OpUConvert))
+  Register Trunc = MRI->createVirtualRegister(GR.getRegClass(WorkingType));
+  if (!selectOpWithSrcs(Trunc, WorkingType, I, {SrcReg}, SPIRV::OpUConvert))
     return false;
 
-  Register LowCount = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
-  if (!selectOpWithSrcs(LowCount, VecI32Type, I, {Trunc}, SPIRV::OpBitCount))
+  Register LowCount = MRI->createVirtualRegister(GR.getRegClass(WorkingType));
+  if (!selectOpWithSrcs(LowCount, WorkingType, I, {Trunc}, SPIRV::OpBitCount))
     return false;
 
   // Shift the high bits over and count them too.
-  Register ShiftAmount = ComponentCount == 1
+  Register ShiftAmount = IsScalar
                              ? GR.getOrCreateConstInt(32, I, SrcType, TII)
                              : GR.getOrCreateConstVector(32, I, SrcType, TII);
-  unsigned ShiftOp = ComponentCount == 1 ? SPIRV::OpShiftRightLogicalS
-                                         : SPIRV::OpShiftRightLogicalV;
+  unsigned ShiftOp =
+      IsScalar ? SPIRV::OpShiftRightLogicalS : SPIRV::OpShiftRightLogicalV;
   Register Shift = MRI->createVirtualRegister(GR.getRegClass(SrcType));
   if (!selectOpWithSrcs(Shift, SrcType, I, {SrcReg, ShiftAmount}, ShiftOp))
     return false;
 
-  Trunc = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
-  if (!selectOpWithSrcs(Trunc, VecI32Type, I, {Shift}, SPIRV::OpUConvert))
+  Trunc = MRI->createVirtualRegister(GR.getRegClass(WorkingType));
+  if (!selectOpWithSrcs(Trunc, WorkingType, I, {Shift}, SPIRV::OpUConvert))
     return false;
 
-  Register HighCount = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
-  if (!selectOpWithSrcs(HighCount, VecI32Type, I, {Trunc}, SPIRV::OpBitCount))
+  Register HighCount = MRI->createVirtualRegister(GR.getRegClass(WorkingType));
+  if (!selectOpWithSrcs(HighCount, WorkingType, I, {Trunc}, SPIRV::OpBitCount))
     return false;
 
   // Add them up and zext or sext back to 64 bit values.
-  Register Sum = MRI->createVirtualRegister(GR.getRegClass(VecI32Type));
-  if (!selectOpWithSrcs(Sum, VecI32Type, I, {HighCount, LowCount},
-                        ComponentCount == 1 ? SPIRV::OpIAddS : SPIRV::OpIAddV))
+  Register Sum = MRI->createVirtualRegister(GR.getRegClass(WorkingType));
+  if (!selectOpWithSrcs(Sum, WorkingType, I, {HighCount, LowCount},
+                        IsScalar ? SPIRV::OpIAddS : SPIRV::OpIAddV))
     return false;
 
   bool IsSigned = GR.isScalarOrVectorSigned(ResType);
diff --git a/llvm/test/CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll b/llvm/test/CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll
index 066d6151ff0d4..42a62dbce3f35 100644
--- a/llvm/test/CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll
+++ b/llvm/test/CodeGen/SPIRV/llvm-intrinsics/ctpop-vk.ll
@@ -1,4 +1,5 @@
-; RUN: llc -verify-machineinstrs -O0 -mtriple=spirv1.6-vulkan1.3-unknown %s -o - | FileCheck %s
+; RUN: llc -verify-machineinstrs -O0 -mtriple=spirv1.6-vulkan1.3-unknown %s -o - | FileCheck %s --check-prefix=CHECK,CHECK-SCALAR
+; RUN: llc -spirv-ext=+SPV_EXT_long_vector -verify-machineinstrs -O0 -mtriple=spirv1.6-vulkan1.3-unknown %s -o - | FileCheck --check-prefix=CHECK,CHECK-VECTOR %s
 ; RUN: %if spirv-tools %{ llc -O0 -mtriple=spirv1.6-vulkan1.3-unknown %s -o - -filetype=obj | spirv-val --target-env vulkan1.3 %}
 
 
@@ -14,111 +15,160 @@
 ; CHECK-DAG: [[i64x4_t:%.+]] = OpTypeVector [[i64_t]] 4
 ; CHECK-DAG: [[i16x3_t:%.+]] = OpTypeVector [[i16_t]] 3
 
+; CHECK-VECTOR-DAG: [[i32_one:%.+]] = OpConstant [[i32_t]] 1
+; CHECK-VECTOR-DAG: [[i16x1_t:%.+]] = OpTypeVectorIdEXT [[i16_t]] [[i32_one]]
+; CHECK-VECTOR-DAG: [[i32x1_t:%.+]] = OpTypeVectorIdEXT [[i32_t]] [[i32_one]]
+; CHECK-VECTOR-DAG: [[i64x1_t:%.+]] = OpTypeVectorIdEXT [[i64_t]] [[i32_one]]
+
 ; CHECK-DAG: [[i64_32:%.+]] = OpConstant [[i64_t]] 32
+; CHECK-VECTOR-DAG: [[i64x1_32:%.+]] = OpConstantComposite [[i64x1_t]] [[i64_32]]
 ; CHECK-DAG: [[i64x2_32:%.+]] = OpConstantComposite [[i64x2_t]] [[i64_32]] [[i64_32]]
 ; CHECK-DAG: [[i64x3_32:%.+]] = OpConstantComposite [[i64x3_t]] [[i64_32]] [[i64_32]] [[i64_32]]
 ; CHECK-DAG: [[i64x4_32:%.+]] = OpConstantComposite [[i64x4_t]] [[i64_32]] [[i64_32]] [[i64_32]] [[i64_32]]
 
-; CHECK-LABEL:  ; -- Begin function test
-
-; CHECK: [[p8:%.+]] = OpFunctionParameter [[i8_t]]
-; CHECK: [[p16:%.+]] = OpFunctionParameter [[i16_t]]
-; CHECK: [[p32:%.+]] = OpFunctionParameter [[i32_t]]
-; CHECK: [[p64:%.+]] = OpFunctionParameter [[i64_t]]
-; CHECK: [[p32x2:%.+]] = OpFunctionParameter [[i32x2_t]]
-; CHECK: [[p64x2:%.+]] = OpFunctionParameter [[i64x2_t]]
-; CHECK: [[p64x3:%.+]] = OpFunctionParameter [[i64x3_t]]
-; CHECK: [[p64x4:%.+]] = OpFunctionParameter [[i64x4_t]]
-; CHECK: [[p16x3:%.+]] = OpFunctionParameter [[i16x3_t]]
-
-; p8
-; CHECK: [[p8_conversion_in:%.+]] = OpUConvert [[i32_t]] [[p8]]
-; CHECK: [[p8_bitcount:%.+]] = OpBitCount [[i32_t]] [[p8_conversion_in]]
-; CHECK: %[[#]] = OpUConvert [[i8_t]] [[p8_bitcount]]
-
-; p16
-; CHECK: [[p16_conversion_in:%.+]] = OpUConvert [[i32_t]] [[p16]]
-; CHECK: [[p16_bitcount:%.+]] = OpBitCount [[i32_t]] [[p16_conversion_in]]
-; CHECK: %[[#]] = OpUConvert [[i16_t]] [[p16_bitcount]]
-
-; p32
-; CHECK: [[p32_bitcount:%.+]] = OpBitCount [[i32_t]] [[p32]]
-
-; p64
-; CHECK: [[p64_trunc_low:%.+]] = OpUConvert [[i32_t]] [[p64]]
-; CHECK: [[p64_low:%.+]] = OpBitCount [[i32_t]] [[p64_trunc_low]]
-; CHECK: [[p64_shift_high:%.+]] = OpShiftRightLogical [[i64_t]] [[p64]] [[i64_32]]
-; CHECK: [[p64_trunc_high:%.+]] = OpUConvert [[i32_t]] [[p64_shift_high]]
-; CHECK: [[p64_high:%.+]] = OpBitCount [[i32_t]] [[p64_trunc_high]]
-; CHECK: [[p64_sum:%.+]] = OpIAdd [[i32_t]] [[p64_high]] [[p64_low]]
-; CHECK: %[[#]] = OpUConvert [[i64_t]] [[p64_sum]]
-
-; p32x2
-; CHECK: [[#]] = OpBitCount [[i32x2_t]] [[p32x2]]
-
-; p64x2
-; CHECK: [[p64x2_trunc_low:%.+]] = OpUConvert [[i32x2_t]] [[p64x2]]
-; CHECK: [[p64x2_low:%.+]] = OpBitCount [[i32x2_t]] [[p64x2_trunc_low]]
-; CHECK: [[p64x2_shift_high:%.+]] = OpShiftRightLogical [[i64x2_t]] [[p64x2]] [[i64x2_32]]
-; CHECK: [[p64x2_trunc_high:%.+]] = OpUConvert [[i32x2_t]] [[p64x2_shift_high]]
-; CHECK: [[p64x2_high:%.+]] = OpBitCount [[i32x2_t]] [[p64x2_trunc_high]]
-; CHECK: [[p64x2_sum:%.+]] = OpIAdd [[i32x2_t]] [[p64x2_high]] [[p64x2_low]]
-; CHECK: %[[#]] = OpUConvert [[i64x2_t]] [[p64x2_sum]]
-
-; p64x3
-; CHECK: [[p64x3_trunc_low:%.+]] = OpUConvert [[i32x3_t]] [[p64x3]]
-; CHECK: [[p64x3_low:%.+]] = OpBitCount [[i32x3_t]] [[p64x3_trunc_low]]
-; CHECK: [[p64x3_shift_high:%.+]] = OpShiftRightLogical [[i64x3_t]] [[p64x3]] [[i64x3_32]]
-; CHECK: [[p64x3_trunc_high:%.+]] = OpUConvert [[i32x3_t]] [[p64x3_shift_high]]
-; CHECK: [[p64x3_high:%.+]] = OpBitCount [[i32x3_t]] [[p64x3_trunc_high]]
-; CHECK: [[p64x3_sum:%.+]] = OpIAdd [[i32x3_t]] [[p64x3_high]] [[p64x3_low]]
-; CHECK: %[[#]] = OpUConvert [[i64x3_t]] [[p64x3_sum]]
-
-; p64x4
-; CHECK: [[p64x4_trunc_low:%.+]] = OpUConvert [[i32x4_t]] [[p64x4]]
-; CHECK: [[p64x4_low:%.+]] = OpBitCount [[i32x4_t]] [[p64x4_trunc_low]]
-; CHECK: [[p64x4_shift_high:%.+]] = OpShiftRightLogical [[i64x4_t]] [[p64x4]] [[i64x4_32]]
-; CHECK: [[p64x4_trunc_high:%.+]] = OpUConvert [[i32x4_t]] [[p64x4_shift_high]]
-; CHECK: [[p64x4_high:%.+]] = OpBitCount [[i32x4_t]] [[p64x4_trunc_high]]
-; CHECK: [[p64x4_sum:%.+]] = OpIAdd [[i32x4_t]] [[p64x4_high]] [[p64x4_low]]
-; CHECK: %[[#]] = OpUConvert [[i64x4_t]] [[p64x4_sum]]
-
-; p16x3
-; CHECK: [[p16_conversion_in:%.+]] = OpUConvert [[i32x3_t]] [[p16x3]]
-; CHECK: [[p16_bitcount:%.+]] = OpBitCount [[i32x3_t]] [[p16_conversion_in]]
-; CHECK: %[[#]] = OpUConvert [[i16x3_t]] [[p16_bitcount]]
-
- at g1 = private global i8  0, align 4
- at g2 = private global i16 0, align 4
- at g3 = private global i32 0, align 4
- at g4 = private global i64 0, align 8
- at g5 = private global <2 x i32> zeroinitializer, align 4
- at g6 = private global <2 x i64> zeroinitializer, align 8
- at g7 = private global <3 x i64> zeroinitializer, align 8
- at g8 = private global <4 x i64> zeroinitializer, align 8
- at g9 = private global <3 x i16> zeroinitializer, align 4
-
-
-define internal void @test(i8 %x8, i16 %x16, i32 %x32, i64 %x64, <2 x i32> %x2i32, <2 x i64> %x2i64, <3 x i64> %x3i64, <4 x i64> %x4i64, <3 x i16> %x3i16) local_unnamed_addr {
+ at g8 = private global i8  0, align 4
+ at g16 = private global i16 0, align 4
+ at g32 = private global i32 0, align 4
+ at g64 = private global i64 0, align 8
+ at g2i32 = private global <2 x i32> zeroinitializer, align 4
+ at g2i64 = private global <2 x i64> zeroinitializer, align 8
+ at g3i64 = private global <3 x i64> zeroinitializer, align 8
+ at g4i64 = private global <4 x i64> zeroinitializer, align 8
+ at g3i16 = private global <3 x i16> zeroinitializer, align 4
+ at g1i16 = private global <1 x i16> zeroinitializer, align 4
+ at g1i32 = private global <1 x i32> zeroinitializer, align 4
+ at g1i64 = private global <1 x i64> zeroinitializer, align 8
+
+define internal void @test(i8 %x8, i16 %x16, i32 %x32, i64 %x64, <2 x i32> %x2i32, <2 x i64> %x2i64, <3 x i64> %x3i64, <4 x i64> %x4i64, <3 x i16> %x3i16, <1 x i16> %x1i16, <1 x i32> %x1i32, <1 x i64> %x1i64) local_unnamed_addr {
 entry:
-  %0 = tail call i8 @llvm.ctpop.i8(i8 %x8)
-  store i8 %0, ptr @g1, align 4
-  %1 = tail call i16 @llvm.ctpop.i16(i16 %x16)
-  store i16 %1, ptr @g2, align 4
-  %2 = tail call i32 @llvm.ctpop.i32(i32 %x32)
-  store i32 %2, ptr @g3, align 4
-  %3 = tail call i64 @llvm.ctpop.i64(i64 %x64)
-  store i64 %3, ptr @g4, align 8
-  %4 = tail call <2 x i32> @llvm.ctpop.v2i32(<2 x i32> %x2i32)
-  store <2 x i32> %4, ptr @g5, align 4
-  %5 = tail call <2 x i64> @llvm.ctpop.v2i64(<2 x i64> %x2i64)
-  store <2 x i64> %5, ptr @g6, align 4
-  %6 = tail call <3 x i64> @llvm.ctpop.v3i64(<3 x i64> %x3i64)
-  store <3 x i64> %6, ptr @g7, align 4
-  %7 = tail call <4 x i64> @llvm.ctpop.v4i64(<4 x i64> %x4i64)
-  store <4 x i64> %7, ptr @g8, align 4
-  %8 = tail call <3 x i16> @llvm.ctpop.v3i16(<3 x i16> %x3i16)
-  store <3 x i16> %8, ptr @g9, align 4
+  ; CHECK-LABEL:  ; -- Begin function test
+  ; CHECK: [[p8:%.+]] = OpFunctionParameter [[i8_t]]
+  ; CHECK: [[p16:%.+]] = OpFunctionParameter [[i16_t]]
+  ; CHECK: [[p32:%.+]] = OpFunctionParameter [[i32_t]]
+  ; CHECK: [[p64:%.+]] = OpFunctionParameter [[i64_t]]
+  ; CHECK: [[p32x2:%.+]] = OpFunctionParameter [[i32x2_t]]
+  ; CHECK: [[p64x2:%.+]] = OpFunctionParameter [[i64x2_t]]
+  ; CHECK: [[p64x3:%.+]] = OpFunctionParameter [[i64x3_t]]
+  ; CHECK: [[p64x4:%.+]] = OpFunctionParameter [[i64x4_t]]
+  ; CHECK: [[p16x3:%.+]] = OpFunctionParameter [[i16x3_t]]
+  ; CHECK-SCALAR: [[p16x1:%.+]] = OpFunctionParameter [[i16_t]]
+  ; CHECK-VECTOR: [[p16x1:%.+]] = OpFunctionParameter [[i16x1_t]]
+  ; CHECK-SCALAR: [[p32x1:%.+]] = OpFunctionParameter [[i32_t]]
+  ; CHECK-VECTOR: [[p32x1:%.+]] = OpFunctionParameter [[i32x1_t]]
+  ; CHECK-SCALAR: [[p64x1:%.+]] = OpFunctionParameter [[i64_t]]
+  ; CHECK-VECTOR: [[p64x1:%.+]] = OpFunctionParameter [[i64x1_t]]
+
+  ; p8
+  ; CHECK: [[p8_conversion_in:%.+]] = OpUConvert [[i32_t]] [[p8]]
+  ; CHECK: [[p8_bitcount:%.+]] = OpBitCount [[i32_t]] [[p8_conversion_in]]
+  ; CHECK: %[[#]] = OpUConvert [[i8_t]] [[p8_bitcount]]
+  %y8 = tail call i8 @llvm.ctpop.i8(i8 %x8)
+  store i8 %y8, ptr @g8, align 4
+
+  ; p16
+  ; CHECK: [[p16_conversion_in:%.+]] = OpUConvert [[i32_t]] [[p16]]
+  ; CHECK: [[p16_bitcount:%.+]] = OpBitCount [[i32_t]] [[p16_conversion_in]]
+  ; CHECK: %[[#]] = OpUConvert [[i16_t]] [[p16_bitcount]]
+  %y16 = tail call i16 @llvm.ctpop.i16(i16 %x16)
+  store i16 %y16, ptr @g16, align 4
+
+  ; p32
+  ; CHECK: [[p32_bitcount:%.+]] = OpBitCount [[i32_t]] [[p32]]
+  %y32 = tail call i32 @llvm.ctpop.i32(i32 %x32)
+  store i32 %y32, ptr @g32, align 4
+
+  ; p64
+  ; CHECK: [[p64_trunc_low:%.+]] = OpUConvert [[i32_t]] [[p64]]
+  ; CHECK: [[p64_low:%.+]] = OpBitCount [[i32_t]] [[p64_trunc_low]]
+  ; CHECK: [[p64_shift_high:%.+]] = OpShiftRightLogical [[i64_t]] [[p64]] [[i64_32]]
+  ; CHECK: [[p64_trunc_high:%.+]] = OpUConvert [[i32_t]] [[p64_shift_high]]
+  ; CHECK: [[p64_high:%.+]] = OpBitCount [[i32_t]] [[p64_trunc_high]]
+  ; CHECK: [[p64_sum:%.+]] = OpIAdd [[i32_t]] [[p64_high]] [[p64_low]]
+  ; CHECK: %[[#]] = OpUConvert [[i64_t]] [[p64_sum]]
+  %y64 = tail call i64 @llvm.ctpop.i64(i64 %x64)
+  store i64 %y64, ptr @g64, align 8
+
+  ; p32x2
+  ; CHECK: [[#]] = OpBitCount [[i32x2_t]] [[p32x2]]
+  %y2i32 = tail call <2 x i32> @llvm.ctpop.v2i32(<2 x i32> %x2i32)
+  store <2 x i32> %y2i32, ptr @g2i32, align 4
+
+  ; p64x2
+  ; CHECK: [[p64x2_trunc_low:%.+]] = OpUConvert [[i32x2_t]] [[p64x2]]
+  ; CHECK: [[p64x2_low:%.+]] = OpBitCount [[i32x2_t]] [[p64x2_trunc_low]]
+  ; CHECK: [[p64x2_shift_high:%.+]] = OpShiftRightLogical [[i64x2_t]] [[p64x2]] [[i64x2_32]]
+  ; CHECK: [[p64x2_trunc_high:%.+]] = OpUConvert [[i32x2_t]] [[p64x2_shift_high]]
+  ; CHECK: [[p64x2_high:%.+]] = OpBitCount [[i32x2_t]] [[p64x2_trunc_high]]
+  ; CHECK: [[p64x2_sum:%.+]] = OpIAdd [[i32x2_t]] [[p64x2_high]] [[p64x2_low]]
+  ; CHECK: %[[#]] = OpUConvert [[i64x2_t]] [[p64x2_sum]]
+  %y2i64 = tail call <2 x i64> @llvm.ctpop.v2i64(<2 x i64> %x2i64)
+  store <2 x i64> %y2i64, ptr @g2i64, align 4
+
+  ; p64x3
+  ; CHECK: [[p64x3_trunc_low:%.+]] = OpUConvert [[i32x3_t]] [[p64x3]]
+  ; CHECK: [[p64x3_low:%.+]] = OpBitCount [[i32x3_t]] [[p64x3_trunc_low]]
+  ; CHECK: [[p64x3_shift_high:%.+]] = OpShiftRightLogical [[i64x3_t]] [[p64x3]] [[i64x3_32]]
+  ; CHECK: [[p64x3_trunc_high:%.+]] = OpUConvert [[i32x3_t]] [[p64x3_shift_high]]
+  ; CHECK: [[p64x3_high:%.+]] = OpBitCount [[i32x3_t]] [[p64x3_trunc_high]]
+  ; CHECK: [[p64x3_sum:%.+]] = OpIAdd [[i32x3_t]] [[p64x3_high]] [[p64x3_low]]
+  ; CHECK: %[[#]] = OpUConvert [[i64x3_t]] [[p64x3_sum]]
+  %y3i64 = tail call <3 x i64> @llvm.ctpop.v3i64(<3 x i64> %x3i64)
+  store <3 x i64> %y3i64, ptr @g3i64, align 4
+
+  ; p64x4
+  ; CHECK: [[p64x4_trunc_low:%.+]] = OpUConvert [[i32x4_t]] [[p64x4]]
+  ; CHECK: [[p64x4_low:%.+]] = OpBitCount [[i32x4_t]] [[p64x4_trunc_low]]
+  ; CHECK: [[p64x4_shift_high:%.+]] = OpShiftRightLogical [[i64x4_t]] [[p64x4]] [[i64x4_32]]
+  ; CHECK: [[p64x4_trunc_high:%.+]] = OpUConvert [[i32x4_t]] [[p64x4_shift_high]]
+  ; CHECK: [[p64x4_high:%.+]] = OpBitCount [[i32x4_t]] [[p64x4_trunc_high]]
+  ; CHECK: [[p64x4_sum:%.+]] = OpIAdd [[i32x4_t]] [[p64x4_high]] [[p64x4_low]]
+  ; CHECK: %[[#]] = OpUConvert [[i64x4_t]] [[p64x4_sum]]
+  %y4i64 = tail call <4 x i64> @llvm.ctpop.v4i64(<4 x i64> %x4i64)
+  store <4 x i64> %y4i64, ptr @g4i64, align 4
+
+  ; p16x3
+  ; CHECK: [[p16_conversion_in:%.+]] = OpUConvert [[i32x3_t]] [[p16x3]]
+  ; CHECK: [[p16_bitcount:%.+]] = OpBitCount [[i32x3_t]] [[p16_conversion_in]]
+  ; CHECK: %[[#]] = OpUConvert [[i16x3_t]] [[p16_bitcount]]
+  %y3i16 = tail call <3 x i16> @llvm.ctpop.v3i16(<3 x i16> %x3i16)
+  store <3 x i16> %y3i16, ptr @g3i16, align 4
+
+  ; p16x1
+  ;
+  ; CHECK-SCALAR: [[p16x1_conversion_in:%.+]] = OpUConvert [[i32_t]] [[p16x1]]
+  ; CHECK-SCALAR: [[p16x1_bitcount:%.+]] = OpBitCount [[i32_t]] [[p16x1_conversion_in]]
+  ; CHECK-SCALAR: %[[#]] = OpUConvert [[i16_t]] [[p16x1_bitcount]]
+  ;
+  ; CHECK-VECTOR: [[p16x1_conversion_in:%.+]] = OpUConvert [[i32x1_t]] [[p16x1]]
+  ; CHECK-VECTOR: [[p16x1_bitcount:%.+]] = OpBitCount [[i32x1_t]] [[p16x1_conversion_in]]
+  ; CHECK-VECTOR: %[[#]] = OpUConvert [[i16x1_t]] [[p16x1_bitcount]]
+  %y1i16 = tail call <1 x i16> @llvm.ctpop.v1i16(<1 x i16> %x1i16)
+  store <1 x i16> %y1i16, ptr @g1i16, align 4
+
+  ; p32x1
+  ; CHECK-SCALAR: [[p32x1_bitcount:%.+]] = OpBitCount [[i32_t]] [[p32x1]]
+  ; CHECK-VECTOR: [[p32x1_bitcount:%.+]] = OpBitCount [[i32x1_t]] [[p32x1]]
+  %y1i32 = tail call <1 x i32> @llvm.ctpop.v1i32(<1 x i32> %x1i32)
+  store <1 x i32> %y1i32, ptr @g1i32, align 4
+
+  ; p64x1
+  ; CHECK-SCALAR: [[p64x1_trunc_low:%.+]] = OpUConvert [[i32_t]] [[p64x1]]
+  ; CHECK-SCALAR: [[p64x1_low:%.+]] = OpBitCount [[i32_t]] [[p64x1_trunc_low]]
+  ; CHECK-SCALAR: [[p64x1_shift_high:%.+]] = OpShiftRightLogical [[i64_t]] [[p64x1]] [[i64_32]]
+  ; CHECK-SCALAR: [[p64x1_trunc_high:%.+]] = OpUConvert [[i32_t]] [[p64x1_shift_high]]
+  ; CHECK-SCALAR: [[p64x1_high:%.+]] = OpBitCount [[i32_t]] [[p64x1_trunc_high]]
+  ; CHECK-SCALAR: [[p64x1_sum:%.+]] = OpIAdd [[i32_t]] [[p64x1_high]] [[p64x1_low]]
+  ; CHECK-SCALAR: %[[#]] = OpUConvert [[i64_t]] [[p64x1_sum]]
+  ;
+  ; CHECK-VECTOR: [[p64x1_trunc_low:%.+]] = OpUConvert [[i32x1_t]] [[p64x1]]
+  ; CHECK-VECTOR: [[p64x1_low:%.+]] = OpBitCount [[i32x1_t]] [[p64x1_trunc_low]]
+  ; CHECK-VECTOR: [[p64x1_shift_high:%.+]] = OpShiftRightLogical [[i64x1_t]] [[p64x1]] [[i64x1_32]]
+  ; CHECK-VECTOR: [[p64x1_trunc_high:%.+]] = OpUConvert [[i32x1_t]] [[p64x1_shift_high]]
+  ; CHECK-VECTOR: [[p64x1_high:%.+]] = OpBitCount [[i32x1_t]] [[p64x1_trunc_high]]
+  ; CHECK-VECTOR: [[p64x1_sum:%.+]] = OpIAdd [[i32x1_t]] [[p64x1_high]] [[p64x1_low]]
+  ; CHECK-VECTOR: %[[#]] = OpUConvert [[i64x1_t]] [[p64x1_sum]]
+  %y1i64 = tail call <1 x i64> @llvm.ctpop.v1i64(<1 x i64> %x1i64)
+  store <1 x i64> %y1i64, ptr @g1i64, align 8
   ret void
 }
 



More information about the llvm-commits mailing list