[llvm] [DAGCombiner] Avoid illegal scalar nodes in estimate refinement (PR #219394)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 14 16:02:42 PDT 2026


https://github.com/MattPD updated https://github.com/llvm/llvm-project/pull/219394

>From de1c8d3ad6af9ffe6f4107468c13b3d3e5f67826 Mon Sep 17 00:00:00 2001
From: "Matt P. Dziubinski" <matt-p.dziubinski at hpe.com>
Date: Fri, 28 Aug 2026 02:00:23 -0500
Subject: [PATCH 1/2] [DAGCombiner] Avoid illegal scalar nodes in estimate
 refinement

On i686 with SSE1 but without SSE2 or x87, `v4f32` is legal while scalar
`f32` has the `TypeSoftenFloat` action. `SelectionDAG::getConstantFP`
represents a vector constant as a `BUILD_VECTOR` of scalar
`ConstantFPSDNode` operands. Estimate refinement can therefore introduce
illegal scalar nodes into a legal vector.

Refinement creates these constants in two phases. For `v4f32`, it
creates them before type legalization. For `v8f32`, it creates them in
the `AfterLegalizeTypes` combine, after type legalization splits the
vector. A change limited to the type legalizer misses the second phase.
A direct softened-integer representation of these constants would
introduce illegal `v4i32` nodes in this configuration.

Add a `DAGCombiner` helper that materializes affected refinement
constants as constant-pool vector loads when the value type is a legal
fixed-length vector whose scalar element action is `TypeSoftenFloat`.
All other types continue to use `SelectionDAG::getConstantFP`.

Use the helper to create the `1.0` constant in `BuildDivEstimate` and
the `-3.0` and `-0.5` constants in `buildSqrtNRTwoConst`. This preserves
requested reciprocal and reciprocal-square-root refinement in both
creation phases.

Fixes https://github.com/llvm/llvm-project/issues/217806

Assisted-by: Claude Opus 5, GPT-5.6 Sol.
---
 llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp |  31 ++-
 ...ate-refinement-illegal-scalar-types-pic.ll | 159 +++++++++++
 ...stimate-refinement-illegal-scalar-types.ll | 251 ++++++++++++++++++
 3 files changed, 438 insertions(+), 3 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll
 create mode 100644 llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll

diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index a9db383ede385..06dc5910d9f4b 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -653,6 +653,11 @@ namespace {
     SDValue BuildUDIV(SDNode *N);
     SDValue BuildSREMPow2(SDNode *N);
     SDValue buildOptimizedSREM(SDValue N0, SDValue N1, SDNode *N);
+    /// Materialize an FP constant for estimate refinement. Before operation
+    /// legalization, the softened-vector path returns an opaque load, so
+    /// callers requiring constant recognition, folding, or SDValue identity
+    /// must not use this helper.
+    SDValue materializeFPConstant(double Value, const SDLoc &DL, EVT VT);
     SDValue BuildLogBase2(SDValue V, const SDLoc &DL,
                           bool KnownNeverZero = false,
                           bool InexpensiveOnly = false,
@@ -31732,6 +31737,26 @@ SDValue DAGCombiner::BuildLogBase2(SDValue V, const SDLoc &DL,
   return LogBase2;
 }
 
+SDValue DAGCombiner::materializeFPConstant(double Value, const SDLoc &DL,
+                                           EVT VT) {
+  if (!VT.isFixedLengthVector() || !TLI.isTypeLegal(VT) ||
+      TLI.getTypeAction(*DAG.getContext(), VT.getScalarType()) !=
+          TargetLowering::TypeSoftenFloat)
+    return DAG.getConstantFP(Value, DL, VT);
+
+  assert(!LegalDAG && "load must be created before operation legalization");
+  Type *ScalarTy = VT.getScalarType().getTypeForEVT(*DAG.getContext());
+  Constant *Scalar = ConstantFP::get(ScalarTy, Value);
+  Constant *Vector =
+      ConstantVector::getSplat(VT.getVectorElementCount(), Scalar);
+  SDValue CPIdx =
+      DAG.getConstantPool(Vector, TLI.getPointerTy(DAG.getDataLayout()));
+  Align Alignment = cast<ConstantPoolSDNode>(CPIdx)->getAlign();
+  return DAG.getLoad(
+      VT, DL, DAG.getEntryNode(), CPIdx,
+      MachinePointerInfo::getConstantPool(DAG.getMachineFunction()), Alignment);
+}
+
 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
 /// For the reciprocal, we need to find the zero of the function:
 ///   F(X) = 1/X - A [which has a zero at X = 1/A]
@@ -31765,7 +31790,7 @@ SDValue DAGCombiner::BuildDivEstimate(SDValue N, SDValue Op,
 
     SDLoc DL(Op);
     if (Iterations) {
-      SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
+      SDValue FPOne = materializeFPConstant(1.0, DL, VT);
 
       // Newton iterations: Est = Est + Est (N - Arg * Est)
       // If this is the last iteration, also multiply by the numerator.
@@ -31843,8 +31868,8 @@ SDValue DAGCombiner::buildSqrtNRTwoConst(SDValue Arg, SDValue Est,
                                          unsigned Iterations, bool Reciprocal) {
   EVT VT = Arg.getValueType();
   SDLoc DL(Arg);
-  SDValue MinusThree = DAG.getConstantFP(-3.0, DL, VT);
-  SDValue MinusHalf = DAG.getConstantFP(-0.5, DL, VT);
+  SDValue MinusThree = materializeFPConstant(-3.0, DL, VT);
+  SDValue MinusHalf = materializeFPConstant(-0.5, DL, VT);
 
   // This routine must enter the loop below to work correctly
   // when (Reciprocal == false).
diff --git a/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll
new file mode 100644
index 0000000000000..1cea8189a6467
--- /dev/null
+++ b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll
@@ -0,0 +1,159 @@
+; NOTE: Do not autogenerate
+; RUN: split-file %s %t
+; RUN: llc -relocation-model=pic %t/estimate.ll -o - | FileCheck %s --check-prefix=PIC --enable-var-scope --implicit-check-not='{{^[[:space:]]+sqrtps[[:space:]]}}' --implicit-check-not='{{^[[:space:]]+divps[[:space:]]}}'
+; RUN: llc -relocation-model=pic %t/fallback.ll -o - | FileCheck %s --check-prefix=FALLBACK
+
+; This test owns refinement constant values and emitted PIC form. It binds each
+; pool definition to the canonical i386 call/pop/GOT expression and @GOTOFF
+; spelling. It checks immediate arithmetic where adjacent and otherwise follows
+; loaded registers to their arithmetic consumers in both vector halves.
+; Whole-input exclusions reject native fallback. The fallback input establishes
+; the excluded operations in the same configuration.
+
+;--- estimate.ll
+
+target triple = "i686-unknown-linux-gnu"
+
+; PIC: [[$V4_THREE:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC: [[$V4_HALF:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-LABEL: rsqrt_v4_default:
+; PIC: calll [[PB:\.L[0-9]+\$pb]]
+; PIC: [[PB]]:
+; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
+; PIC: [[TMP:\.Ltmp[0-9]+]]:
+; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
+; PIC: rsqrtps
+; PIC: addps [[$V4_THREE]]@GOTOFF([[GOT]]), {{%xmm[0-9]+}}
+; PIC: mulps [[$V4_HALF]]@GOTOFF([[GOT]]), {{%xmm[0-9]+}}
+; PIC: retl
+define <4 x float> @rsqrt_v4_default(
+    <4 x float> %n, <4 x float> %x) #0 {
+  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+  %q = fdiv arcp ninf <4 x float> %n, %sqrt
+  ret <4 x float> %q
+}
+
+; PIC: [[$V8_HALF:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC: [[$V8_THREE:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-LABEL: rsqrt_v8_default:
+; PIC: calll [[PB:\.L[0-9]+\$pb]]
+; PIC: [[PB]]:
+; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
+; PIC: [[TMP:\.Ltmp[0-9]+]]:
+; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
+; PIC: rsqrtps
+; PIC-NEXT: movaps [[$V8_HALF]]@GOTOFF([[GOT]]), [[V8_HALF_REG:%xmm[0-9]+]]
+; PIC: mulps [[V8_HALF_REG]], {{%xmm[0-9]+}}
+; PIC: movaps [[$V8_THREE]]@GOTOFF([[GOT]]), [[V8_THREE_REG:%xmm[0-9]+]]
+; PIC-NEXT: addps [[V8_THREE_REG]], {{%xmm[0-9]+}}
+; PIC: rsqrtps
+; PIC: mulps {{%xmm[0-9]+}}, [[V8_HALF_REG]]
+; PIC: addps [[V8_THREE_REG]], {{%xmm[0-9]+}}
+; PIC: retl
+define <8 x float> @rsqrt_v8_default(
+    <8 x float> %n, <8 x float> %x) #0 {
+  %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
+  %q = fdiv arcp ninf <8 x float> %n, %sqrt
+  ret <8 x float> %q
+}
+
+; PIC: [[$V4_ONE:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-LABEL: div_v4_steps_2:
+; PIC: calll [[PB:\.L[0-9]+\$pb]]
+; PIC: [[PB]]:
+; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
+; PIC: [[TMP:\.Ltmp[0-9]+]]:
+; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
+; PIC: rcpps
+; PIC: movaps [[$V4_ONE]]@GOTOFF([[GOT]]), [[V4_ONE_REG:%xmm[0-9]+]]
+; PIC-NEXT: subps {{%xmm[0-9]+}}, [[V4_ONE_REG]]
+; PIC: retl
+define <4 x float> @div_v4_steps_2(
+    <4 x float> %n, <4 x float> %d) #1 {
+  %q = fdiv arcp ninf <4 x float> %n, %d
+  ret <4 x float> %q
+}
+
+; PIC: [[$V8_ONE:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-LABEL: div_v8_steps_2:
+; PIC: calll [[PB:\.L[0-9]+\$pb]]
+; PIC: [[PB]]:
+; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
+; PIC: [[TMP:\.Ltmp[0-9]+]]:
+; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
+; PIC: rcpps
+; PIC: movaps [[$V8_ONE]]@GOTOFF([[GOT]]), [[V8_ONE_REG:%xmm[0-9]+]]
+; PIC: movaps [[V8_ONE_REG]], [[V8_ONE_COPY:%xmm[0-9]+]]
+; PIC: subps {{%xmm[0-9]+}}, [[V8_ONE_COPY]]
+; PIC: rcpps
+; PIC: subps {{%xmm[0-9]+}}, [[V8_ONE_REG]]
+; PIC: retl
+define <8 x float> @div_v8_steps_2(
+    <8 x float> %n, <8 x float> %d) #1 {
+  %q = fdiv arcp ninf <8 x float> %n, %d
+  ret <8 x float> %q
+}
+
+attributes #0 = {
+  "reciprocal-estimates"="vec-sqrtf"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #1 = {
+  "reciprocal-estimates"="vec-divf:2"
+  "target-features"="+sse,-sse2,-x87"
+}
+
+;--- fallback.ll
+
+target triple = "i686-unknown-linux-gnu"
+
+; FALLBACK-LABEL: fallback_sqrt_v4:
+; FALLBACK: {{^[[:space:]]+sqrtps[[:space:]]}}
+; FALLBACK: {{^[[:space:]]+rcpps[[:space:]]}}
+define <4 x float> @fallback_sqrt_v4(
+    <4 x float> %n, <4 x float> %x) #0 {
+  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+  %q = fdiv arcp ninf <4 x float> %n, %sqrt
+  ret <4 x float> %q
+}
+
+; FALLBACK-LABEL: fallback_div_v4:
+; FALLBACK: {{^[[:space:]]+divps[[:space:]]}}
+define <4 x float> @fallback_div_v4(
+    <4 x float> %n, <4 x float> %d) #1 {
+  %q = fdiv arcp ninf <4 x float> %n, %d
+  ret <4 x float> %q
+}
+
+attributes #0 = {
+  "reciprocal-estimates"="!vec-sqrtf"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #1 = {
+  "reciprocal-estimates"="!vec-divf"
+  "target-features"="+sse,-sse2,-x87"
+}
diff --git a/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll
new file mode 100644
index 0000000000000..b55e40957622d
--- /dev/null
+++ b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll
@@ -0,0 +1,251 @@
+; NOTE: Do not autogenerate
+; RUN: split-file %s %t
+; RUN: llc %t/estimate.ll -o - | FileCheck %s --check-prefix=ASM --enable-var-scope --implicit-check-not='{{^[[:space:]]+sqrtps[[:space:]]}}' --implicit-check-not='{{^[[:space:]]+divps[[:space:]]}}'
+; RUN: llc %t/fallback.ll -o - | FileCheck %s --check-prefix=FALLBACK
+
+; On i686 with SSE1 but no SSE2 or x87, v4f32 is legal while scalar f32
+; requires softening. The v4f32 cases create refinement constants before type
+; legalization. The v8f32 cases split first and create them during the
+; AfterLegalizeTypes combine. Assembly checks preserve working boundaries and
+; exclude fallback. The fallback input positively establishes the excluded
+; native operations. The companion PIC test checks refinement constant values,
+; relocation form, and local arithmetic use.
+
+;--- estimate.ll
+
+target triple = "i686-unknown-linux-gnu"
+
+; ASM-LABEL: rsqrt_v4_steps_0:
+; ASM:       # %bb.0:
+; ASM-NEXT:    rsqrtps %xmm1, %xmm1
+; ASM-NEXT:    mulps %xmm1, %xmm0
+; ASM-NEXT:    retl
+define <4 x float> @rsqrt_v4_steps_0(
+    <4 x float> %n, <4 x float> %x) #0 {
+  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+  %q = fdiv arcp ninf <4 x float> %n, %sqrt
+  ret <4 x float> %q
+}
+
+; ASM-LABEL: rsqrt_v4_default:
+; ASM:       rsqrtps
+; ASM-NOT:   {{^[[:space:]]+sqrtps}}
+; ASM-NOT:   divps
+; ASM:       retl
+define <4 x float> @rsqrt_v4_default(
+    <4 x float> %n, <4 x float> %x) #1 {
+  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+  %q = fdiv arcp ninf <4 x float> %n, %sqrt
+  ret <4 x float> %q
+}
+
+; ASM-LABEL: rsqrt_v4_steps_2:
+; ASM:       rsqrtps
+; ASM-NOT:   {{^[[:space:]]+sqrtps}}
+; ASM-NOT:   divps
+; ASM:       retl
+define <4 x float> @rsqrt_v4_steps_2(
+    <4 x float> %n, <4 x float> %x) #2 {
+  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+  %q = fdiv arcp ninf <4 x float> %n, %sqrt
+  ret <4 x float> %q
+}
+
+; ASM-LABEL: rsqrt_v8_steps_0:
+; ASM:         rsqrtps %xmm2, %xmm2
+; ASM-NEXT:    rsqrtps 16(%esp), %xmm3
+; ASM-NEXT:    mulps %xmm3, %xmm1
+; ASM-NEXT:    mulps %xmm2, %xmm0
+; ASM:         retl
+define <8 x float> @rsqrt_v8_steps_0(
+    <8 x float> %n, <8 x float> %x) #3 {
+  %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
+  %q = fdiv arcp ninf <8 x float> %n, %sqrt
+  ret <8 x float> %q
+}
+
+; ASM-LABEL: rsqrt_v8_default:
+; ASM-COUNT-2: rsqrtps
+; ASM-NOT:   {{^[[:space:]]+sqrtps}}
+; ASM-NOT:   divps
+; ASM:       retl
+define <8 x float> @rsqrt_v8_default(
+    <8 x float> %n, <8 x float> %x) #4 {
+  %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
+  %q = fdiv arcp ninf <8 x float> %n, %sqrt
+  ret <8 x float> %q
+}
+
+; ASM-LABEL: rsqrt_v8_steps_2:
+; ASM-COUNT-2: rsqrtps
+; ASM-NOT:   {{^[[:space:]]+sqrtps}}
+; ASM-NOT:   divps
+; ASM:       retl
+define <8 x float> @rsqrt_v8_steps_2(
+    <8 x float> %n, <8 x float> %x) #5 {
+  %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
+  %q = fdiv arcp ninf <8 x float> %n, %sqrt
+  ret <8 x float> %q
+}
+
+; ASM-LABEL: div_v4_steps_0:
+; ASM:       # %bb.0:
+; ASM-NEXT:    rcpps %xmm1, %xmm1
+; ASM-NEXT:    mulps %xmm1, %xmm0
+; ASM-NEXT:    retl
+define <4 x float> @div_v4_steps_0(
+    <4 x float> %n, <4 x float> %d) #6 {
+  %q = fdiv arcp ninf <4 x float> %n, %d
+  ret <4 x float> %q
+}
+
+; ASM-LABEL: div_v4_steps_1:
+; ASM:       # %bb.0:
+; ASM-NEXT:    rcpps %xmm1, %xmm2
+; ASM-NEXT:    movaps %xmm0, %xmm3
+; ASM-NEXT:    mulps %xmm2, %xmm3
+; ASM-NEXT:    mulps %xmm3, %xmm1
+; ASM-NEXT:    subps %xmm1, %xmm0
+; ASM-NEXT:    mulps %xmm2, %xmm0
+; ASM-NEXT:    addps %xmm3, %xmm0
+; ASM-NEXT:    retl
+define <4 x float> @div_v4_steps_1(
+    <4 x float> %n, <4 x float> %d) #7 {
+  %q = fdiv arcp ninf <4 x float> %n, %d
+  ret <4 x float> %q
+}
+
+; ASM-LABEL: div_v4_steps_2:
+; ASM:       rcpps
+; ASM-NOT:   divps
+; ASM:       retl
+define <4 x float> @div_v4_steps_2(
+    <4 x float> %n, <4 x float> %d) #8 {
+  %q = fdiv arcp ninf <4 x float> %n, %d
+  ret <4 x float> %q
+}
+
+; ASM-LABEL: div_v8_steps_0:
+; ASM:         rcpps 16(%esp), %xmm3
+; ASM-NEXT:    mulps %xmm3, %xmm1
+; ASM-NEXT:    rcpps %xmm2, %xmm2
+; ASM-NEXT:    mulps %xmm2, %xmm0
+; ASM:         retl
+define <8 x float> @div_v8_steps_0(
+    <8 x float> %n, <8 x float> %d) #9 {
+  %q = fdiv arcp ninf <8 x float> %n, %d
+  ret <8 x float> %q
+}
+
+; ASM-LABEL: div_v8_steps_1:
+; ASM:         rcpps %xmm2, %xmm3
+; ASM-NEXT:    movaps %xmm0, %xmm4
+; ASM-NEXT:    mulps %xmm3, %xmm4
+; ASM-NEXT:    mulps %xmm4, %xmm2
+; ASM-NEXT:    subps %xmm2, %xmm0
+; ASM-NEXT:    mulps %xmm3, %xmm0
+; ASM-NEXT:    addps %xmm4, %xmm0
+; ASM-NEXT:    movaps 16(%esp), %xmm2
+; ASM-NEXT:    rcpps %xmm2, %xmm3
+; ASM-NEXT:    movaps %xmm1, %xmm4
+; ASM-NEXT:    mulps %xmm3, %xmm4
+; ASM-NEXT:    mulps %xmm4, %xmm2
+; ASM-NEXT:    subps %xmm2, %xmm1
+; ASM-NEXT:    mulps %xmm3, %xmm1
+; ASM-NEXT:    addps %xmm4, %xmm1
+; ASM:         retl
+define <8 x float> @div_v8_steps_1(
+    <8 x float> %n, <8 x float> %d) #10 {
+  %q = fdiv arcp ninf <8 x float> %n, %d
+  ret <8 x float> %q
+}
+
+; ASM-LABEL: div_v8_steps_2:
+; ASM-COUNT-2: rcpps
+; ASM-NOT:   divps
+; ASM:       retl
+define <8 x float> @div_v8_steps_2(
+    <8 x float> %n, <8 x float> %d) #11 {
+  %q = fdiv arcp ninf <8 x float> %n, %d
+  ret <8 x float> %q
+}
+
+attributes #0 = {
+  "reciprocal-estimates"="vec-sqrtf:0"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #1 = {
+  "reciprocal-estimates"="vec-sqrtf"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #2 = {
+  "reciprocal-estimates"="vec-sqrtf:2"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #3 = {
+  "reciprocal-estimates"="vec-sqrtf:0"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #4 = {
+  "reciprocal-estimates"="vec-sqrtf"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #5 = {
+  "reciprocal-estimates"="vec-sqrtf:2"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #6 = {
+  "reciprocal-estimates"="vec-divf:0"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #7 = {
+  "reciprocal-estimates"="vec-divf:1"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #8 = {
+  "reciprocal-estimates"="vec-divf:2"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #9 = {
+  "reciprocal-estimates"="vec-divf:0"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #10 = {
+  "reciprocal-estimates"="vec-divf:1"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #11 = {
+  "reciprocal-estimates"="vec-divf:2"
+  "target-features"="+sse,-sse2,-x87"
+}
+
+;--- fallback.ll
+
+target triple = "i686-unknown-linux-gnu"
+
+; FALLBACK-LABEL: fallback_sqrt_v4:
+; FALLBACK: {{^[[:space:]]+sqrtps[[:space:]]}}
+; FALLBACK: {{^[[:space:]]+rcpps[[:space:]]}}
+define <4 x float> @fallback_sqrt_v4(
+    <4 x float> %n, <4 x float> %x) #0 {
+  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+  %q = fdiv arcp ninf <4 x float> %n, %sqrt
+  ret <4 x float> %q
+}
+
+; FALLBACK-LABEL: fallback_div_v4:
+; FALLBACK: {{^[[:space:]]+divps[[:space:]]}}
+define <4 x float> @fallback_div_v4(
+    <4 x float> %n, <4 x float> %d) #1 {
+  %q = fdiv arcp ninf <4 x float> %n, %d
+  ret <4 x float> %q
+}
+
+attributes #0 = {
+  "reciprocal-estimates"="!vec-sqrtf"
+  "target-features"="+sse,-sse2,-x87"
+}
+attributes #1 = {
+  "reciprocal-estimates"="!vec-divf"
+  "target-features"="+sse,-sse2,-x87"
+}

>From 6eb40c39b8e810b290f57fb3d248a649f08fc9d0 Mon Sep 17 00:00:00 2001
From: "Matt P. Dziubinski" <matt-p.dziubinski at hpe.com>
Date: Mon, 14 Sep 2026 18:02:21 -0500
Subject: [PATCH 2/2] [DAGCombiner] Comment the pool-load predicate and merge
 the X86 tests

Add a comment stating why a legal fixed-length vector with a softened
element type uses a constant-pool load.

Replace the two hand-written X86 tests with one test generated by
`update_llc_test_checks.py`. The test runs under the static and PIC
relocation models, and keeps the fallback functions.

Assisted-by: Claude Opus 5, GPT-5.6 Sol.
---
 llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp |   5 +
 ...ate-refinement-illegal-scalar-types-pic.ll | 159 -----
 ...stimate-refinement-illegal-scalar-types.ll | 584 +++++++++++++-----
 3 files changed, 424 insertions(+), 324 deletions(-)
 delete mode 100644 llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll

diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index 06dc5910d9f4b..7bf65cff2bd3e 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -31739,6 +31739,11 @@ SDValue DAGCombiner::BuildLogBase2(SDValue V, const SDLoc &DL,
 
 SDValue DAGCombiner::materializeFPConstant(double Value, const SDLoc &DL,
                                            EVT VT) {
+  // SelectionDAG::getConstantFP represents a fixed-length vector constant as a
+  // BUILD_VECTOR of scalar ConstantFP operands, and SoftenFloatOperand has no
+  // BUILD_VECTOR case. A legal fixed-length vector whose element type softens
+  // therefore takes a constant-pool load here. The vector legality check keeps
+  // the helper to that case. Every other type keeps getConstantFP.
   if (!VT.isFixedLengthVector() || !TLI.isTypeLegal(VT) ||
       TLI.getTypeAction(*DAG.getContext(), VT.getScalarType()) !=
           TargetLowering::TypeSoftenFloat)
diff --git a/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll
deleted file mode 100644
index 1cea8189a6467..0000000000000
--- a/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll
+++ /dev/null
@@ -1,159 +0,0 @@
-; NOTE: Do not autogenerate
-; RUN: split-file %s %t
-; RUN: llc -relocation-model=pic %t/estimate.ll -o - | FileCheck %s --check-prefix=PIC --enable-var-scope --implicit-check-not='{{^[[:space:]]+sqrtps[[:space:]]}}' --implicit-check-not='{{^[[:space:]]+divps[[:space:]]}}'
-; RUN: llc -relocation-model=pic %t/fallback.ll -o - | FileCheck %s --check-prefix=FALLBACK
-
-; This test owns refinement constant values and emitted PIC form. It binds each
-; pool definition to the canonical i386 call/pop/GOT expression and @GOTOFF
-; spelling. It checks immediate arithmetic where adjacent and otherwise follows
-; loaded registers to their arithmetic consumers in both vector halves.
-; Whole-input exclusions reject native fallback. The fallback input establishes
-; the excluded operations in the same configuration.
-
-;--- estimate.ll
-
-target triple = "i686-unknown-linux-gnu"
-
-; PIC: [[$V4_THREE:\.LCPI[0-9]+_[0-9]+]]:
-; PIC-NEXT: .long 0xc0400000
-; PIC-NEXT: .long 0xc0400000
-; PIC-NEXT: .long 0xc0400000
-; PIC-NEXT: .long 0xc0400000
-; PIC: [[$V4_HALF:\.LCPI[0-9]+_[0-9]+]]:
-; PIC-NEXT: .long 0xbf000000
-; PIC-NEXT: .long 0xbf000000
-; PIC-NEXT: .long 0xbf000000
-; PIC-NEXT: .long 0xbf000000
-; PIC-LABEL: rsqrt_v4_default:
-; PIC: calll [[PB:\.L[0-9]+\$pb]]
-; PIC: [[PB]]:
-; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
-; PIC: [[TMP:\.Ltmp[0-9]+]]:
-; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
-; PIC: rsqrtps
-; PIC: addps [[$V4_THREE]]@GOTOFF([[GOT]]), {{%xmm[0-9]+}}
-; PIC: mulps [[$V4_HALF]]@GOTOFF([[GOT]]), {{%xmm[0-9]+}}
-; PIC: retl
-define <4 x float> @rsqrt_v4_default(
-    <4 x float> %n, <4 x float> %x) #0 {
-  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
-  %q = fdiv arcp ninf <4 x float> %n, %sqrt
-  ret <4 x float> %q
-}
-
-; PIC: [[$V8_HALF:\.LCPI[0-9]+_[0-9]+]]:
-; PIC-NEXT: .long 0xbf000000
-; PIC-NEXT: .long 0xbf000000
-; PIC-NEXT: .long 0xbf000000
-; PIC-NEXT: .long 0xbf000000
-; PIC: [[$V8_THREE:\.LCPI[0-9]+_[0-9]+]]:
-; PIC-NEXT: .long 0xc0400000
-; PIC-NEXT: .long 0xc0400000
-; PIC-NEXT: .long 0xc0400000
-; PIC-NEXT: .long 0xc0400000
-; PIC-LABEL: rsqrt_v8_default:
-; PIC: calll [[PB:\.L[0-9]+\$pb]]
-; PIC: [[PB]]:
-; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
-; PIC: [[TMP:\.Ltmp[0-9]+]]:
-; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
-; PIC: rsqrtps
-; PIC-NEXT: movaps [[$V8_HALF]]@GOTOFF([[GOT]]), [[V8_HALF_REG:%xmm[0-9]+]]
-; PIC: mulps [[V8_HALF_REG]], {{%xmm[0-9]+}}
-; PIC: movaps [[$V8_THREE]]@GOTOFF([[GOT]]), [[V8_THREE_REG:%xmm[0-9]+]]
-; PIC-NEXT: addps [[V8_THREE_REG]], {{%xmm[0-9]+}}
-; PIC: rsqrtps
-; PIC: mulps {{%xmm[0-9]+}}, [[V8_HALF_REG]]
-; PIC: addps [[V8_THREE_REG]], {{%xmm[0-9]+}}
-; PIC: retl
-define <8 x float> @rsqrt_v8_default(
-    <8 x float> %n, <8 x float> %x) #0 {
-  %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
-  %q = fdiv arcp ninf <8 x float> %n, %sqrt
-  ret <8 x float> %q
-}
-
-; PIC: [[$V4_ONE:\.LCPI[0-9]+_[0-9]+]]:
-; PIC-NEXT: .long 0x3f800000
-; PIC-NEXT: .long 0x3f800000
-; PIC-NEXT: .long 0x3f800000
-; PIC-NEXT: .long 0x3f800000
-; PIC-LABEL: div_v4_steps_2:
-; PIC: calll [[PB:\.L[0-9]+\$pb]]
-; PIC: [[PB]]:
-; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
-; PIC: [[TMP:\.Ltmp[0-9]+]]:
-; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
-; PIC: rcpps
-; PIC: movaps [[$V4_ONE]]@GOTOFF([[GOT]]), [[V4_ONE_REG:%xmm[0-9]+]]
-; PIC-NEXT: subps {{%xmm[0-9]+}}, [[V4_ONE_REG]]
-; PIC: retl
-define <4 x float> @div_v4_steps_2(
-    <4 x float> %n, <4 x float> %d) #1 {
-  %q = fdiv arcp ninf <4 x float> %n, %d
-  ret <4 x float> %q
-}
-
-; PIC: [[$V8_ONE:\.LCPI[0-9]+_[0-9]+]]:
-; PIC-NEXT: .long 0x3f800000
-; PIC-NEXT: .long 0x3f800000
-; PIC-NEXT: .long 0x3f800000
-; PIC-NEXT: .long 0x3f800000
-; PIC-LABEL: div_v8_steps_2:
-; PIC: calll [[PB:\.L[0-9]+\$pb]]
-; PIC: [[PB]]:
-; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
-; PIC: [[TMP:\.Ltmp[0-9]+]]:
-; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
-; PIC: rcpps
-; PIC: movaps [[$V8_ONE]]@GOTOFF([[GOT]]), [[V8_ONE_REG:%xmm[0-9]+]]
-; PIC: movaps [[V8_ONE_REG]], [[V8_ONE_COPY:%xmm[0-9]+]]
-; PIC: subps {{%xmm[0-9]+}}, [[V8_ONE_COPY]]
-; PIC: rcpps
-; PIC: subps {{%xmm[0-9]+}}, [[V8_ONE_REG]]
-; PIC: retl
-define <8 x float> @div_v8_steps_2(
-    <8 x float> %n, <8 x float> %d) #1 {
-  %q = fdiv arcp ninf <8 x float> %n, %d
-  ret <8 x float> %q
-}
-
-attributes #0 = {
-  "reciprocal-estimates"="vec-sqrtf"
-  "target-features"="+sse,-sse2,-x87"
-}
-attributes #1 = {
-  "reciprocal-estimates"="vec-divf:2"
-  "target-features"="+sse,-sse2,-x87"
-}
-
-;--- fallback.ll
-
-target triple = "i686-unknown-linux-gnu"
-
-; FALLBACK-LABEL: fallback_sqrt_v4:
-; FALLBACK: {{^[[:space:]]+sqrtps[[:space:]]}}
-; FALLBACK: {{^[[:space:]]+rcpps[[:space:]]}}
-define <4 x float> @fallback_sqrt_v4(
-    <4 x float> %n, <4 x float> %x) #0 {
-  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
-  %q = fdiv arcp ninf <4 x float> %n, %sqrt
-  ret <4 x float> %q
-}
-
-; FALLBACK-LABEL: fallback_div_v4:
-; FALLBACK: {{^[[:space:]]+divps[[:space:]]}}
-define <4 x float> @fallback_div_v4(
-    <4 x float> %n, <4 x float> %d) #1 {
-  %q = fdiv arcp ninf <4 x float> %n, %d
-  ret <4 x float> %q
-}
-
-attributes #0 = {
-  "reciprocal-estimates"="!vec-sqrtf"
-  "target-features"="+sse,-sse2,-x87"
-}
-attributes #1 = {
-  "reciprocal-estimates"="!vec-divf"
-  "target-features"="+sse,-sse2,-x87"
-}
diff --git a/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll
index b55e40957622d..ce6a3d8ad02a7 100644
--- a/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll
+++ b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll
@@ -1,251 +1,505 @@
-; NOTE: Do not autogenerate
-; RUN: split-file %s %t
-; RUN: llc %t/estimate.ll -o - | FileCheck %s --check-prefix=ASM --enable-var-scope --implicit-check-not='{{^[[:space:]]+sqrtps[[:space:]]}}' --implicit-check-not='{{^[[:space:]]+divps[[:space:]]}}'
-; RUN: llc %t/fallback.ll -o - | FileCheck %s --check-prefix=FALLBACK
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s | FileCheck %s --check-prefixes=CHECK,STATIC
+; RUN: llc -relocation-model=pic < %s | FileCheck %s --check-prefixes=CHECK,PIC
 
-; On i686 with SSE1 but no SSE2 or x87, v4f32 is legal while scalar f32
-; requires softening. The v4f32 cases create refinement constants before type
-; legalization. The v8f32 cases split first and create them during the
-; AfterLegalizeTypes combine. Assembly checks preserve working boundaries and
-; exclude fallback. The fallback input positively establishes the excluded
-; native operations. The companion PIC test checks refinement constant values,
-; relocation form, and local arithmetic use.
-
-;--- estimate.ll
+; On i686 with SSE1 but no SSE2 or x87, v4f32 is legal while scalar f32 has
+; the TypeSoftenFloat action. The v4f32 functions create refinement constants
+; before type legalization. The v8f32 functions split first and create them in
+; the AfterLegalizeTypes combine. The refinement constants become constant-pool
+; loads created in DAGCombiner, so the PIC run checks that they take the GOT
+; relative form. The fallback functions disable one estimate each:
+; fallback_sqrt_v4 keeps the native sqrtps while its division still uses the
+; reciprocal estimate, and fallback_div_v4 keeps the native divps.
 
 target triple = "i686-unknown-linux-gnu"
 
-; ASM-LABEL: rsqrt_v4_steps_0:
-; ASM:       # %bb.0:
-; ASM-NEXT:    rsqrtps %xmm1, %xmm1
-; ASM-NEXT:    mulps %xmm1, %xmm0
-; ASM-NEXT:    retl
-define <4 x float> @rsqrt_v4_steps_0(
-    <4 x float> %n, <4 x float> %x) #0 {
+define <4 x float> @rsqrt_v4_steps_0(<4 x float> %n, <4 x float> %x) #0 {
+; CHECK-LABEL: rsqrt_v4_steps_0:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    rsqrtps %xmm1, %xmm1
+; CHECK-NEXT:    mulps %xmm1, %xmm0
+; CHECK-NEXT:    retl
   %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
   %q = fdiv arcp ninf <4 x float> %n, %sqrt
   ret <4 x float> %q
 }
 
-; ASM-LABEL: rsqrt_v4_default:
-; ASM:       rsqrtps
-; ASM-NOT:   {{^[[:space:]]+sqrtps}}
-; ASM-NOT:   divps
-; ASM:       retl
-define <4 x float> @rsqrt_v4_default(
-    <4 x float> %n, <4 x float> %x) #1 {
+define <4 x float> @rsqrt_v4_default(<4 x float> %n, <4 x float> %x) #1 {
+; STATIC-LABEL: rsqrt_v4_default:
+; STATIC:       # %bb.0:
+; STATIC-NEXT:    rsqrtps %xmm1, %xmm2
+; STATIC-NEXT:    mulps %xmm2, %xmm1
+; STATIC-NEXT:    mulps %xmm2, %xmm1
+; STATIC-NEXT:    addps {{\.?LCPI[0-9]+_[0-9]+}}, %xmm1
+; STATIC-NEXT:    mulps {{\.?LCPI[0-9]+_[0-9]+}}, %xmm2
+; STATIC-NEXT:    mulps %xmm1, %xmm2
+; STATIC-NEXT:    mulps %xmm2, %xmm0
+; STATIC-NEXT:    retl
+;
+; PIC-LABEL: rsqrt_v4_default:
+; PIC:       # %bb.0:
+; PIC-NEXT:    calll .L1$pb
+; PIC-NEXT:  .L1$pb:
+; PIC-NEXT:    popl %eax
+; PIC-NEXT:  .Ltmp0:
+; PIC-NEXT:    addl $_GLOBAL_OFFSET_TABLE_+(.Ltmp0-.L1$pb), %eax
+; PIC-NEXT:    rsqrtps %xmm1, %xmm2
+; PIC-NEXT:    mulps %xmm2, %xmm1
+; PIC-NEXT:    mulps %xmm2, %xmm1
+; PIC-NEXT:    addps {{\.?LCPI[0-9]+_[0-9]+}}@GOTOFF(%eax), %xmm1
+; PIC-NEXT:    mulps {{\.?LCPI[0-9]+_[0-9]+}}@GOTOFF(%eax), %xmm2
+; PIC-NEXT:    mulps %xmm1, %xmm2
+; PIC-NEXT:    mulps %xmm2, %xmm0
+; PIC-NEXT:    retl
   %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
   %q = fdiv arcp ninf <4 x float> %n, %sqrt
   ret <4 x float> %q
 }
 
-; ASM-LABEL: rsqrt_v4_steps_2:
-; ASM:       rsqrtps
-; ASM-NOT:   {{^[[:space:]]+sqrtps}}
-; ASM-NOT:   divps
-; ASM:       retl
-define <4 x float> @rsqrt_v4_steps_2(
-    <4 x float> %n, <4 x float> %x) #2 {
+define <4 x float> @rsqrt_v4_steps_2(<4 x float> %n, <4 x float> %x) #2 {
+; STATIC-LABEL: rsqrt_v4_steps_2:
+; STATIC:       # %bb.0:
+; STATIC-NEXT:    rsqrtps %xmm1, %xmm4
+; STATIC-NEXT:    movaps {{.*#+}} xmm3 = [-5.0E-1,-5.0E-1,-5.0E-1,-5.0E-1]
+; STATIC-NEXT:    movaps %xmm1, %xmm2
+; STATIC-NEXT:    mulps %xmm4, %xmm2
+; STATIC-NEXT:    mulps %xmm4, %xmm2
+; STATIC-NEXT:    mulps %xmm3, %xmm4
+; STATIC-NEXT:    movaps {{.*#+}} xmm5 = [-3.0E+0,-3.0E+0,-3.0E+0,-3.0E+0]
+; STATIC-NEXT:    addps %xmm5, %xmm2
+; STATIC-NEXT:    mulps %xmm4, %xmm2
+; STATIC-NEXT:    mulps %xmm2, %xmm1
+; STATIC-NEXT:    mulps %xmm2, %xmm1
+; STATIC-NEXT:    addps %xmm1, %xmm5
+; STATIC-NEXT:    mulps %xmm3, %xmm2
+; STATIC-NEXT:    mulps %xmm5, %xmm2
+; STATIC-NEXT:    mulps %xmm2, %xmm0
+; STATIC-NEXT:    retl
+;
+; PIC-LABEL: rsqrt_v4_steps_2:
+; PIC:       # %bb.0:
+; PIC-NEXT:    calll .L2$pb
+; PIC-NEXT:  .L2$pb:
+; PIC-NEXT:    popl %eax
+; PIC-NEXT:  .Ltmp1:
+; PIC-NEXT:    addl $_GLOBAL_OFFSET_TABLE_+(.Ltmp1-.L2$pb), %eax
+; PIC-NEXT:    rsqrtps %xmm1, %xmm4
+; PIC-NEXT:    movaps {{.*#+}} xmm3 = [-5.0E-1,-5.0E-1,-5.0E-1,-5.0E-1]
+; PIC-NEXT:    movaps %xmm1, %xmm2
+; PIC-NEXT:    mulps %xmm4, %xmm2
+; PIC-NEXT:    mulps %xmm4, %xmm2
+; PIC-NEXT:    mulps %xmm3, %xmm4
+; PIC-NEXT:    movaps {{.*#+}} xmm5 = [-3.0E+0,-3.0E+0,-3.0E+0,-3.0E+0]
+; PIC-NEXT:    addps %xmm5, %xmm2
+; PIC-NEXT:    mulps %xmm4, %xmm2
+; PIC-NEXT:    mulps %xmm2, %xmm1
+; PIC-NEXT:    mulps %xmm2, %xmm1
+; PIC-NEXT:    addps %xmm1, %xmm5
+; PIC-NEXT:    mulps %xmm3, %xmm2
+; PIC-NEXT:    mulps %xmm5, %xmm2
+; PIC-NEXT:    mulps %xmm2, %xmm0
+; PIC-NEXT:    retl
   %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
   %q = fdiv arcp ninf <4 x float> %n, %sqrt
   ret <4 x float> %q
 }
 
-; ASM-LABEL: rsqrt_v8_steps_0:
-; ASM:         rsqrtps %xmm2, %xmm2
-; ASM-NEXT:    rsqrtps 16(%esp), %xmm3
-; ASM-NEXT:    mulps %xmm3, %xmm1
-; ASM-NEXT:    mulps %xmm2, %xmm0
-; ASM:         retl
-define <8 x float> @rsqrt_v8_steps_0(
-    <8 x float> %n, <8 x float> %x) #3 {
+define <8 x float> @rsqrt_v8_steps_0(<8 x float> %n, <8 x float> %x) #0 {
+; CHECK-LABEL: rsqrt_v8_steps_0:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    subl $12, %esp
+; CHECK-NEXT:    rsqrtps %xmm2, %xmm2
+; CHECK-NEXT:    rsqrtps {{[0-9]+}}(%esp), %xmm3
+; CHECK-NEXT:    mulps %xmm3, %xmm1
+; CHECK-NEXT:    mulps %xmm2, %xmm0
+; CHECK-NEXT:    addl $12, %esp
+; CHECK-NEXT:    retl
   %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
   %q = fdiv arcp ninf <8 x float> %n, %sqrt
   ret <8 x float> %q
 }
 
-; ASM-LABEL: rsqrt_v8_default:
-; ASM-COUNT-2: rsqrtps
-; ASM-NOT:   {{^[[:space:]]+sqrtps}}
-; ASM-NOT:   divps
-; ASM:       retl
-define <8 x float> @rsqrt_v8_default(
-    <8 x float> %n, <8 x float> %x) #4 {
+define <8 x float> @rsqrt_v8_default(<8 x float> %n, <8 x float> %x) #1 {
+; STATIC-LABEL: rsqrt_v8_default:
+; STATIC:       # %bb.0:
+; STATIC-NEXT:    subl $12, %esp
+; STATIC-NEXT:    movaps {{[0-9]+}}(%esp), %xmm3
+; STATIC-NEXT:    rsqrtps %xmm3, %xmm5
+; STATIC-NEXT:    movaps {{.*#+}} xmm4 = [-5.0E-1,-5.0E-1,-5.0E-1,-5.0E-1]
+; STATIC-NEXT:    mulps %xmm5, %xmm3
+; STATIC-NEXT:    mulps %xmm5, %xmm3
+; STATIC-NEXT:    mulps %xmm4, %xmm5
+; STATIC-NEXT:    movaps {{.*#+}} xmm6 = [-3.0E+0,-3.0E+0,-3.0E+0,-3.0E+0]
+; STATIC-NEXT:    addps %xmm6, %xmm3
+; STATIC-NEXT:    mulps %xmm5, %xmm3
+; STATIC-NEXT:    rsqrtps %xmm2, %xmm5
+; STATIC-NEXT:    mulps %xmm5, %xmm4
+; STATIC-NEXT:    mulps %xmm5, %xmm2
+; STATIC-NEXT:    mulps %xmm2, %xmm5
+; STATIC-NEXT:    addps %xmm6, %xmm5
+; STATIC-NEXT:    mulps %xmm4, %xmm5
+; STATIC-NEXT:    mulps %xmm5, %xmm0
+; STATIC-NEXT:    mulps %xmm3, %xmm1
+; STATIC-NEXT:    addl $12, %esp
+; STATIC-NEXT:    retl
+;
+; PIC-LABEL: rsqrt_v8_default:
+; PIC:       # %bb.0:
+; PIC-NEXT:    subl $12, %esp
+; PIC-NEXT:    calll .L4$pb
+; PIC-NEXT:  .L4$pb:
+; PIC-NEXT:    popl %eax
+; PIC-NEXT:  .Ltmp2:
+; PIC-NEXT:    addl $_GLOBAL_OFFSET_TABLE_+(.Ltmp2-.L4$pb), %eax
+; PIC-NEXT:    movaps {{[0-9]+}}(%esp), %xmm3
+; PIC-NEXT:    rsqrtps %xmm3, %xmm5
+; PIC-NEXT:    movaps {{.*#+}} xmm4 = [-5.0E-1,-5.0E-1,-5.0E-1,-5.0E-1]
+; PIC-NEXT:    mulps %xmm5, %xmm3
+; PIC-NEXT:    mulps %xmm5, %xmm3
+; PIC-NEXT:    mulps %xmm4, %xmm5
+; PIC-NEXT:    movaps {{.*#+}} xmm6 = [-3.0E+0,-3.0E+0,-3.0E+0,-3.0E+0]
+; PIC-NEXT:    addps %xmm6, %xmm3
+; PIC-NEXT:    mulps %xmm5, %xmm3
+; PIC-NEXT:    rsqrtps %xmm2, %xmm5
+; PIC-NEXT:    mulps %xmm5, %xmm4
+; PIC-NEXT:    mulps %xmm5, %xmm2
+; PIC-NEXT:    mulps %xmm2, %xmm5
+; PIC-NEXT:    addps %xmm6, %xmm5
+; PIC-NEXT:    mulps %xmm4, %xmm5
+; PIC-NEXT:    mulps %xmm5, %xmm0
+; PIC-NEXT:    mulps %xmm3, %xmm1
+; PIC-NEXT:    addl $12, %esp
+; PIC-NEXT:    retl
   %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
   %q = fdiv arcp ninf <8 x float> %n, %sqrt
   ret <8 x float> %q
 }
 
-; ASM-LABEL: rsqrt_v8_steps_2:
-; ASM-COUNT-2: rsqrtps
-; ASM-NOT:   {{^[[:space:]]+sqrtps}}
-; ASM-NOT:   divps
-; ASM:       retl
-define <8 x float> @rsqrt_v8_steps_2(
-    <8 x float> %n, <8 x float> %x) #5 {
+define <8 x float> @rsqrt_v8_steps_2(<8 x float> %n, <8 x float> %x) #2 {
+; STATIC-LABEL: rsqrt_v8_steps_2:
+; STATIC:       # %bb.0:
+; STATIC-NEXT:    subl $12, %esp
+; STATIC-NEXT:    movaps {{[0-9]+}}(%esp), %xmm6
+; STATIC-NEXT:    rsqrtps %xmm6, %xmm7
+; STATIC-NEXT:    movaps {{.*#+}} xmm4 = [-5.0E-1,-5.0E-1,-5.0E-1,-5.0E-1]
+; STATIC-NEXT:    movaps %xmm6, %xmm3
+; STATIC-NEXT:    mulps %xmm7, %xmm3
+; STATIC-NEXT:    mulps %xmm7, %xmm3
+; STATIC-NEXT:    mulps %xmm4, %xmm7
+; STATIC-NEXT:    movaps {{.*#+}} xmm5 = [-3.0E+0,-3.0E+0,-3.0E+0,-3.0E+0]
+; STATIC-NEXT:    addps %xmm5, %xmm3
+; STATIC-NEXT:    mulps %xmm7, %xmm3
+; STATIC-NEXT:    mulps %xmm3, %xmm6
+; STATIC-NEXT:    mulps %xmm3, %xmm6
+; STATIC-NEXT:    addps %xmm5, %xmm6
+; STATIC-NEXT:    mulps %xmm4, %xmm3
+; STATIC-NEXT:    mulps %xmm6, %xmm3
+; STATIC-NEXT:    rsqrtps %xmm2, %xmm7
+; STATIC-NEXT:    movaps %xmm2, %xmm6
+; STATIC-NEXT:    mulps %xmm7, %xmm6
+; STATIC-NEXT:    mulps %xmm7, %xmm6
+; STATIC-NEXT:    mulps %xmm4, %xmm7
+; STATIC-NEXT:    addps %xmm5, %xmm6
+; STATIC-NEXT:    mulps %xmm7, %xmm6
+; STATIC-NEXT:    mulps %xmm6, %xmm2
+; STATIC-NEXT:    mulps %xmm6, %xmm2
+; STATIC-NEXT:    addps %xmm2, %xmm5
+; STATIC-NEXT:    mulps %xmm4, %xmm6
+; STATIC-NEXT:    mulps %xmm5, %xmm6
+; STATIC-NEXT:    mulps %xmm6, %xmm0
+; STATIC-NEXT:    mulps %xmm3, %xmm1
+; STATIC-NEXT:    addl $12, %esp
+; STATIC-NEXT:    retl
+;
+; PIC-LABEL: rsqrt_v8_steps_2:
+; PIC:       # %bb.0:
+; PIC-NEXT:    subl $12, %esp
+; PIC-NEXT:    calll .L5$pb
+; PIC-NEXT:  .L5$pb:
+; PIC-NEXT:    popl %eax
+; PIC-NEXT:  .Ltmp3:
+; PIC-NEXT:    addl $_GLOBAL_OFFSET_TABLE_+(.Ltmp3-.L5$pb), %eax
+; PIC-NEXT:    movaps {{[0-9]+}}(%esp), %xmm6
+; PIC-NEXT:    rsqrtps %xmm6, %xmm7
+; PIC-NEXT:    movaps {{.*#+}} xmm4 = [-5.0E-1,-5.0E-1,-5.0E-1,-5.0E-1]
+; PIC-NEXT:    movaps %xmm6, %xmm3
+; PIC-NEXT:    mulps %xmm7, %xmm3
+; PIC-NEXT:    mulps %xmm7, %xmm3
+; PIC-NEXT:    mulps %xmm4, %xmm7
+; PIC-NEXT:    movaps {{.*#+}} xmm5 = [-3.0E+0,-3.0E+0,-3.0E+0,-3.0E+0]
+; PIC-NEXT:    addps %xmm5, %xmm3
+; PIC-NEXT:    mulps %xmm7, %xmm3
+; PIC-NEXT:    mulps %xmm3, %xmm6
+; PIC-NEXT:    mulps %xmm3, %xmm6
+; PIC-NEXT:    addps %xmm5, %xmm6
+; PIC-NEXT:    mulps %xmm4, %xmm3
+; PIC-NEXT:    mulps %xmm6, %xmm3
+; PIC-NEXT:    rsqrtps %xmm2, %xmm7
+; PIC-NEXT:    movaps %xmm2, %xmm6
+; PIC-NEXT:    mulps %xmm7, %xmm6
+; PIC-NEXT:    mulps %xmm7, %xmm6
+; PIC-NEXT:    mulps %xmm4, %xmm7
+; PIC-NEXT:    addps %xmm5, %xmm6
+; PIC-NEXT:    mulps %xmm7, %xmm6
+; PIC-NEXT:    mulps %xmm6, %xmm2
+; PIC-NEXT:    mulps %xmm6, %xmm2
+; PIC-NEXT:    addps %xmm2, %xmm5
+; PIC-NEXT:    mulps %xmm4, %xmm6
+; PIC-NEXT:    mulps %xmm5, %xmm6
+; PIC-NEXT:    mulps %xmm6, %xmm0
+; PIC-NEXT:    mulps %xmm3, %xmm1
+; PIC-NEXT:    addl $12, %esp
+; PIC-NEXT:    retl
   %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
   %q = fdiv arcp ninf <8 x float> %n, %sqrt
   ret <8 x float> %q
 }
 
-; ASM-LABEL: div_v4_steps_0:
-; ASM:       # %bb.0:
-; ASM-NEXT:    rcpps %xmm1, %xmm1
-; ASM-NEXT:    mulps %xmm1, %xmm0
-; ASM-NEXT:    retl
-define <4 x float> @div_v4_steps_0(
-    <4 x float> %n, <4 x float> %d) #6 {
+define <4 x float> @div_v4_steps_0(<4 x float> %n, <4 x float> %d) #3 {
+; CHECK-LABEL: div_v4_steps_0:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    rcpps %xmm1, %xmm1
+; CHECK-NEXT:    mulps %xmm1, %xmm0
+; CHECK-NEXT:    retl
   %q = fdiv arcp ninf <4 x float> %n, %d
   ret <4 x float> %q
 }
 
-; ASM-LABEL: div_v4_steps_1:
-; ASM:       # %bb.0:
-; ASM-NEXT:    rcpps %xmm1, %xmm2
-; ASM-NEXT:    movaps %xmm0, %xmm3
-; ASM-NEXT:    mulps %xmm2, %xmm3
-; ASM-NEXT:    mulps %xmm3, %xmm1
-; ASM-NEXT:    subps %xmm1, %xmm0
-; ASM-NEXT:    mulps %xmm2, %xmm0
-; ASM-NEXT:    addps %xmm3, %xmm0
-; ASM-NEXT:    retl
-define <4 x float> @div_v4_steps_1(
-    <4 x float> %n, <4 x float> %d) #7 {
+define <4 x float> @div_v4_steps_1(<4 x float> %n, <4 x float> %d) #4 {
+; CHECK-LABEL: div_v4_steps_1:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    rcpps %xmm1, %xmm2
+; CHECK-NEXT:    movaps %xmm0, %xmm3
+; CHECK-NEXT:    mulps %xmm2, %xmm3
+; CHECK-NEXT:    mulps %xmm3, %xmm1
+; CHECK-NEXT:    subps %xmm1, %xmm0
+; CHECK-NEXT:    mulps %xmm2, %xmm0
+; CHECK-NEXT:    addps %xmm3, %xmm0
+; CHECK-NEXT:    retl
   %q = fdiv arcp ninf <4 x float> %n, %d
   ret <4 x float> %q
 }
 
-; ASM-LABEL: div_v4_steps_2:
-; ASM:       rcpps
-; ASM-NOT:   divps
-; ASM:       retl
-define <4 x float> @div_v4_steps_2(
-    <4 x float> %n, <4 x float> %d) #8 {
+define <4 x float> @div_v4_steps_2(<4 x float> %n, <4 x float> %d) #5 {
+; STATIC-LABEL: div_v4_steps_2:
+; STATIC:       # %bb.0:
+; STATIC-NEXT:    rcpps %xmm1, %xmm2
+; STATIC-NEXT:    movaps %xmm1, %xmm3
+; STATIC-NEXT:    mulps %xmm2, %xmm3
+; STATIC-NEXT:    movaps {{.*#+}} xmm4 = [1.0E+0,1.0E+0,1.0E+0,1.0E+0]
+; STATIC-NEXT:    subps %xmm3, %xmm4
+; STATIC-NEXT:    mulps %xmm2, %xmm4
+; STATIC-NEXT:    addps %xmm2, %xmm4
+; STATIC-NEXT:    movaps %xmm0, %xmm2
+; STATIC-NEXT:    mulps %xmm4, %xmm2
+; STATIC-NEXT:    mulps %xmm2, %xmm1
+; STATIC-NEXT:    subps %xmm1, %xmm0
+; STATIC-NEXT:    mulps %xmm4, %xmm0
+; STATIC-NEXT:    addps %xmm2, %xmm0
+; STATIC-NEXT:    retl
+;
+; PIC-LABEL: div_v4_steps_2:
+; PIC:       # %bb.0:
+; PIC-NEXT:    calll .L8$pb
+; PIC-NEXT:  .L8$pb:
+; PIC-NEXT:    popl %eax
+; PIC-NEXT:  .Ltmp4:
+; PIC-NEXT:    addl $_GLOBAL_OFFSET_TABLE_+(.Ltmp4-.L8$pb), %eax
+; PIC-NEXT:    rcpps %xmm1, %xmm2
+; PIC-NEXT:    movaps %xmm1, %xmm3
+; PIC-NEXT:    mulps %xmm2, %xmm3
+; PIC-NEXT:    movaps {{.*#+}} xmm4 = [1.0E+0,1.0E+0,1.0E+0,1.0E+0]
+; PIC-NEXT:    subps %xmm3, %xmm4
+; PIC-NEXT:    mulps %xmm2, %xmm4
+; PIC-NEXT:    addps %xmm2, %xmm4
+; PIC-NEXT:    movaps %xmm0, %xmm2
+; PIC-NEXT:    mulps %xmm4, %xmm2
+; PIC-NEXT:    mulps %xmm2, %xmm1
+; PIC-NEXT:    subps %xmm1, %xmm0
+; PIC-NEXT:    mulps %xmm4, %xmm0
+; PIC-NEXT:    addps %xmm2, %xmm0
+; PIC-NEXT:    retl
   %q = fdiv arcp ninf <4 x float> %n, %d
   ret <4 x float> %q
 }
 
-; ASM-LABEL: div_v8_steps_0:
-; ASM:         rcpps 16(%esp), %xmm3
-; ASM-NEXT:    mulps %xmm3, %xmm1
-; ASM-NEXT:    rcpps %xmm2, %xmm2
-; ASM-NEXT:    mulps %xmm2, %xmm0
-; ASM:         retl
-define <8 x float> @div_v8_steps_0(
-    <8 x float> %n, <8 x float> %d) #9 {
+define <8 x float> @div_v8_steps_0(<8 x float> %n, <8 x float> %d) #3 {
+; CHECK-LABEL: div_v8_steps_0:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    subl $12, %esp
+; CHECK-NEXT:    rcpps {{[0-9]+}}(%esp), %xmm3
+; CHECK-NEXT:    mulps %xmm3, %xmm1
+; CHECK-NEXT:    rcpps %xmm2, %xmm2
+; CHECK-NEXT:    mulps %xmm2, %xmm0
+; CHECK-NEXT:    addl $12, %esp
+; CHECK-NEXT:    retl
   %q = fdiv arcp ninf <8 x float> %n, %d
   ret <8 x float> %q
 }
 
-; ASM-LABEL: div_v8_steps_1:
-; ASM:         rcpps %xmm2, %xmm3
-; ASM-NEXT:    movaps %xmm0, %xmm4
-; ASM-NEXT:    mulps %xmm3, %xmm4
-; ASM-NEXT:    mulps %xmm4, %xmm2
-; ASM-NEXT:    subps %xmm2, %xmm0
-; ASM-NEXT:    mulps %xmm3, %xmm0
-; ASM-NEXT:    addps %xmm4, %xmm0
-; ASM-NEXT:    movaps 16(%esp), %xmm2
-; ASM-NEXT:    rcpps %xmm2, %xmm3
-; ASM-NEXT:    movaps %xmm1, %xmm4
-; ASM-NEXT:    mulps %xmm3, %xmm4
-; ASM-NEXT:    mulps %xmm4, %xmm2
-; ASM-NEXT:    subps %xmm2, %xmm1
-; ASM-NEXT:    mulps %xmm3, %xmm1
-; ASM-NEXT:    addps %xmm4, %xmm1
-; ASM:         retl
-define <8 x float> @div_v8_steps_1(
-    <8 x float> %n, <8 x float> %d) #10 {
+define <8 x float> @div_v8_steps_1(<8 x float> %n, <8 x float> %d) #4 {
+; CHECK-LABEL: div_v8_steps_1:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    subl $12, %esp
+; CHECK-NEXT:    rcpps %xmm2, %xmm3
+; CHECK-NEXT:    movaps %xmm0, %xmm4
+; CHECK-NEXT:    mulps %xmm3, %xmm4
+; CHECK-NEXT:    mulps %xmm4, %xmm2
+; CHECK-NEXT:    subps %xmm2, %xmm0
+; CHECK-NEXT:    mulps %xmm3, %xmm0
+; CHECK-NEXT:    addps %xmm4, %xmm0
+; CHECK-NEXT:    movaps {{[0-9]+}}(%esp), %xmm2
+; CHECK-NEXT:    rcpps %xmm2, %xmm3
+; CHECK-NEXT:    movaps %xmm1, %xmm4
+; CHECK-NEXT:    mulps %xmm3, %xmm4
+; CHECK-NEXT:    mulps %xmm4, %xmm2
+; CHECK-NEXT:    subps %xmm2, %xmm1
+; CHECK-NEXT:    mulps %xmm3, %xmm1
+; CHECK-NEXT:    addps %xmm4, %xmm1
+; CHECK-NEXT:    addl $12, %esp
+; CHECK-NEXT:    retl
   %q = fdiv arcp ninf <8 x float> %n, %d
   ret <8 x float> %q
 }
 
-; ASM-LABEL: div_v8_steps_2:
-; ASM-COUNT-2: rcpps
-; ASM-NOT:   divps
-; ASM:       retl
-define <8 x float> @div_v8_steps_2(
-    <8 x float> %n, <8 x float> %d) #11 {
+define <8 x float> @div_v8_steps_2(<8 x float> %n, <8 x float> %d) #5 {
+; STATIC-LABEL: div_v8_steps_2:
+; STATIC:       # %bb.0:
+; STATIC-NEXT:    subl $12, %esp
+; STATIC-NEXT:    rcpps %xmm2, %xmm4
+; STATIC-NEXT:    movaps %xmm2, %xmm5
+; STATIC-NEXT:    mulps %xmm4, %xmm5
+; STATIC-NEXT:    movaps {{.*#+}} xmm3 = [1.0E+0,1.0E+0,1.0E+0,1.0E+0]
+; STATIC-NEXT:    movaps %xmm3, %xmm6
+; STATIC-NEXT:    subps %xmm5, %xmm6
+; STATIC-NEXT:    mulps %xmm4, %xmm6
+; STATIC-NEXT:    addps %xmm4, %xmm6
+; STATIC-NEXT:    movaps %xmm0, %xmm4
+; STATIC-NEXT:    mulps %xmm6, %xmm4
+; STATIC-NEXT:    mulps %xmm4, %xmm2
+; STATIC-NEXT:    subps %xmm2, %xmm0
+; STATIC-NEXT:    mulps %xmm6, %xmm0
+; STATIC-NEXT:    addps %xmm4, %xmm0
+; STATIC-NEXT:    movaps {{[0-9]+}}(%esp), %xmm2
+; STATIC-NEXT:    rcpps %xmm2, %xmm4
+; STATIC-NEXT:    movaps %xmm2, %xmm5
+; STATIC-NEXT:    mulps %xmm4, %xmm5
+; STATIC-NEXT:    subps %xmm5, %xmm3
+; STATIC-NEXT:    mulps %xmm4, %xmm3
+; STATIC-NEXT:    addps %xmm4, %xmm3
+; STATIC-NEXT:    movaps %xmm1, %xmm4
+; STATIC-NEXT:    mulps %xmm3, %xmm4
+; STATIC-NEXT:    mulps %xmm4, %xmm2
+; STATIC-NEXT:    subps %xmm2, %xmm1
+; STATIC-NEXT:    mulps %xmm3, %xmm1
+; STATIC-NEXT:    addps %xmm4, %xmm1
+; STATIC-NEXT:    addl $12, %esp
+; STATIC-NEXT:    retl
+;
+; PIC-LABEL: div_v8_steps_2:
+; PIC:       # %bb.0:
+; PIC-NEXT:    subl $12, %esp
+; PIC-NEXT:    calll .L11$pb
+; PIC-NEXT:  .L11$pb:
+; PIC-NEXT:    popl %eax
+; PIC-NEXT:  .Ltmp5:
+; PIC-NEXT:    addl $_GLOBAL_OFFSET_TABLE_+(.Ltmp5-.L11$pb), %eax
+; PIC-NEXT:    rcpps %xmm2, %xmm4
+; PIC-NEXT:    movaps %xmm2, %xmm5
+; PIC-NEXT:    mulps %xmm4, %xmm5
+; PIC-NEXT:    movaps {{.*#+}} xmm3 = [1.0E+0,1.0E+0,1.0E+0,1.0E+0]
+; PIC-NEXT:    movaps %xmm3, %xmm6
+; PIC-NEXT:    subps %xmm5, %xmm6
+; PIC-NEXT:    mulps %xmm4, %xmm6
+; PIC-NEXT:    addps %xmm4, %xmm6
+; PIC-NEXT:    movaps %xmm0, %xmm4
+; PIC-NEXT:    mulps %xmm6, %xmm4
+; PIC-NEXT:    mulps %xmm4, %xmm2
+; PIC-NEXT:    subps %xmm2, %xmm0
+; PIC-NEXT:    mulps %xmm6, %xmm0
+; PIC-NEXT:    addps %xmm4, %xmm0
+; PIC-NEXT:    movaps {{[0-9]+}}(%esp), %xmm2
+; PIC-NEXT:    rcpps %xmm2, %xmm4
+; PIC-NEXT:    movaps %xmm2, %xmm5
+; PIC-NEXT:    mulps %xmm4, %xmm5
+; PIC-NEXT:    subps %xmm5, %xmm3
+; PIC-NEXT:    mulps %xmm4, %xmm3
+; PIC-NEXT:    addps %xmm4, %xmm3
+; PIC-NEXT:    movaps %xmm1, %xmm4
+; PIC-NEXT:    mulps %xmm3, %xmm4
+; PIC-NEXT:    mulps %xmm4, %xmm2
+; PIC-NEXT:    subps %xmm2, %xmm1
+; PIC-NEXT:    mulps %xmm3, %xmm1
+; PIC-NEXT:    addps %xmm4, %xmm1
+; PIC-NEXT:    addl $12, %esp
+; PIC-NEXT:    retl
   %q = fdiv arcp ninf <8 x float> %n, %d
   ret <8 x float> %q
 }
 
+define <4 x float> @fallback_sqrt_v4(<4 x float> %n, <4 x float> %x) #6 {
+; CHECK-LABEL: fallback_sqrt_v4:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    sqrtps %xmm1, %xmm1
+; CHECK-NEXT:    rcpps %xmm1, %xmm2
+; CHECK-NEXT:    movaps %xmm0, %xmm3
+; CHECK-NEXT:    mulps %xmm2, %xmm3
+; CHECK-NEXT:    mulps %xmm3, %xmm1
+; CHECK-NEXT:    subps %xmm1, %xmm0
+; CHECK-NEXT:    mulps %xmm2, %xmm0
+; CHECK-NEXT:    addps %xmm3, %xmm0
+; CHECK-NEXT:    retl
+  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+  %q = fdiv arcp ninf <4 x float> %n, %sqrt
+  ret <4 x float> %q
+}
+
+define <4 x float> @fallback_div_v4(<4 x float> %n, <4 x float> %d) #7 {
+; CHECK-LABEL: fallback_div_v4:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    divps %xmm1, %xmm0
+; CHECK-NEXT:    retl
+  %q = fdiv arcp ninf <4 x float> %n, %d
+  ret <4 x float> %q
+}
+
 attributes #0 = {
+  nounwind
   "reciprocal-estimates"="vec-sqrtf:0"
   "target-features"="+sse,-sse2,-x87"
 }
 attributes #1 = {
+  nounwind
   "reciprocal-estimates"="vec-sqrtf"
   "target-features"="+sse,-sse2,-x87"
 }
 attributes #2 = {
+  nounwind
   "reciprocal-estimates"="vec-sqrtf:2"
   "target-features"="+sse,-sse2,-x87"
 }
 attributes #3 = {
-  "reciprocal-estimates"="vec-sqrtf:0"
-  "target-features"="+sse,-sse2,-x87"
-}
-attributes #4 = {
-  "reciprocal-estimates"="vec-sqrtf"
-  "target-features"="+sse,-sse2,-x87"
-}
-attributes #5 = {
-  "reciprocal-estimates"="vec-sqrtf:2"
-  "target-features"="+sse,-sse2,-x87"
-}
-attributes #6 = {
+  nounwind
   "reciprocal-estimates"="vec-divf:0"
   "target-features"="+sse,-sse2,-x87"
 }
-attributes #7 = {
-  "reciprocal-estimates"="vec-divf:1"
-  "target-features"="+sse,-sse2,-x87"
-}
-attributes #8 = {
-  "reciprocal-estimates"="vec-divf:2"
-  "target-features"="+sse,-sse2,-x87"
-}
-attributes #9 = {
-  "reciprocal-estimates"="vec-divf:0"
-  "target-features"="+sse,-sse2,-x87"
-}
-attributes #10 = {
+attributes #4 = {
+  nounwind
   "reciprocal-estimates"="vec-divf:1"
   "target-features"="+sse,-sse2,-x87"
 }
-attributes #11 = {
+attributes #5 = {
+  nounwind
   "reciprocal-estimates"="vec-divf:2"
   "target-features"="+sse,-sse2,-x87"
 }
-
-;--- fallback.ll
-
-target triple = "i686-unknown-linux-gnu"
-
-; FALLBACK-LABEL: fallback_sqrt_v4:
-; FALLBACK: {{^[[:space:]]+sqrtps[[:space:]]}}
-; FALLBACK: {{^[[:space:]]+rcpps[[:space:]]}}
-define <4 x float> @fallback_sqrt_v4(
-    <4 x float> %n, <4 x float> %x) #0 {
-  %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
-  %q = fdiv arcp ninf <4 x float> %n, %sqrt
-  ret <4 x float> %q
-}
-
-; FALLBACK-LABEL: fallback_div_v4:
-; FALLBACK: {{^[[:space:]]+divps[[:space:]]}}
-define <4 x float> @fallback_div_v4(
-    <4 x float> %n, <4 x float> %d) #1 {
-  %q = fdiv arcp ninf <4 x float> %n, %d
-  ret <4 x float> %q
-}
-
-attributes #0 = {
+attributes #6 = {
+  nounwind
   "reciprocal-estimates"="!vec-sqrtf"
   "target-features"="+sse,-sse2,-x87"
 }
-attributes #1 = {
+attributes #7 = {
+  nounwind
   "reciprocal-estimates"="!vec-divf"
   "target-features"="+sse,-sse2,-x87"
 }



More information about the llvm-commits mailing list