[llvm] [DAGCombiner] Avoid illegal scalar nodes in estimate refinement (PR #219394)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 28 00:01:42 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-selectiondag
Author: Matt (MattPD)
<details>
<summary>Changes</summary>
On i686 with SSE1 but without SSE2 or x87, `v4f32` is legal while scalar `f32` has the `TypeSoftenFloat` action. `SelectionDAG::getConstantFP` represents a vector constant as a `BUILD_VECTOR` of scalar `ConstantFPSDNode` operands. Estimate refinement can therefore introduce illegal scalar nodes into a legal vector.
Refinement can create these constants before type legalization for `v4f32` and during the `AfterLegalizeTypes` combine after type legalization splits `v8f32`. A change limited to the type legalizer misses the second phase. A direct softened-integer representation of these constants would introduce illegal `v4i32` nodes in this configuration.
Add a `DAGCombiner` helper that materializes affected refinement constants as constant-pool vector loads when the value type is a legal fixed-length vector whose scalar element action is `TypeSoftenFloat`. All other types continue to use `SelectionDAG::getConstantFP`.
Use the helper to create the `1.0` constant in `BuildDivEstimate` and the `-3.0` and `-0.5` constants in `buildSqrtNRTwoConst`. This preserves requested reciprocal and reciprocal-square-root refinement in both creation phases.
Fixes https://github.com/llvm/llvm-project/issues/217806
Assisted-by: Claude Opus 5, GPT-5.6 Sol.
---
Patch is 26.87 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/219394.diff
3 Files Affected:
- (modified) llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp (+28-3)
- (added) llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll (+160)
- (added) llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll (+422)
``````````diff
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index a9db383ede385..06dc5910d9f4b 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -653,6 +653,11 @@ namespace {
SDValue BuildUDIV(SDNode *N);
SDValue BuildSREMPow2(SDNode *N);
SDValue buildOptimizedSREM(SDValue N0, SDValue N1, SDNode *N);
+ /// Materialize an FP constant for estimate refinement. Before operation
+ /// legalization, the softened-vector path returns an opaque load, so
+ /// callers requiring constant recognition, folding, or SDValue identity
+ /// must not use this helper.
+ SDValue materializeFPConstant(double Value, const SDLoc &DL, EVT VT);
SDValue BuildLogBase2(SDValue V, const SDLoc &DL,
bool KnownNeverZero = false,
bool InexpensiveOnly = false,
@@ -31732,6 +31737,26 @@ SDValue DAGCombiner::BuildLogBase2(SDValue V, const SDLoc &DL,
return LogBase2;
}
+SDValue DAGCombiner::materializeFPConstant(double Value, const SDLoc &DL,
+ EVT VT) {
+ if (!VT.isFixedLengthVector() || !TLI.isTypeLegal(VT) ||
+ TLI.getTypeAction(*DAG.getContext(), VT.getScalarType()) !=
+ TargetLowering::TypeSoftenFloat)
+ return DAG.getConstantFP(Value, DL, VT);
+
+ assert(!LegalDAG && "load must be created before operation legalization");
+ Type *ScalarTy = VT.getScalarType().getTypeForEVT(*DAG.getContext());
+ Constant *Scalar = ConstantFP::get(ScalarTy, Value);
+ Constant *Vector =
+ ConstantVector::getSplat(VT.getVectorElementCount(), Scalar);
+ SDValue CPIdx =
+ DAG.getConstantPool(Vector, TLI.getPointerTy(DAG.getDataLayout()));
+ Align Alignment = cast<ConstantPoolSDNode>(CPIdx)->getAlign();
+ return DAG.getLoad(
+ VT, DL, DAG.getEntryNode(), CPIdx,
+ MachinePointerInfo::getConstantPool(DAG.getMachineFunction()), Alignment);
+}
+
/// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
/// For the reciprocal, we need to find the zero of the function:
/// F(X) = 1/X - A [which has a zero at X = 1/A]
@@ -31765,7 +31790,7 @@ SDValue DAGCombiner::BuildDivEstimate(SDValue N, SDValue Op,
SDLoc DL(Op);
if (Iterations) {
- SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
+ SDValue FPOne = materializeFPConstant(1.0, DL, VT);
// Newton iterations: Est = Est + Est (N - Arg * Est)
// If this is the last iteration, also multiply by the numerator.
@@ -31843,8 +31868,8 @@ SDValue DAGCombiner::buildSqrtNRTwoConst(SDValue Arg, SDValue Est,
unsigned Iterations, bool Reciprocal) {
EVT VT = Arg.getValueType();
SDLoc DL(Arg);
- SDValue MinusThree = DAG.getConstantFP(-3.0, DL, VT);
- SDValue MinusHalf = DAG.getConstantFP(-0.5, DL, VT);
+ SDValue MinusThree = materializeFPConstant(-3.0, DL, VT);
+ SDValue MinusHalf = materializeFPConstant(-0.5, DL, VT);
// This routine must enter the loop below to work correctly
// when (Reciprocal == false).
diff --git a/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll
new file mode 100644
index 0000000000000..bc367a263caf7
--- /dev/null
+++ b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types-pic.ll
@@ -0,0 +1,160 @@
+; NOTE: Do not autogenerate
+; RUN: split-file %s %t
+; RUN: llc -O3 -verify-machineinstrs -relocation-model=pic %t/estimate.ll -o - | FileCheck %s --check-prefix=PIC --enable-var-scope --implicit-check-not='{{^[[:space:]]+sqrtps[[:space:]]}}' --implicit-check-not='{{^[[:space:]]+divps[[:space:]]}}'
+; RUN: llc -O3 -verify-machineinstrs -relocation-model=pic %t/fallback.ll -o - | FileCheck %s --check-prefix=FALLBACK
+
+; This test owns emitted PIC form, not post-RA value continuity. It binds each
+; pool definition to the canonical i386 call/pop/GOT expression and @GOTOFF
+; spelling, with immediate arithmetic form where adjacent. The non-PIC test
+; owns semantic dataflow. Whole-input exclusions reject native fallback. The
+; fallback input establishes the excluded operations in the same configuration.
+
+;--- estimate.ll
+
+target triple = "i686-unknown-linux-gnu"
+
+declare <4 x float> @llvm.sqrt.v4f32(<4 x float>)
+declare <8 x float> @llvm.sqrt.v8f32(<8 x float>)
+
+; PIC: [[$V4_THREE:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC: [[$V4_HALF:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-LABEL: rsqrt_v4_default:
+; PIC: calll [[PB:\.L[0-9]+\$pb]]
+; PIC: [[PB]]:
+; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
+; PIC: [[TMP:\.Ltmp[0-9]+]]:
+; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
+; PIC: rsqrtps
+; PIC: addps [[$V4_THREE]]@GOTOFF([[GOT]]), {{%xmm[0-9]+}}
+; PIC: mulps [[$V4_HALF]]@GOTOFF([[GOT]]), {{%xmm[0-9]+}}
+; PIC: retl
+define <4 x float> @rsqrt_v4_default(
+ <4 x float> %n, <4 x float> %x) #0 {
+ %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+ %q = fdiv arcp ninf <4 x float> %n, %sqrt
+ ret <4 x float> %q
+}
+
+; PIC: [[$V8_HALF:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC-NEXT: .long 0xbf000000
+; PIC: [[$V8_THREE:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-NEXT: .long 0xc0400000
+; PIC-LABEL: rsqrt_v8_default:
+; PIC: calll [[PB:\.L[0-9]+\$pb]]
+; PIC: [[PB]]:
+; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
+; PIC: [[TMP:\.Ltmp[0-9]+]]:
+; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
+; PIC: rsqrtps
+; PIC-NEXT: movaps [[$V8_HALF]]@GOTOFF([[GOT]]), {{%xmm[0-9]+}}
+; PIC: mulps
+; PIC: movaps [[$V8_THREE]]@GOTOFF([[GOT]]), [[V8_THREE_REG:%xmm[0-9]+]]
+; PIC-NEXT: addps [[V8_THREE_REG]], {{%xmm[0-9]+}}
+; PIC: rsqrtps
+; PIC: retl
+define <8 x float> @rsqrt_v8_default(
+ <8 x float> %n, <8 x float> %x) #0 {
+ %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
+ %q = fdiv arcp ninf <8 x float> %n, %sqrt
+ ret <8 x float> %q
+}
+
+; PIC: [[$V4_ONE:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-LABEL: div_v4_steps_2:
+; PIC: calll [[PB:\.L[0-9]+\$pb]]
+; PIC: [[PB]]:
+; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
+; PIC: [[TMP:\.Ltmp[0-9]+]]:
+; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
+; PIC: rcpps
+; PIC: movaps [[$V4_ONE]]@GOTOFF([[GOT]]), [[V4_ONE_REG:%xmm[0-9]+]]
+; PIC-NEXT: subps {{%xmm[0-9]+}}, [[V4_ONE_REG]]
+; PIC: retl
+define <4 x float> @div_v4_steps_2(
+ <4 x float> %n, <4 x float> %d) #1 {
+ %q = fdiv arcp ninf <4 x float> %n, %d
+ ret <4 x float> %q
+}
+
+; PIC: [[$V8_ONE:\.LCPI[0-9]+_[0-9]+]]:
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-NEXT: .long 0x3f800000
+; PIC-LABEL: div_v8_steps_2:
+; PIC: calll [[PB:\.L[0-9]+\$pb]]
+; PIC: [[PB]]:
+; PIC-NEXT: popl [[GOT:%e[a-z0-9]+]]
+; PIC: [[TMP:\.Ltmp[0-9]+]]:
+; PIC-NEXT: addl $_GLOBAL_OFFSET_TABLE_+([[TMP]]-[[PB]]), [[GOT]]
+; PIC: rcpps
+; PIC: movaps [[$V8_ONE]]@GOTOFF([[GOT]]), {{%xmm[0-9]+}}
+; PIC: subps
+; PIC: rcpps
+; PIC: subps
+; PIC: retl
+define <8 x float> @div_v8_steps_2(
+ <8 x float> %n, <8 x float> %d) #1 {
+ %q = fdiv arcp ninf <8 x float> %n, %d
+ ret <8 x float> %q
+}
+
+attributes #0 = {
+ "reciprocal-estimates"="vec-sqrtf"
+ "target-features"="+sse,-sse2,-x87"
+}
+attributes #1 = {
+ "reciprocal-estimates"="vec-divf:2"
+ "target-features"="+sse,-sse2,-x87"
+}
+
+;--- fallback.ll
+
+target triple = "i686-unknown-linux-gnu"
+
+declare <4 x float> @llvm.sqrt.v4f32(<4 x float>)
+
+; FALLBACK-LABEL: fallback_sqrt_v4:
+; FALLBACK: {{^[[:space:]]+sqrtps[[:space:]]}}
+; FALLBACK: {{^[[:space:]]+rcpps[[:space:]]}}
+define <4 x float> @fallback_sqrt_v4(
+ <4 x float> %n, <4 x float> %x) #0 {
+ %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+ %q = fdiv arcp ninf <4 x float> %n, %sqrt
+ ret <4 x float> %q
+}
+
+; FALLBACK-LABEL: fallback_div_v4:
+; FALLBACK: {{^[[:space:]]+divps[[:space:]]}}
+define <4 x float> @fallback_div_v4(
+ <4 x float> %n, <4 x float> %d) #1 {
+ %q = fdiv arcp ninf <4 x float> %n, %d
+ ret <4 x float> %q
+}
+
+attributes #0 = {
+ "reciprocal-estimates"="!vec-sqrtf"
+ "target-features"="+sse,-sse2,-x87"
+}
+attributes #1 = {
+ "reciprocal-estimates"="!vec-divf"
+ "target-features"="+sse,-sse2,-x87"
+}
diff --git a/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll
new file mode 100644
index 0000000000000..68e90c18e4aec
--- /dev/null
+++ b/llvm/test/CodeGen/X86/estimate-refinement-illegal-scalar-types.ll
@@ -0,0 +1,422 @@
+; NOTE: Do not autogenerate
+; RUN: split-file %s %t
+; RUN: llc -O3 -verify-machineinstrs -stop-after=finalize-isel %t/estimate.ll -o - | FileCheck %s --check-prefix=MIR --enable-var-scope
+; RUN: llc -O3 -verify-machineinstrs %t/estimate.ll -o - | FileCheck %s --check-prefix=ASM --enable-var-scope --implicit-check-not='{{^[[:space:]]+sqrtps[[:space:]]}}' --implicit-check-not='{{^[[:space:]]+divps[[:space:]]}}'
+; RUN: llc -O3 -verify-machineinstrs %t/fallback.ll -o - | FileCheck %s --check-prefix=FALLBACK
+
+; On i686 with SSE1 but no SSE2 or x87, v4f32 is legal while scalar f32
+; requires softening. The v4f32 cases create refinement constants before type
+; legalization. The v8f32 cases split first and create them during the
+; AfterLegalizeTypes combine. MIR checks bind constants and complete refinement
+; dataflow; assembly checks preserve working boundaries and exclude fallback.
+; The fallback input positively establishes the excluded native operations.
+
+;--- estimate.ll
+
+target triple = "i686-unknown-linux-gnu"
+
+declare <4 x float> @llvm.sqrt.v4f32(<4 x float>)
+declare <8 x float> @llvm.sqrt.v8f32(<8 x float>)
+
+; ASM-LABEL: rsqrt_v4_steps_0:
+; ASM: # %bb.0:
+; ASM-NEXT: rsqrtps %xmm1, %xmm1
+; ASM-NEXT: mulps %xmm1, %xmm0
+; ASM-NEXT: retl
+define <4 x float> @rsqrt_v4_steps_0(
+ <4 x float> %n, <4 x float> %x) #0 {
+ %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+ %q = fdiv arcp ninf <4 x float> %n, %sqrt
+ ret <4 x float> %q
+}
+
+; MIR-LABEL: name: rsqrt_v4_default
+; MIR: constants:
+; MIR-NEXT: - id: [[THREE:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float -3.000000e+00)'
+; MIR: - id: [[HALF:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float -5.000000e-01)'
+; MIR: body:
+; MIR: %[[ARG:[0-9]+]]:vr128 = COPY $xmm1
+; MIR: %[[NUM:[0-9]+]]:vr128 = COPY $xmm0
+; MIR: %[[EST:[0-9]+]]:vr128 = {{.*}}RSQRTPSr %[[ARG]]
+; MIR: %[[AE:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG]], %[[EST]],
+; MIR: %[[AEE:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE]], %[[EST]],
+; MIR: %[[RHS:[0-9]+]]:vr128 = {{.*}}ADDPSrm %[[AEE]], $noreg, 1, $noreg, %const.[[THREE]], $noreg,
+; MIR: %[[LHS:[0-9]+]]:vr128 = {{.*}}MULPSrm %[[EST]], $noreg, 1, $noreg, %const.[[HALF]], $noreg,
+; MIR: %[[REFINED:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS]], {{(killed )?}}%[[RHS]],
+; MIR: %[[RESULT:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[NUM]], {{(killed )?}}%[[REFINED]],
+; MIR-NEXT: $xmm0 = COPY %[[RESULT]]
+; MIR-NEXT: RET 0, $xmm0
+; ASM-LABEL: rsqrt_v4_default:
+; ASM: rsqrtps
+; ASM-NOT: {{^[[:space:]]+sqrtps}}
+; ASM-NOT: divps
+; ASM: retl
+define <4 x float> @rsqrt_v4_default(
+ <4 x float> %n, <4 x float> %x) #1 {
+ %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+ %q = fdiv arcp ninf <4 x float> %n, %sqrt
+ ret <4 x float> %q
+}
+
+; MIR-LABEL: name: rsqrt_v4_steps_2
+; MIR: constants:
+; MIR-NEXT: - id: [[HALF:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float -5.000000e-01)'
+; MIR: - id: [[THREE:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float -3.000000e+00)'
+; MIR: body:
+; MIR: %[[ARG:[0-9]+]]:vr128 = COPY $xmm1
+; MIR: %[[NUM:[0-9]+]]:vr128 = COPY $xmm0
+; MIR: %[[EST:[0-9]+]]:vr128 = {{.*}}RSQRTPSr %[[ARG]]
+; MIR: %[[HALF_LOAD:[0-9]+]]:vr128 = MOVAPSrm $noreg, 1, $noreg, %const.[[HALF]], $noreg
+; MIR: %[[LHS1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[EST]], %[[HALF_LOAD]],
+; MIR: %[[AE1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG]], %[[EST]],
+; MIR: %[[AEE1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE1]], %[[EST]],
+; MIR: %[[THREE_LOAD:[0-9]+]]:vr128 = MOVAPSrm $noreg, 1, $noreg, %const.[[THREE]], $noreg
+; MIR: %[[RHS1:[0-9]+]]:vr128 = {{.*}}ADDPSrr %[[AEE1]], %[[THREE_LOAD]],
+; MIR: %[[REF1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS1]], {{(killed )?}}%[[RHS1]],
+; MIR: %[[AE2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG]], %[[REF1]],
+; MIR: %[[AEE2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE2]], %[[REF1]],
+; MIR: %[[RHS2:[0-9]+]]:vr128 = {{.*}}ADDPSrr %[[AEE2]], %[[THREE_LOAD]],
+; MIR: %[[LHS2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[REF1]], %[[HALF_LOAD]],
+; MIR: %[[REF2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS2]], {{(killed )?}}%[[RHS2]],
+; MIR: %[[RESULT:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[NUM]], {{(killed )?}}%[[REF2]],
+; MIR-NEXT: $xmm0 = COPY %[[RESULT]]
+; MIR-NEXT: RET 0, $xmm0
+; ASM-LABEL: rsqrt_v4_steps_2:
+; ASM: rsqrtps
+; ASM-NOT: {{^[[:space:]]+sqrtps}}
+; ASM-NOT: divps
+; ASM: retl
+define <4 x float> @rsqrt_v4_steps_2(
+ <4 x float> %n, <4 x float> %x) #2 {
+ %sqrt = call afn ninf <4 x float> @llvm.sqrt.v4f32(<4 x float> %x)
+ %q = fdiv arcp ninf <4 x float> %n, %sqrt
+ ret <4 x float> %q
+}
+
+; ASM-LABEL: rsqrt_v8_steps_0:
+; ASM: rsqrtps %xmm2, %xmm2
+; ASM-NEXT: rsqrtps 16(%esp), %xmm3
+; ASM-NEXT: mulps %xmm3, %xmm1
+; ASM-NEXT: mulps %xmm2, %xmm0
+; ASM: retl
+define <8 x float> @rsqrt_v8_steps_0(
+ <8 x float> %n, <8 x float> %x) #3 {
+ %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
+ %q = fdiv arcp ninf <8 x float> %n, %sqrt
+ ret <8 x float> %q
+}
+
+; MIR-LABEL: name: rsqrt_v8_default
+; MIR: constants:
+; MIR-NEXT: - id: [[HALF:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float -5.000000e-01)'
+; MIR: - id: [[THREE:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float -3.000000e+00)'
+; MIR: body:
+; MIR: %[[ARG_LO:[0-9]+]]:vr128 = COPY $xmm2
+; MIR: %[[NUM_HI:[0-9]+]]:vr128 = COPY $xmm1
+; MIR: %[[NUM_LO:[0-9]+]]:vr128 = COPY $xmm0
+; MIR: %[[ARG_HI:[0-9]+]]:vr128 = MOVAPSrm %fixed-stack.{{[0-9]+}}, 1, $noreg, 0, $noreg
+; MIR: %[[EST_HI:[0-9]+]]:vr128 = {{.*}}RSQRTPSr %[[ARG_HI]]
+; MIR: %[[HALF_LOAD:[0-9]+]]:vr128 = MOVAPSrm $noreg, 1, $noreg, %const.[[HALF]], $noreg
+; MIR: %[[LHS_HI:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[EST_HI]], %[[HALF_LOAD]],
+; MIR: %[[AE_HI:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG_HI]], %[[EST_HI]],
+; MIR: %[[AEE_HI:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE_HI]], %[[EST_HI]],
+; MIR: %[[THREE_LOAD:[0-9]+]]:vr128 = MOVAPSrm $noreg, 1, $noreg, %const.[[THREE]], $noreg
+; MIR: %[[RHS_HI:[0-9]+]]:vr128 = {{.*}}ADDPSrr %[[AEE_HI]], %[[THREE_LOAD]],
+; MIR: %[[REF_HI:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS_HI]], {{(killed )?}}%[[RHS_HI]],
+; MIR: %[[EST_LO:[0-9]+]]:vr128 = {{.*}}RSQRTPSr %[[ARG_LO]]
+; MIR: %[[LHS_LO:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[EST_LO]], %[[HALF_LOAD]],
+; MIR: %[[AE_LO:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG_LO]], %[[EST_LO]],
+; MIR: %[[AEE_LO:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE_LO]], %[[EST_LO]],
+; MIR: %[[RHS_LO:[0-9]+]]:vr128 = {{.*}}ADDPSrr %[[AEE_LO]], %[[THREE_LOAD]],
+; MIR: %[[REF_LO:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS_LO]], {{(killed )?}}%[[RHS_LO]],
+; MIR: %[[RESULT_LO:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[NUM_LO]], {{(killed )?}}%[[REF_LO]],
+; MIR-NEXT: %[[RESULT_HI:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[NUM_HI]], {{(killed )?}}%[[REF_HI]],
+; MIR-NEXT: $xmm0 = COPY %[[RESULT_LO]]
+; MIR-NEXT: $xmm1 = COPY %[[RESULT_HI]]
+; MIR-NEXT: RET 0, $xmm0, $xmm1
+; ASM-LABEL: rsqrt_v8_default:
+; ASM-COUNT-2: rsqrtps
+; ASM-NOT: {{^[[:space:]]+sqrtps}}
+; ASM-NOT: divps
+; ASM: retl
+define <8 x float> @rsqrt_v8_default(
+ <8 x float> %n, <8 x float> %x) #4 {
+ %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
+ %q = fdiv arcp ninf <8 x float> %n, %sqrt
+ ret <8 x float> %q
+}
+
+; MIR-LABEL: name: rsqrt_v8_steps_2
+; MIR: constants:
+; MIR-NEXT: - id: [[HALF:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float -5.000000e-01)'
+; MIR: - id: [[THREE:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float -3.000000e+00)'
+; MIR: body:
+; MIR: %[[ARG_LO:[0-9]+]]:vr128 = COPY $xmm2
+; MIR: %[[NUM_HI:[0-9]+]]:vr128 = COPY $xmm1
+; MIR: %[[NUM_LO:[0-9]+]]:vr128 = COPY $xmm0
+; MIR: %[[ARG_HI:[0-9]+]]:vr128 = MOVAPSrm %fixed-stack.{{[0-9]+}}, 1, $noreg, 0, $noreg
+; MIR: %[[EST_HI:[0-9]+]]:vr128 = {{.*}}RSQRTPSr %[[ARG_HI]]
+; MIR: %[[HALF_LOAD:[0-9]+]]:vr128 = MOVAPSrm $noreg, 1, $noreg, %const.[[HALF]], $noreg
+; MIR: %[[LHS_HI1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[EST_HI]], %[[HALF_LOAD]],
+; MIR: %[[AE_HI1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG_HI]], %[[EST_HI]],
+; MIR: %[[AEE_HI1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE_HI1]], %[[EST_HI]],
+; MIR: %[[THREE_LOAD:[0-9]+]]:vr128 = MOVAPSrm $noreg, 1, $noreg, %const.[[THREE]], $noreg
+; MIR: %[[RHS_HI1:[0-9]+]]:vr128 = {{.*}}ADDPSrr %[[AEE_HI1]], %[[THREE_LOAD]],
+; MIR: %[[REF_HI1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS_HI1]], {{(killed )?}}%[[RHS_HI1]],
+; MIR: %[[AE_HI2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG_HI]], %[[REF_HI1]],
+; MIR: %[[AEE_HI2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE_HI2]], %[[REF_HI1]],
+; MIR: %[[RHS_HI2:[0-9]+]]:vr128 = {{.*}}ADDPSrr %[[AEE_HI2]], %[[THREE_LOAD]],
+; MIR: %[[LHS_HI2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[REF_HI1]], %[[HALF_LOAD]],
+; MIR: %[[REF_HI2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS_HI2]], {{(killed )?}}%[[RHS_HI2]],
+; MIR: %[[EST_LO:[0-9]+]]:vr128 = {{.*}}RSQRTPSr %[[ARG_LO]]
+; MIR: %[[LHS_LO1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[EST_LO]], %[[HALF_LOAD]],
+; MIR: %[[AE_LO1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG_LO]], %[[EST_LO]],
+; MIR: %[[AEE_LO1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE_LO1]], %[[EST_LO]],
+; MIR: %[[RHS_LO1:[0-9]+]]:vr128 = {{.*}}ADDPSrr %[[AEE_LO1]], %[[THREE_LOAD]],
+; MIR: %[[REF_LO1:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS_LO1]], {{(killed )?}}%[[RHS_LO1]],
+; MIR: %[[AE_LO2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[ARG_LO]], %[[REF_LO1]],
+; MIR: %[[AEE_LO2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[AE_LO2]], %[[REF_LO1]],
+; MIR: %[[RHS_LO2:[0-9]+]]:vr128 = {{.*}}ADDPSrr %[[AEE_LO2]], %[[THREE_LOAD]],
+; MIR: %[[LHS_LO2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[REF_LO1]], %[[HALF_LOAD]],
+; MIR: %[[REF_LO2:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[LHS_LO2]], {{(killed )?}}%[[RHS_LO2]],
+; MIR: %[[RESULT_LO:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[NUM_LO]], {{(killed )?}}%[[REF_LO2]],
+; MIR-NEXT: %[[RESULT_HI:[0-9]+]]:vr128 = {{.*}}MULPSrr %[[NUM_HI]], {{(killed )?}}%[[REF_HI2]],
+; MIR-NEXT: $xmm0 = COPY %[[RESULT_LO]]
+; MIR-NEXT: $xmm1 = COPY %[[RESULT_HI]]
+; MIR-NEXT: RET 0, $xmm0, $xmm1
+; ASM-LABEL: rsqrt_v8_steps_2:
+; ASM-COUNT-2: rsqrtps
+; ASM-NOT: {{^[[:space:]]+sqrtps}}
+; ASM-NOT: divps
+; ASM: retl
+define <8 x float> @rsqrt_v8_steps_2(
+ <8 x float> %n, <8 x float> %x) #5 {
+ %sqrt = call afn ninf <8 x float> @llvm.sqrt.v8f32(<8 x float> %x)
+ %q = fdiv arcp ninf <8 x float> %n, %sqrt
+ ret <8 x float> %q
+}
+
+; ASM-LABEL: div_v4_steps_0:
+; ASM: # %bb.0:
+; ASM-NEXT: rcpps %xmm1, %xmm1
+; ASM-NEXT: mulps %xmm1, %xmm0
+; ASM-NEXT: retl
+define <4 x float> @div_v4_steps_0(
+ <4 x float> %n, <4 x float> %d) #6 {
+ %q = fdiv arcp ninf <4 x float> %n, %d
+ ret <4 x float> %q
+}
+
+; ASM-LABEL: div_v4_steps_1:
+; ASM: # %bb.0:
+; ASM-NEXT: rcpps %xmm1, %xmm2
+; ASM-NEXT: movaps %xmm0, %xmm3
+; ASM-NEXT: mulps %xmm2, %xmm3
+; ASM-NEXT: mulps %xmm3, %xmm1
+; ASM-NEXT: subps %xmm1, %xmm0
+; ASM-NEXT: mulps %xmm2, %xmm0
+; ASM-NEXT: addps %xmm3, %xmm0
+; ASM-NEXT: retl
+define <4 x float> @div_v4_steps_1(
+ <4 x float> %n, <4 x float> %d) #7 {
+ %q = fdiv arcp ninf <4 x float> %n, %d
+ ret <4 x float> %q
+}
+
+; MIR-LABEL: name: div_v4_steps_2
+; MIR: constants:
+; MIR-NEXT: - id: [[ONE:[0-9]+]]
+; MIR-NEXT: value: '<4 x float> splat (float 1.000000e+00)'
+; MIR: body:
+; MIR: %[[DIVISOR:[0-9]+]]:vr128 = COPY $xmm1
+; M...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/219394
More information about the llvm-commits
mailing list