[llvm] [X86] Faster truncf and roundf on x86 SSE2 (PR #226513)

Ralf Jung via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 27 03:51:58 PDT 2026


================
@@ -23178,6 +23190,56 @@ SDValue X86TargetLowering::lowerFaddFsub(SDValue Op, SelectionDAG &DAG) const {
   return lowerAddSubToHorizontalOp(Op, SDLoc(Op), DAG, Subtarget);
 }
 
+static SDValue lowerFTRUNC_FROUND_SSE2(SDValue Op, SelectionDAG &DAG) {
+  SDLoc DL(Op);
+  SDValue N0 = Op.getOperand(0);
+  MVT VT = Op.getSimpleValueType();
+  bool IsRound = Op.getOpcode() == ISD::FROUND;
+
+  SDValue Abs = DAG.getNode(ISD::FABS, DL, VT, N0);
+  SDValue AbsBiased = Abs;
+  if (IsRound) {
+    const fltSemantics &Sem = VT.getFltSemantics();
+    APFloat Bias = APFloat(0.5f);
+    bool Ignored;
+    Bias.convert(Sem, APFloat::rmNearestTiesToEven, &Ignored);
+    Bias.next(/*nextDown*/ true);
+    AbsBiased =
+        DAG.getNode(ISD::FADD, DL, VT, Abs, DAG.getConstantFP(Bias, DL, VT));
+  }
+
+  MVT IntVT;
+  if (VT == MVT::f32)
+    IntVT = MVT::i32;
+  else if (VT == MVT::f64)
+    IntVT = MVT::i64;
+  else if (VT == MVT::v4f32)
+    IntVT = MVT::v4i32;
+  else if (VT == MVT::v2f64)
+    IntVT = MVT::v2i64;
+  else
+    llvm_unreachable("Unexpected type");
+
+  const fltSemantics &Sem = VT.getFltSemantics();
+  // Any threshold in [2^23, 2^31] for float (or [2^52, 2^63] for double) is
+  // correct since all FP values at or above 2^23 (2^52) are already integers.
+  APFloat Bound = VT.getScalarType() == MVT::f32 ? APFloat(Sem, "0x1.0p31")
+                                                 : APFloat(Sem, "0x1.0p63");
+  SDValue Threshold = DAG.getConstantFP(Bound, DL, VT);
+
+  EVT CCVT = DAG.getTargetLoweringInfo().getSetCCResultType(
+      DAG.getDataLayout(), *DAG.getContext(), VT);
+  SDValue IsLarge = DAG.getSetCC(DL, CCVT, Abs, Threshold, ISD::SETUGE);
----------------
RalfJung wrote:

The issue mentions that this makes assumptions about how Abs treats NaN. That should be guaranteed (abs is a "bitwise" operation and so its effect is defined in terms of the bit representation, not the abstract float value), but seems worth a comment here to explain how we're relying on it.

https://github.com/llvm/llvm-project/pull/226513


More information about the llvm-commits mailing list