[flang-commits] [flang] [llvm] [Flang][Runtime] Add fast path for formatted real input (PR #229335)

via flang-commits flang-commits at lists.llvm.org
Tue Oct 6 02:00:48 PDT 2026


https://github.com/ShashwathiNavada updated https://github.com/llvm/llvm-project/pull/229335

>From 11360732fce056f2a8a5d9652826b999c42ceff3 Mon Sep 17 00:00:00 2001
From: Shashwathi N <shashwathinavada at gmail.com>
Date: Tue, 6 Oct 2026 03:06:15 -0500
Subject: [PATCH] [Flang][Runtime] Add fast path for formatted real input

---
 flang-rt/lib/runtime/edit-input.cpp     |  59 +++++++----
 flang-rt/lib/runtime/io-api.cpp         |  54 ++++++----
 flang/lib/Decimal/decimal-to-binary.cpp | 133 ++++++++++++++++++++++++
 3 files changed, 209 insertions(+), 37 deletions(-)

diff --git a/flang-rt/lib/runtime/edit-input.cpp b/flang-rt/lib/runtime/edit-input.cpp
index 6a05974893e517..8b717e957494dc 100644
--- a/flang-rt/lib/runtime/edit-input.cpp
+++ b/flang-rt/lib/runtime/edit-input.cpp
@@ -559,43 +559,64 @@ static RT_API_ATTRS ScannedRealInput ScanRealInput(
 
 static RT_API_ATTRS void RaiseFPExceptions(
     decimal::ConversionResultFlags flags) {
-#undef RAISE
 #if defined(RT_DEVICE_COMPILATION)
   Terminator terminator(__FILE__, __LINE__);
 #define RAISE(e) \
   terminator.Crash( \
       "not implemented yet: raising FP exception in device code: %s", #e);
-#else // !defined(RT_DEVICE_COMPILATION)
-#ifdef feraiseexcept // a macro in some environments; omit std::
-#define RAISE feraiseexcept
-#else
-#define RAISE std::feraiseexcept
-#endif
-#endif // !defined(RT_DEVICE_COMPILATION)
-
-// Some environment (e.g. emscripten, musl) don't define FE_OVERFLOW as allowed
-// by c99 (but not c++11) :-/
-#if defined(FE_OVERFLOW) || defined(RT_DEVICE_COMPILATION)
+  // Some environment (e.g. emscripten, musl) don't define FE_OVERFLOW as
+  // allowed by c99 (but not c++11) :-/
   if (flags & decimal::ConversionResultFlags::Overflow) {
     RAISE(FE_OVERFLOW);
   }
-#endif
-#if defined(FE_UNDERFLOW) || defined(RT_DEVICE_COMPILATION)
   if (flags & decimal::ConversionResultFlags::Underflow) {
     RAISE(FE_UNDERFLOW);
   }
-#endif
-#if defined(FE_INEXACT) || defined(RT_DEVICE_COMPILATION)
   if (flags & decimal::ConversionResultFlags::Inexact) {
     RAISE(FE_INEXACT);
   }
-#endif
-#if defined(FE_INVALID) || defined(RT_DEVICE_COMPILATION)
   if (flags & decimal::ConversionResultFlags::Invalid) {
     RAISE(FE_INVALID);
   }
-#endif
 #undef RAISE
+#else // !defined(RT_DEVICE_COMPILATION)
+  // Collect the requested exceptions, then raise only those not already set.
+  // feraiseexcept on an already-set flag is a no-op yet still costs a libm
+  // call, which dominates formatted real input of many inexact values.
+  int excepts{0};
+#if defined(FE_OVERFLOW)
+  if (flags & decimal::ConversionResultFlags::Overflow) {
+    excepts |= FE_OVERFLOW;
+  }
+#endif
+#if defined(FE_UNDERFLOW)
+  if (flags & decimal::ConversionResultFlags::Underflow) {
+    excepts |= FE_UNDERFLOW;
+  }
+#endif
+#if defined(FE_INEXACT)
+  if (flags & decimal::ConversionResultFlags::Inexact) {
+    excepts |= FE_INEXACT;
+  }
+#endif
+#if defined(FE_INVALID)
+  if (flags & decimal::ConversionResultFlags::Invalid) {
+    excepts |= FE_INVALID;
+  }
+#endif
+#ifdef fetestexcept // a macro in some environments; omit std::
+  excepts &= ~fetestexcept(excepts);
+#else
+  excepts &= ~std::fetestexcept(excepts);
+#endif
+  if (excepts) {
+#ifdef feraiseexcept // a macro in some environments; omit std::
+    feraiseexcept(excepts);
+#else
+    std::feraiseexcept(excepts);
+#endif
+  }
+#endif // !defined(RT_DEVICE_COMPILATION)
 }
 
 // If no special modes are in effect and the form of the input value
diff --git a/flang-rt/lib/runtime/io-api.cpp b/flang-rt/lib/runtime/io-api.cpp
index 28f16872ddada5..f0996b2c9b0fc6 100644
--- a/flang-rt/lib/runtime/io-api.cpp
+++ b/flang-rt/lib/runtime/io-api.cpp
@@ -1128,23 +1128,25 @@ bool IODEF(InputInteger)(Cookie cookie, std::int64_t &n, int kind) {
 }
 
 bool IODEF(InputReal32)(Cookie cookie, float &x) {
-  if (!cookie->CheckFormattedStmtType<Direction::Input>("InputReal32")) {
-    return false;
+  IoStatementState &io{*cookie};
+  if (io.BeginReadingRecord()) {
+    if (auto edit{io.GetNextDataEdit()}) {
+      return edit->descriptor == DataEdit::ListDirectedNullValue ||
+          EditRealInput<4>(io, *edit, reinterpret_cast<void *>(&x));
+    }
   }
-  StaticDescriptor<0> staticDescriptor;
-  Descriptor &descriptor{staticDescriptor.descriptor()};
-  descriptor.Establish(TypeCategory::Real, 4, reinterpret_cast<void *>(&x), 0);
-  return descr::DescriptorIO<Direction::Input>(*cookie, descriptor);
+  return false;
 }
 
 bool IODEF(InputReal64)(Cookie cookie, double &x) {
-  if (!cookie->CheckFormattedStmtType<Direction::Input>("InputReal64")) {
-    return false;
+  IoStatementState &io{*cookie};
+  if (io.BeginReadingRecord()) {
+    if (auto edit{io.GetNextDataEdit()}) {
+      return edit->descriptor == DataEdit::ListDirectedNullValue ||
+          EditRealInput<8>(io, *edit, reinterpret_cast<void *>(&x));
+    }
   }
-  StaticDescriptor<0> staticDescriptor;
-  Descriptor &descriptor{staticDescriptor.descriptor()};
-  descriptor.Establish(TypeCategory::Real, 8, reinterpret_cast<void *>(&x), 0);
-  return descr::DescriptorIO<Direction::Input>(*cookie, descriptor);
+  return false;
 }
 
 bool IODEF(InputComplex32)(Cookie cookie, float z[2]) {
@@ -1183,13 +1185,29 @@ bool IODEF(OutputCharacter)(
 
 bool IODEF(InputCharacter)(
     Cookie cookie, char *x, std::size_t length, int kind) {
-  if (!cookie->CheckFormattedStmtType<Direction::Input>("InputCharacter")) {
-    return false;
+  IoStatementState &io{*cookie};
+  if (io.BeginReadingRecord()) {
+    if (auto edit{io.GetNextDataEdit()}) {
+      if (edit->descriptor == DataEdit::ListDirectedNullValue) {
+        return true;
+      }
+      switch (kind) {
+      case 1:
+        return EditCharacterInput(io, *edit, x, length);
+      case 2:
+        return EditCharacterInput(
+            io, *edit, reinterpret_cast<char16_t *>(x), length);
+      case 4:
+        return EditCharacterInput(
+            io, *edit, reinterpret_cast<char32_t *>(x), length);
+      default:
+        io.GetIoErrorHandler().Crash(
+            "InputCharacter: bad character kind %d", kind);
+        return false;
+      }
+    }
   }
-  StaticDescriptor<0> staticDescriptor;
-  Descriptor &descriptor{staticDescriptor.descriptor()};
-  descriptor.Establish(kind, length, reinterpret_cast<void *>(x), 0);
-  return descr::DescriptorIO<Direction::Input>(*cookie, descriptor);
+  return false;
 }
 
 bool IODEF(InputAscii)(Cookie cookie, char *x, std::size_t length) {
diff --git a/flang/lib/Decimal/decimal-to-binary.cpp b/flang/lib/Decimal/decimal-to-binary.cpp
index 42998b171f7885..ab45e4b4566460 100644
--- a/flang/lib/Decimal/decimal-to-binary.cpp
+++ b/flang/lib/Decimal/decimal-to-binary.cpp
@@ -448,10 +448,143 @@ BigRadixFloatingPointNumber<PREC, LOG10RADIX>::ConvertToBinary() {
   return f.ToBinary(isNegative_, rounding_);
 }
 
+#if defined(__SIZEOF_INT128__)
+// Clinger fast path for IEEE double, round-to-nearest: when the significand is
+// exactly representable (<= 2**53) and 10**|exponent| is exactly representable
+// (|exponent| <= 22), a single correctly-rounded IEEE multiply or divide yields
+// the correctly-rounded result.  Inexactness is determined algebraically so the
+// flag matches the general path.  On success advances p exactly as ParseNumber
+// would and returns true; on failure leaves p unchanged.
+template <int PREC>
+static RT_API_ATTRS bool TryClingerFastPath(const char *&p, const char *end,
+    enum FortranRounding rounding, ConversionToBinaryResult<PREC> &result) {
+  if constexpr (PREC != 53) {
+    return false;
+  } else {
+    if (rounding != RoundNearest) {
+      return false;
+    }
+    static constexpr double pow10[]{1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8,
+        1e9, 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20,
+        1e21, 1e22};
+    static constexpr std::uint64_t pow5[]{1u, 5u, 25u, 125u, 625u, 3125u,
+        15625u, 78125u, 390625u, 1953125u, 9765625u, 48828125u, 244140625u,
+        1220703125u, 6103515625u, 30517578125u, 152587890625u, 762939453125u,
+        3814697265625u, 19073486328125u, 95367431640625u, 476837158203125u,
+        2384185791015625u};
+    constexpr std::uint64_t safeLimit{std::uint64_t{1} << 53};
+    const char *q{p};
+    if (end && q >= end) {
+      return false;
+    }
+    for (; q != end && *q == ' '; ++q) {
+    }
+    if (q == end) {
+      return false;
+    }
+    bool isNegative{*q == '-'};
+    if (*q == '-' || *q == '+') {
+      ++q;
+    }
+    std::uint64_t mantissa{0};
+    int fracDigits{0};
+    bool anyDigit{false};
+    for (; q != end && *q >= '0' && *q <= '9'; ++q) {
+      if (mantissa > safeLimit) {
+        return false; // too many significant digits for the exact fast path
+      }
+      mantissa = mantissa * 10 + (*q - '0');
+      anyDigit = true;
+    }
+    if (q != end && *q == '.') {
+      ++q;
+      for (; q != end && *q >= '0' && *q <= '9'; ++q) {
+        if (mantissa > safeLimit) {
+          return false;
+        }
+        mantissa = mantissa * 10 + (*q - '0');
+        ++fracDigits;
+        anyDigit = true;
+      }
+    }
+    if (!anyDigit) {
+      return false;
+    }
+    const char *tail{q}; // position after the significand
+    int exp10{-fracDigits};
+    if (q != end) {
+      char c{*q};
+      if (c == 'e' || c == 'E' || c == 'd' || c == 'D' || c == 'q' ||
+          c == 'Q') {
+        const char *e{q + 1};
+        bool negExpo{e != end && *e == '-'};
+        if (e != end && (*e == '-' || *e == '+')) {
+          ++e;
+        }
+        if (e != end && *e >= '0' && *e <= '9') {
+          int expo{0};
+          for (; e != end && *e >= '0' && *e <= '9'; ++e) {
+            if (expo < 100000) {
+              expo = 10 * expo + (*e - '0');
+            }
+          }
+          tail = e; // exponent consumed
+          exp10 += negExpo ? -expo : expo;
+        }
+      }
+    }
+    if (mantissa == 0) {
+      double zero{isNegative ? -0.0 : 0.0};
+      result = {BinaryFloatingPointNumber<PREC>{zero}, Exact};
+      p = tail;
+      return true;
+    }
+    if (mantissa > safeLimit || exp10 < -22 || exp10 > 22) {
+      return false;
+    }
+    double m{static_cast<double>(mantissa)};
+    double value;
+    bool inexact;
+    if (exp10 == 0) {
+      value = m;
+      inexact = false;
+    } else if (exp10 > 0) {
+      value = m * pow10[exp10];
+      // Exact iff mantissa * 5**exp10 has at most 53 significant bits.
+      unsigned __int128 prod{
+          static_cast<unsigned __int128>(mantissa) * pow5[exp10]};
+      auto lo{static_cast<std::uint64_t>(prod)};
+      auto hi{static_cast<std::uint64_t>(prod >> 64)};
+      int highBit{
+          hi != 0 ? 127 - __builtin_clzll(hi) : 63 - __builtin_clzll(lo)};
+      int lowBit{lo != 0 ? __builtin_ctzll(lo) : 64 + __builtin_ctzll(hi)};
+      inexact = highBit - lowBit + 1 > 53;
+    } else { // exp10 < 0
+      value = m / pow10[-exp10];
+      // Exact iff 5**(-exp10) divides the significand.
+      inexact = mantissa % pow5[-exp10] != 0;
+    }
+    if (isNegative) {
+      value = -value;
+    }
+    result = {
+        BinaryFloatingPointNumber<PREC>{value}, inexact ? Inexact : Exact};
+    p = tail;
+    return true;
+  }
+}
+#endif // __SIZEOF_INT128__
+
 template <int PREC, int LOG10RADIX>
 ConversionToBinaryResult<PREC>
 BigRadixFloatingPointNumber<PREC, LOG10RADIX>::ConvertToBinary(
     const char *&p, const char *limit) {
+#if defined(__SIZEOF_INT128__)
+  if (ConversionToBinaryResult<PREC> fast;
+      TryClingerFastPath<PREC>(p, limit, rounding_, fast)) {
+    return fast;
+  }
+#endif
   bool inexact{false};
   if (ParseNumber(p, inexact, limit)) {
     auto result{ConvertToBinary()};



More information about the flang-commits mailing list