[flang-commits] [flang] [llvm] [Flang][Runtime] Add fast path for formatted real input (PR #229335)
via flang-commits
flang-commits at lists.llvm.org
Tue Oct 6 02:00:48 PDT 2026
https://github.com/ShashwathiNavada updated https://github.com/llvm/llvm-project/pull/229335
>From 11360732fce056f2a8a5d9652826b999c42ceff3 Mon Sep 17 00:00:00 2001
From: Shashwathi N <shashwathinavada at gmail.com>
Date: Tue, 6 Oct 2026 03:06:15 -0500
Subject: [PATCH] [Flang][Runtime] Add fast path for formatted real input
---
flang-rt/lib/runtime/edit-input.cpp | 59 +++++++----
flang-rt/lib/runtime/io-api.cpp | 54 ++++++----
flang/lib/Decimal/decimal-to-binary.cpp | 133 ++++++++++++++++++++++++
3 files changed, 209 insertions(+), 37 deletions(-)
diff --git a/flang-rt/lib/runtime/edit-input.cpp b/flang-rt/lib/runtime/edit-input.cpp
index 6a05974893e517..8b717e957494dc 100644
--- a/flang-rt/lib/runtime/edit-input.cpp
+++ b/flang-rt/lib/runtime/edit-input.cpp
@@ -559,43 +559,64 @@ static RT_API_ATTRS ScannedRealInput ScanRealInput(
static RT_API_ATTRS void RaiseFPExceptions(
decimal::ConversionResultFlags flags) {
-#undef RAISE
#if defined(RT_DEVICE_COMPILATION)
Terminator terminator(__FILE__, __LINE__);
#define RAISE(e) \
terminator.Crash( \
"not implemented yet: raising FP exception in device code: %s", #e);
-#else // !defined(RT_DEVICE_COMPILATION)
-#ifdef feraiseexcept // a macro in some environments; omit std::
-#define RAISE feraiseexcept
-#else
-#define RAISE std::feraiseexcept
-#endif
-#endif // !defined(RT_DEVICE_COMPILATION)
-
-// Some environment (e.g. emscripten, musl) don't define FE_OVERFLOW as allowed
-// by c99 (but not c++11) :-/
-#if defined(FE_OVERFLOW) || defined(RT_DEVICE_COMPILATION)
+ // Some environment (e.g. emscripten, musl) don't define FE_OVERFLOW as
+ // allowed by c99 (but not c++11) :-/
if (flags & decimal::ConversionResultFlags::Overflow) {
RAISE(FE_OVERFLOW);
}
-#endif
-#if defined(FE_UNDERFLOW) || defined(RT_DEVICE_COMPILATION)
if (flags & decimal::ConversionResultFlags::Underflow) {
RAISE(FE_UNDERFLOW);
}
-#endif
-#if defined(FE_INEXACT) || defined(RT_DEVICE_COMPILATION)
if (flags & decimal::ConversionResultFlags::Inexact) {
RAISE(FE_INEXACT);
}
-#endif
-#if defined(FE_INVALID) || defined(RT_DEVICE_COMPILATION)
if (flags & decimal::ConversionResultFlags::Invalid) {
RAISE(FE_INVALID);
}
-#endif
#undef RAISE
+#else // !defined(RT_DEVICE_COMPILATION)
+ // Collect the requested exceptions, then raise only those not already set.
+ // feraiseexcept on an already-set flag is a no-op yet still costs a libm
+ // call, which dominates formatted real input of many inexact values.
+ int excepts{0};
+#if defined(FE_OVERFLOW)
+ if (flags & decimal::ConversionResultFlags::Overflow) {
+ excepts |= FE_OVERFLOW;
+ }
+#endif
+#if defined(FE_UNDERFLOW)
+ if (flags & decimal::ConversionResultFlags::Underflow) {
+ excepts |= FE_UNDERFLOW;
+ }
+#endif
+#if defined(FE_INEXACT)
+ if (flags & decimal::ConversionResultFlags::Inexact) {
+ excepts |= FE_INEXACT;
+ }
+#endif
+#if defined(FE_INVALID)
+ if (flags & decimal::ConversionResultFlags::Invalid) {
+ excepts |= FE_INVALID;
+ }
+#endif
+#ifdef fetestexcept // a macro in some environments; omit std::
+ excepts &= ~fetestexcept(excepts);
+#else
+ excepts &= ~std::fetestexcept(excepts);
+#endif
+ if (excepts) {
+#ifdef feraiseexcept // a macro in some environments; omit std::
+ feraiseexcept(excepts);
+#else
+ std::feraiseexcept(excepts);
+#endif
+ }
+#endif // !defined(RT_DEVICE_COMPILATION)
}
// If no special modes are in effect and the form of the input value
diff --git a/flang-rt/lib/runtime/io-api.cpp b/flang-rt/lib/runtime/io-api.cpp
index 28f16872ddada5..f0996b2c9b0fc6 100644
--- a/flang-rt/lib/runtime/io-api.cpp
+++ b/flang-rt/lib/runtime/io-api.cpp
@@ -1128,23 +1128,25 @@ bool IODEF(InputInteger)(Cookie cookie, std::int64_t &n, int kind) {
}
bool IODEF(InputReal32)(Cookie cookie, float &x) {
- if (!cookie->CheckFormattedStmtType<Direction::Input>("InputReal32")) {
- return false;
+ IoStatementState &io{*cookie};
+ if (io.BeginReadingRecord()) {
+ if (auto edit{io.GetNextDataEdit()}) {
+ return edit->descriptor == DataEdit::ListDirectedNullValue ||
+ EditRealInput<4>(io, *edit, reinterpret_cast<void *>(&x));
+ }
}
- StaticDescriptor<0> staticDescriptor;
- Descriptor &descriptor{staticDescriptor.descriptor()};
- descriptor.Establish(TypeCategory::Real, 4, reinterpret_cast<void *>(&x), 0);
- return descr::DescriptorIO<Direction::Input>(*cookie, descriptor);
+ return false;
}
bool IODEF(InputReal64)(Cookie cookie, double &x) {
- if (!cookie->CheckFormattedStmtType<Direction::Input>("InputReal64")) {
- return false;
+ IoStatementState &io{*cookie};
+ if (io.BeginReadingRecord()) {
+ if (auto edit{io.GetNextDataEdit()}) {
+ return edit->descriptor == DataEdit::ListDirectedNullValue ||
+ EditRealInput<8>(io, *edit, reinterpret_cast<void *>(&x));
+ }
}
- StaticDescriptor<0> staticDescriptor;
- Descriptor &descriptor{staticDescriptor.descriptor()};
- descriptor.Establish(TypeCategory::Real, 8, reinterpret_cast<void *>(&x), 0);
- return descr::DescriptorIO<Direction::Input>(*cookie, descriptor);
+ return false;
}
bool IODEF(InputComplex32)(Cookie cookie, float z[2]) {
@@ -1183,13 +1185,29 @@ bool IODEF(OutputCharacter)(
bool IODEF(InputCharacter)(
Cookie cookie, char *x, std::size_t length, int kind) {
- if (!cookie->CheckFormattedStmtType<Direction::Input>("InputCharacter")) {
- return false;
+ IoStatementState &io{*cookie};
+ if (io.BeginReadingRecord()) {
+ if (auto edit{io.GetNextDataEdit()}) {
+ if (edit->descriptor == DataEdit::ListDirectedNullValue) {
+ return true;
+ }
+ switch (kind) {
+ case 1:
+ return EditCharacterInput(io, *edit, x, length);
+ case 2:
+ return EditCharacterInput(
+ io, *edit, reinterpret_cast<char16_t *>(x), length);
+ case 4:
+ return EditCharacterInput(
+ io, *edit, reinterpret_cast<char32_t *>(x), length);
+ default:
+ io.GetIoErrorHandler().Crash(
+ "InputCharacter: bad character kind %d", kind);
+ return false;
+ }
+ }
}
- StaticDescriptor<0> staticDescriptor;
- Descriptor &descriptor{staticDescriptor.descriptor()};
- descriptor.Establish(kind, length, reinterpret_cast<void *>(x), 0);
- return descr::DescriptorIO<Direction::Input>(*cookie, descriptor);
+ return false;
}
bool IODEF(InputAscii)(Cookie cookie, char *x, std::size_t length) {
diff --git a/flang/lib/Decimal/decimal-to-binary.cpp b/flang/lib/Decimal/decimal-to-binary.cpp
index 42998b171f7885..ab45e4b4566460 100644
--- a/flang/lib/Decimal/decimal-to-binary.cpp
+++ b/flang/lib/Decimal/decimal-to-binary.cpp
@@ -448,10 +448,143 @@ BigRadixFloatingPointNumber<PREC, LOG10RADIX>::ConvertToBinary() {
return f.ToBinary(isNegative_, rounding_);
}
+#if defined(__SIZEOF_INT128__)
+// Clinger fast path for IEEE double, round-to-nearest: when the significand is
+// exactly representable (<= 2**53) and 10**|exponent| is exactly representable
+// (|exponent| <= 22), a single correctly-rounded IEEE multiply or divide yields
+// the correctly-rounded result. Inexactness is determined algebraically so the
+// flag matches the general path. On success advances p exactly as ParseNumber
+// would and returns true; on failure leaves p unchanged.
+template <int PREC>
+static RT_API_ATTRS bool TryClingerFastPath(const char *&p, const char *end,
+ enum FortranRounding rounding, ConversionToBinaryResult<PREC> &result) {
+ if constexpr (PREC != 53) {
+ return false;
+ } else {
+ if (rounding != RoundNearest) {
+ return false;
+ }
+ static constexpr double pow10[]{1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8,
+ 1e9, 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20,
+ 1e21, 1e22};
+ static constexpr std::uint64_t pow5[]{1u, 5u, 25u, 125u, 625u, 3125u,
+ 15625u, 78125u, 390625u, 1953125u, 9765625u, 48828125u, 244140625u,
+ 1220703125u, 6103515625u, 30517578125u, 152587890625u, 762939453125u,
+ 3814697265625u, 19073486328125u, 95367431640625u, 476837158203125u,
+ 2384185791015625u};
+ constexpr std::uint64_t safeLimit{std::uint64_t{1} << 53};
+ const char *q{p};
+ if (end && q >= end) {
+ return false;
+ }
+ for (; q != end && *q == ' '; ++q) {
+ }
+ if (q == end) {
+ return false;
+ }
+ bool isNegative{*q == '-'};
+ if (*q == '-' || *q == '+') {
+ ++q;
+ }
+ std::uint64_t mantissa{0};
+ int fracDigits{0};
+ bool anyDigit{false};
+ for (; q != end && *q >= '0' && *q <= '9'; ++q) {
+ if (mantissa > safeLimit) {
+ return false; // too many significant digits for the exact fast path
+ }
+ mantissa = mantissa * 10 + (*q - '0');
+ anyDigit = true;
+ }
+ if (q != end && *q == '.') {
+ ++q;
+ for (; q != end && *q >= '0' && *q <= '9'; ++q) {
+ if (mantissa > safeLimit) {
+ return false;
+ }
+ mantissa = mantissa * 10 + (*q - '0');
+ ++fracDigits;
+ anyDigit = true;
+ }
+ }
+ if (!anyDigit) {
+ return false;
+ }
+ const char *tail{q}; // position after the significand
+ int exp10{-fracDigits};
+ if (q != end) {
+ char c{*q};
+ if (c == 'e' || c == 'E' || c == 'd' || c == 'D' || c == 'q' ||
+ c == 'Q') {
+ const char *e{q + 1};
+ bool negExpo{e != end && *e == '-'};
+ if (e != end && (*e == '-' || *e == '+')) {
+ ++e;
+ }
+ if (e != end && *e >= '0' && *e <= '9') {
+ int expo{0};
+ for (; e != end && *e >= '0' && *e <= '9'; ++e) {
+ if (expo < 100000) {
+ expo = 10 * expo + (*e - '0');
+ }
+ }
+ tail = e; // exponent consumed
+ exp10 += negExpo ? -expo : expo;
+ }
+ }
+ }
+ if (mantissa == 0) {
+ double zero{isNegative ? -0.0 : 0.0};
+ result = {BinaryFloatingPointNumber<PREC>{zero}, Exact};
+ p = tail;
+ return true;
+ }
+ if (mantissa > safeLimit || exp10 < -22 || exp10 > 22) {
+ return false;
+ }
+ double m{static_cast<double>(mantissa)};
+ double value;
+ bool inexact;
+ if (exp10 == 0) {
+ value = m;
+ inexact = false;
+ } else if (exp10 > 0) {
+ value = m * pow10[exp10];
+ // Exact iff mantissa * 5**exp10 has at most 53 significant bits.
+ unsigned __int128 prod{
+ static_cast<unsigned __int128>(mantissa) * pow5[exp10]};
+ auto lo{static_cast<std::uint64_t>(prod)};
+ auto hi{static_cast<std::uint64_t>(prod >> 64)};
+ int highBit{
+ hi != 0 ? 127 - __builtin_clzll(hi) : 63 - __builtin_clzll(lo)};
+ int lowBit{lo != 0 ? __builtin_ctzll(lo) : 64 + __builtin_ctzll(hi)};
+ inexact = highBit - lowBit + 1 > 53;
+ } else { // exp10 < 0
+ value = m / pow10[-exp10];
+ // Exact iff 5**(-exp10) divides the significand.
+ inexact = mantissa % pow5[-exp10] != 0;
+ }
+ if (isNegative) {
+ value = -value;
+ }
+ result = {
+ BinaryFloatingPointNumber<PREC>{value}, inexact ? Inexact : Exact};
+ p = tail;
+ return true;
+ }
+}
+#endif // __SIZEOF_INT128__
+
template <int PREC, int LOG10RADIX>
ConversionToBinaryResult<PREC>
BigRadixFloatingPointNumber<PREC, LOG10RADIX>::ConvertToBinary(
const char *&p, const char *limit) {
+#if defined(__SIZEOF_INT128__)
+ if (ConversionToBinaryResult<PREC> fast;
+ TryClingerFastPath<PREC>(p, limit, rounding_, fast)) {
+ return fast;
+ }
+#endif
bool inexact{false};
if (ParseNumber(p, inexact, limit)) {
auto result{ConvertToBinary()};
More information about the flang-commits
mailing list