[llvm] [LoopIdiom] Fix 64-to-32-bit stride truncation in strlen idiom recognition (PR #227406)

via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 10:56:15 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Szymon Sobieszek (Kavu849)

<details>
<summary>Changes</summary>

This patch fixes a silent miscompile where loops with massive strides (e.g., 0x100000001 or 0x2000000000000001) were incorrectly optimized into strlen calls.

Previously, getZExtValue() was truncated to a 32-bit unsigned integer, causing some massive strides to appear as a stride of 1. Furthermore, the OpWidth check relied on multiplication (StepSize * 8), which is vulnerable to 64-bit integer overflow.

By upgrading StepSize to uint64_t and using division (OpWidth / 8), we ensure the optimization is safely aborted for massive strides without risking arithmetic overflow.

---
Full diff: https://github.com/llvm/llvm-project/pull/227406.diff


2 Files Affected:

- (modified) llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp (+2-2) 
- (added) llvm/test/Transforms/LoopIdiom/strlen-large-stride.ll (+41) 


``````````diff
diff --git a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
index 8c098e0d68d72..3e0e648e56770 100644
--- a/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopIdiomRecognize.cpp
@@ -2095,12 +2095,12 @@ class StrlenVerifier {
 
     LLVM_DEBUG(dbgs() << "pointer load scev: " << *LoadEv << "\n");
 
-    unsigned StepSize = Step->getZExtValue();
+    uint64_t StepSize = Step->getZExtValue();
 
     // Verify that StepSize is consistent with platform char width.
     OpWidth = OperandType->getIntegerBitWidth();
     unsigned WcharSize = TLI->getWCharSize(*LoopLoad->getModule());
-    if (OpWidth != StepSize * 8)
+    if (OpWidth % 8 != 0 || StepSize != OpWidth / 8)
       return false;
     if (OpWidth != 8 && OpWidth != 16 && OpWidth != 32)
       return false;
diff --git a/llvm/test/Transforms/LoopIdiom/strlen-large-stride.ll b/llvm/test/Transforms/LoopIdiom/strlen-large-stride.ll
new file mode 100644
index 0000000000000..ad56b1a5c9b3b
--- /dev/null
+++ b/llvm/test/Transforms/LoopIdiom/strlen-large-stride.ll
@@ -0,0 +1,41 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes='loop(loop-idiom)' < %s -S | FileCheck %s
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; Test a stride of 0x100000001 (4294967297).
+; The loop should not be transformed into a strlen call.
+define i64 @large_stride(ptr %src) {
+; CHECK-LABEL: @large_stride(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[WHILE_COND:%.*]]
+; CHECK:       while.cond:
+; CHECK-NEXT:    [[P_0:%.*]] = phi ptr [ [[SRC:%.*]], [[ENTRY:%.*]] ], [ [[INCDEC_PTR:%.*]], [[WHILE_COND]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = load i8, ptr [[P_0]], align 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i8 [[TMP0]], 0
+; CHECK-NEXT:    [[INCDEC_PTR]] = getelementptr inbounds i8, ptr [[P_0]], i64 4294967297
+; CHECK-NEXT:    br i1 [[CMP]], label [[WHILE_COND]], label [[WHILE_END:%.*]]
+; CHECK:       while.end:
+; CHECK-NEXT:    [[P_0_LCSSA:%.*]] = phi ptr [ [[P_0]], [[WHILE_COND]] ]
+; CHECK-NEXT:    [[SUB_PTR_LHS_CAST:%.*]] = ptrtoint ptr [[P_0_LCSSA]] to i64
+; CHECK-NEXT:    [[SUB_PTR_RHS_CAST:%.*]] = ptrtoint ptr [[SRC]] to i64
+; CHECK-NEXT:    [[SUB_PTR_SUB:%.*]] = sub i64 [[SUB_PTR_LHS_CAST]], [[SUB_PTR_RHS_CAST]]
+; CHECK-NEXT:    ret i64 [[SUB_PTR_SUB]]
+;
+entry:
+  br label %while.cond
+
+while.cond:
+  %p.0 = phi ptr [ %src, %entry ], [ %incdec.ptr, %while.cond ]
+  %0 = load i8, ptr %p.0, align 1
+  %cmp = icmp ne i8 %0, 0
+  %incdec.ptr = getelementptr inbounds i8, ptr %p.0, i64 4294967297
+  br i1 %cmp, label %while.cond, label %while.end
+
+while.end:
+  %sub.ptr.lhs.cast = ptrtoint ptr %p.0 to i64
+  %sub.ptr.rhs.cast = ptrtoint ptr %src to i64
+  %sub.ptr.sub = sub i64 %sub.ptr.lhs.cast, %sub.ptr.rhs.cast
+  ret i64 %sub.ptr.sub
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/227406


More information about the llvm-commits mailing list