[llvm] [X86] Prevent NUW flag forwarding in ADD->SUB peephole (PR #201612)

via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 13 09:53:59 PDT 2026


https://github.com/hesam-oxe updated https://github.com/llvm/llvm-project/pull/201612

>From f04117ce46c7e4f9ce1ff02bbc72eb7f62785a1a Mon Sep 17 00:00:00 2001
From: hesam-oxe <chngyzkhanwhsht at gmail.com>
Date: Sun, 13 Sep 2026 16:53:48 +0000
Subject: [PATCH] [X86] Prevent NUW flag forwarding in ADD->SUB peephole

The peephole that converts 'add X, 128' (and the 0x80000000 64-bit
variant) into 'sub X, -128' must not fire when the ADD carries NUW:
ISel copies the matched node's poison-generating flags onto the
selected MachineInstr, and the SUB then carries a flag whose wrap
condition was only valid for the ADD form (see #160217).

Guard the patterns with an add_without_nuw PatFrag so NUW adds keep
the ADD form instead of silently inheriting an incorrect SUB flag.
Update avx512vnni-combine.ll accordingly (the loop-carried add there
has NUW from the vnni combining, so it now stays 'addq $128').

Rebased onto current main; patterns removed upstream in the meantime
(the GR64/-128 SUB64ri32 and SUB64ri32_ND cases) are left deleted, and
the 0x80000000 block follows main's HasNDD predicate rename.
---
 llvm/lib/Target/X86/X86InstrCompiler.td     | 35 ++++++++++++---------
 llvm/test/CodeGen/X86/avx512vnni-combine.ll |  2 +-
 2 files changed, 22 insertions(+), 15 deletions(-)

diff --git a/llvm/lib/Target/X86/X86InstrCompiler.td b/llvm/lib/Target/X86/X86InstrCompiler.td
index 45619663c45cb..9909ea764b017 100644
--- a/llvm/lib/Target/X86/X86InstrCompiler.td
+++ b/llvm/lib/Target/X86/X86InstrCompiler.td
@@ -1625,6 +1625,13 @@ def : Pat<(xor GR16:$src1, -32768),
 def : Pat<(xor GR32:$src1, -2147483648),
           (ADD32ri GR32:$src1, -2147483648)>;
 }
+// Match an `add` that does NOT have the NUW (No Unsigned Wrap) flag.
+// This prevents the NUW flag from being incorrectly forwarded to SUB
+// instructions when we convert `add X, 128` to `SUB X, -128`.
+def add_without_nuw : PatFrag<(ops node:$src1, node:$src2),
+                              (add node:$src1, node:$src2), [{
+  return !N->getFlags().hasNoUnsignedWrap();
+}]>;
 
 //===----------------------------------------------------------------------===//
 // Some peepholes
@@ -1633,9 +1640,9 @@ def : Pat<(xor GR32:$src1, -2147483648),
 // Odd encoding trick: -128 fits into an 8-bit immediate field while
 // +128 doesn't, so in this special case use a sub instead of an add.
 let Predicates = [NoNDD] in {
-  def : Pat<(add GR16:$src1, 128),
+  def : Pat<(add_without_nuw GR16:$src1, 128),
             (SUB16ri GR16:$src1, -128)>;
-  def : Pat<(add GR32:$src1, 128),
+  def : Pat<(add_without_nuw GR32:$src1, 128),
             (SUB32ri GR32:$src1, -128)>;
   def : Pat<(add GR64:$src1, 128),
             (SUB64ri32 GR64:$src1, -128)>;
@@ -1648,9 +1655,9 @@ let Predicates = [NoNDD] in {
             (SUB64ri32 GR64:$src1, -128)>;
 }
 let Predicates = [HasNDD] in {
-  def : Pat<(add GR16:$src1, 128),
+  def : Pat<(add_without_nuw GR16:$src1, 128),
             (SUB16ri_ND GR16:$src1, -128)>;
-  def : Pat<(add GR32:$src1, 128),
+  def : Pat<(add_without_nuw GR32:$src1, 128),
             (SUB32ri_ND GR32:$src1, -128)>;
   def : Pat<(add GR64:$src1, 128),
             (SUB64ri32_ND GR64:$src1, -128)>;
@@ -1662,39 +1669,39 @@ let Predicates = [HasNDD] in {
   def : Pat<(X86add_flag_nocf GR64:$src1, 128),
             (SUB64ri32_ND GR64:$src1, -128)>;
 }
-def : Pat<(store (add (loadi16 addr:$dst), 128), addr:$dst),
+def : Pat<(store (add_without_nuw (loadi16 addr:$dst), 128), addr:$dst),
           (SUB16mi addr:$dst, -128)>;
-def : Pat<(store (add (loadi32 addr:$dst), 128), addr:$dst),
+def : Pat<(store (add_without_nuw (loadi32 addr:$dst), 128), addr:$dst),
           (SUB32mi addr:$dst, -128)>;
-def : Pat<(store (add (loadi64 addr:$dst), 128), addr:$dst),
+def : Pat<(store (add_without_nuw (loadi64 addr:$dst), 128), addr:$dst),
           (SUB64mi32 addr:$dst, -128)>;
 let Predicates = [HasNDD] in {
-  def : Pat<(add (loadi16 ndd_addr:$src), 128),
+  def : Pat<(add_without_nuw (loadi16 ndd_addr:$src), 128),
             (SUB16mi_ND ndd_addr:$src, -128)>;
-  def : Pat<(add (loadi32 ndd_addr:$src), 128),
+  def : Pat<(add_without_nuw (loadi32 ndd_addr:$src), 128),
             (SUB32mi_ND ndd_addr:$src, -128)>;
-  def : Pat<(add (loadi64 ndd_addr:$src), 128),
+  def : Pat<(add_without_nuw (loadi64 ndd_addr:$src), 128),
             (SUB64mi32_ND ndd_addr:$src, -128)>;
 }
 
 // The same trick applies for 32-bit immediate fields in 64-bit
 // instructions.
 let Predicates = [NoNDD] in {
-  def : Pat<(add GR64:$src1, 0x0000000080000000),
+  def : Pat<(add_without_nuw GR64:$src1, 0x0000000080000000),
             (SUB64ri32 GR64:$src1, 0xffffffff80000000)>;
   def : Pat<(X86add_flag_nocf GR64:$src1, 0x0000000080000000),
             (SUB64ri32 GR64:$src1, 0xffffffff80000000)>;
 }
 let Predicates = [HasNDD] in {
-  def : Pat<(add GR64:$src1, 0x0000000080000000),
+  def : Pat<(add_without_nuw GR64:$src1, 0x0000000080000000),
             (SUB64ri32_ND GR64:$src1, 0xffffffff80000000)>;
   def : Pat<(X86add_flag_nocf GR64:$src1, 0x0000000080000000),
             (SUB64ri32_ND GR64:$src1, 0xffffffff80000000)>;
 }
-def : Pat<(store (add (loadi64 addr:$dst), 0x0000000080000000), addr:$dst),
+def : Pat<(store (add_without_nuw (loadi64 addr:$dst), 0x0000000080000000), addr:$dst),
           (SUB64mi32 addr:$dst, 0xffffffff80000000)>;
 let Predicates = [HasNDD] in {
-  def : Pat<(add(loadi64 ndd_addr:$src), 0x0000000080000000),
+  def : Pat<(add_without_nuw (loadi64 ndd_addr:$src), 0x0000000080000000),
             (SUB64mi32_ND ndd_addr:$src, 0xffffffff80000000)>;
 }
 
diff --git a/llvm/test/CodeGen/X86/avx512vnni-combine.ll b/llvm/test/CodeGen/X86/avx512vnni-combine.ll
index b7d950e994241..a3345fb239b30 100644
--- a/llvm/test/CodeGen/X86/avx512vnni-combine.ll
+++ b/llvm/test/CodeGen/X86/avx512vnni-combine.ll
@@ -189,7 +189,7 @@ define void @bar_512(i32 %0, ptr %1, <8 x i64> %2, ptr %3) {
 ; CHECK-NEXT:    vpaddd %zmm2, %zmm1, %zmm1
 ; CHECK-NEXT:    vmovdqa64 %zmm1, (%rsi,%r8)
 ; CHECK-NEXT:    addq $2, %rcx
-; CHECK-NEXT:    subq $-128, %r8
+; CHECK-NEXT:    addq $128, %r8
 ; CHECK-NEXT:    cmpq %rcx, %rdi
 ; CHECK-NEXT:    jne .LBB2_7
 ; CHECK-NEXT:  .LBB2_3:



More information about the llvm-commits mailing list