[llvm] [X86] Use MOV32rr instead of COPY when folding 32-bit identity ops… (PR #221865)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 7 19:49:50 PDT 2026


https://github.com/AZero13 updated https://github.com/llvm/llvm-project/pull/221865

>From 6d8fc6a372ec9f518352545c16ddd6d88ffc38e7 Mon Sep 17 00:00:00 2001
From: AZero13 <gfunni234 at gmail.com>
Date: Mon, 7 Sep 2026 19:47:46 -0400
Subject: [PATCH 1/3] Precommit test showing PeepholeOptimizer folding missing
 zero-extension

---
 .../X86/peephole-fold-or32-zero-extend.ll     | 72 +++++++++++++++++++
 1 file changed, 72 insertions(+)
 create mode 100644 llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll

diff --git a/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll b/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
new file mode 100644
index 0000000000000..2ce27650edbd8
--- /dev/null
+++ b/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
@@ -0,0 +1,72 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu | FileCheck %s
+
+; Verify that the PeepholeOptimizer does not fold a 32-bit identity operation
+; (like OR32rr with 0) into a COPY when the source register's upper 32 bits
+; may contain garbage. On x86-64, 32-bit ALU instructions implicitly zero the
+; upper 32 bits; a COPY does not. Dropping the zero-extension breaks
+; SUBREG_TO_REG and causes miscompilation (e.g., bad jump table indices).
+
+define i64 @test_peephole_or32_fold(i64 %j) {
+; CHECK-LABEL: test_peephole_or32_fold:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    movl $2, %eax
+; CHECK-NEXT:    movl %edi, %ecx
+; CHECK-NEXT:    orl $-50, %ecx
+; CHECK-NEXT:    movl $2863311531, %edx # imm = 0xAAAAAAAB
+; CHECK-NEXT:    imulq %rcx, %rdx
+; CHECK-NEXT:    shrq $34, %rdx
+; CHECK-NEXT:    addl %edx, %edx
+; CHECK-NEXT:    leal (%rdx,%rdx,2), %edx
+; CHECK-NEXT:    subl %edx, %ecx
+; CHECK-NEXT:    jmpq *.LJTI0_0(,%rcx,8)
+; CHECK-NEXT:    .p2align 4
+; CHECK-NEXT:  .LBB0_3: # %sw.epilog
+; CHECK-NEXT:    # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    decq %rax
+; CHECK-NEXT:    je .LBB0_4
+; CHECK-NEXT:  # %bb.1: # %for.body
+; CHECK-NEXT:    # in Loop: Header=BB0_3 Depth=1
+; CHECK-NEXT:    movl $2863311531, %ecx # imm = 0xAAAAAAAB
+; CHECK-NEXT:    imulq %rdi, %rcx
+; CHECK-NEXT:    shrq $34, %rcx
+; CHECK-NEXT:    addl %ecx, %ecx
+; CHECK-NEXT:    leal (%rcx,%rcx,2), %ecx
+; CHECK-NEXT:    movl %edi, %edx
+; CHECK-NEXT:    subl %ecx, %edx
+; CHECK-NEXT:    jmpq *.LJTI0_0(,%rdx,8)
+; CHECK-NEXT:  .LBB0_4: # %common.ret
+; CHECK-NEXT:    xorl %eax, %eax
+; CHECK-NEXT:    retq
+; CHECK-NEXT:  .LBB0_2: # %default.unreachable48
+entry:
+  br label %for.body
+
+common.ret:                                       ; preds = %sw.epilog, %for.body
+  ret i64 0
+
+for.body:                                         ; preds = %sw.epilog, %entry
+  %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %sw.epilog ]
+  %add = phi i32 [ -50, %entry ], [ 0, %sw.epilog ]
+  %conv = trunc i64 %j to i32
+  %sub = or i32 %add, %conv
+  %rem = urem i32 %sub, 6
+  switch i32 %rem, label %default.unreachable48 [
+    i32 3, label %sw.epilog
+    i32 0, label %common.ret
+    i32 4, label %sw.bb15
+    i32 2, label %sw.bb15
+  ]
+
+sw.bb15:                                          ; preds = %for.body, %for.body
+  %xor16 = xor i32 0, 0
+  br label %sw.epilog
+
+default.unreachable48:                            ; preds = %for.body
+  unreachable
+
+sw.epilog:                                        ; preds = %sw.bb15, %for.body
+  %indvars.iv.next = add i64 %indvars.iv, 1
+  %exitcond.not = icmp eq i64 %indvars.iv, 1
+  br i1 %exitcond.not, label %common.ret, label %for.body
+}

>From 37cbba60ee9fd5d516f4279dfe609a617c4934cd Mon Sep 17 00:00:00 2001
From: AZero13 <gfunni234 at gmail.com>
Date: Mon, 7 Sep 2026 22:25:17 -0400
Subject: [PATCH 2/3] [X86] Use MOV32rr instead of COPY when folding 32-bit
 identity ops on 64-bit targets

Converting a 32-bit ALU instruction into a COPY completely destroys the hardware's implicit zero-extending property. This causes subsequent SUBREG_TO_REG instructions to silently operate on garbage upper bits when they are coalesced.

Fixes #213979
---
 llvm/lib/Target/X86/X86InstrInfo.cpp          | 39 ++++++++++++++-----
 .../X86/peephole-fold-or32-zero-extend.ll     | 16 ++++----
 2 files changed, 37 insertions(+), 18 deletions(-)

diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 11a37efd28948..675379d401f74 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -5870,14 +5870,18 @@ static bool canConvert2Copy(unsigned Opc) {
   switch (Opc) {
   default:
     return false;
-  CASE_ND(ADD64ri32)
-  CASE_ND(SUB64ri32)
-  CASE_ND(OR64ri32)
-  CASE_ND(XOR64ri32)
-  CASE_ND(ADD32ri)
-  CASE_ND(SUB32ri)
-  CASE_ND(OR32ri)
-  CASE_ND(XOR32ri)
+    CASE_ND(ADD64ri32)
+    CASE_ND(SUB64ri32)
+    CASE_ND(OR64ri32)
+    CASE_ND(XOR64ri32)
+    CASE_ND(ADD32ri)
+    CASE_ND(SUB32ri)
+    CASE_ND(OR32ri)
+    CASE_ND(XOR32ri)
+    CASE_ND(ADD32rr)
+    CASE_ND(SUB32rr)
+    CASE_ND(OR32rr)
+    CASE_ND(XOR32rr)
     return true;
   }
 }
@@ -6072,8 +6076,23 @@ bool X86InstrInfo::foldImmediateImpl(MachineInstr &UseMI, MachineInstr *DefMI,
         UseMI.registerDefIsDead(X86::EFLAGS, /*TRI=*/nullptr)) {
       //          %100 = add %101, 0
       //    ==>
-      //          %100 = COPY %101
-      UseMI.setDesc(get(TargetOpcode::COPY));
+      //          %100 = COPY %101 (or MOV32rr on 64-bit targets)
+      unsigned CopyOpc = TargetOpcode::COPY;
+      if (Subtarget.is64Bit()) {
+        switch (NewOpc) {
+        case X86::ADD32ri:
+        case X86::SUB32ri:
+        case X86::OR32ri:
+        case X86::XOR32ri:
+        case X86::ADD32rr:
+        case X86::SUB32rr:
+        case X86::OR32rr:
+        case X86::XOR32rr:
+          CopyOpc = X86::MOV32rr;
+          break;
+        }
+      }
+      UseMI.setDesc(get(CopyOpc));
       UseMI.removeOperand(
           UseMI.findRegisterUseOperandIdx(Reg, /*TRI=*/nullptr));
       UseMI.removeOperand(
diff --git a/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll b/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
index 2ce27650edbd8..a8537990f5a65 100644
--- a/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
+++ b/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
@@ -27,14 +27,14 @@ define i64 @test_peephole_or32_fold(i64 %j) {
 ; CHECK-NEXT:    je .LBB0_4
 ; CHECK-NEXT:  # %bb.1: # %for.body
 ; CHECK-NEXT:    # in Loop: Header=BB0_3 Depth=1
-; CHECK-NEXT:    movl $2863311531, %ecx # imm = 0xAAAAAAAB
-; CHECK-NEXT:    imulq %rdi, %rcx
-; CHECK-NEXT:    shrq $34, %rcx
-; CHECK-NEXT:    addl %ecx, %ecx
-; CHECK-NEXT:    leal (%rcx,%rcx,2), %ecx
-; CHECK-NEXT:    movl %edi, %edx
-; CHECK-NEXT:    subl %ecx, %edx
-; CHECK-NEXT:    jmpq *.LJTI0_0(,%rdx,8)
+; CHECK-NEXT:    movl %edi, %ecx
+; CHECK-NEXT:    movl $2863311531, %edx # imm = 0xAAAAAAAB
+; CHECK-NEXT:    imulq %rcx, %rdx
+; CHECK-NEXT:    shrq $34, %rdx
+; CHECK-NEXT:    addl %edx, %edx
+; CHECK-NEXT:    leal (%rdx,%rdx,2), %edx
+; CHECK-NEXT:    subl %edx, %ecx
+; CHECK-NEXT:    jmpq *.LJTI0_0(,%rcx,8)
 ; CHECK-NEXT:  .LBB0_4: # %common.ret
 ; CHECK-NEXT:    xorl %eax, %eax
 ; CHECK-NEXT:    retq

>From 886c803917ead39f504d5ba1253e7bf518ed90d7 Mon Sep 17 00:00:00 2001
From: AZero13 <gfunni234 at gmail.com>
Date: Mon, 7 Sep 2026 22:49:39 -0400
Subject: [PATCH 3/3] [X86] Fold OR32ri/XOR32ri with 0 into OR32rr when EFLAGS
 are live

If a 32-bit (or 64/16/8-bit) OR or XOR with an immediate 0 is encountered,
and EFLAGS are required by a downstream instruction, this patch folds it
into an ORrr with the same source register.

For example,  becomes . This saves 1
byte of instruction encoding size (e.g. 3 bytes to 2 bytes) while setting
the condition codes identically. (Note: we cannot do this for ADD/SUB,
because they explicitly set AF=0, while OR leaves AF undefined).

Suggested by Sei K.
---
 llvm/lib/Target/X86/X86InstrInfo.cpp | 20 ++++++++++++++++++--
 1 file changed, 18 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 675379d401f74..de32cc20bbbda 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -6114,8 +6114,24 @@ bool X86InstrInfo::foldImmediateImpl(MachineInstr &UseMI, MachineInstr *DefMI,
         commuteInstruction(UseMI);
 
       assert(UseMI.getOperand(ImmOpNum).getReg() == Reg);
-      UseMI.setDesc(get(NewOpc));
-      UseMI.getOperand(ImmOpNum).ChangeToImmediate(ImmVal);
+
+      unsigned NewOpcrr = 0;
+      if (ImmVal == 0) {
+        switch (NewOpc) {
+        case X86::OR64ri32: case X86::XOR64ri32: NewOpcrr = X86::OR64rr; break;
+        case X86::OR32ri:   case X86::XOR32ri:   NewOpcrr = X86::OR32rr; break;
+        case X86::OR16ri:   case X86::XOR16ri:   NewOpcrr = X86::OR16rr; break;
+        case X86::OR8ri:    case X86::XOR8ri:    NewOpcrr = X86::OR8rr;  break;
+        }
+      }
+
+      if (NewOpcrr) {
+        UseMI.setDesc(get(NewOpcrr));
+        UseMI.getOperand(ImmOpNum).ChangeToRegister(UseMI.getOperand(1).getReg(), false);
+      } else {
+        UseMI.setDesc(get(NewOpc));
+        UseMI.getOperand(ImmOpNum).ChangeToImmediate(ImmVal);
+      }
     }
   }
 



More information about the llvm-commits mailing list