[llvm] [X86] Use MOV32rr instead of COPY when folding 32-bit identity ops… (PR #221865)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 7 19:49:50 PDT 2026
https://github.com/AZero13 updated https://github.com/llvm/llvm-project/pull/221865
>From 6d8fc6a372ec9f518352545c16ddd6d88ffc38e7 Mon Sep 17 00:00:00 2001
From: AZero13 <gfunni234 at gmail.com>
Date: Mon, 7 Sep 2026 19:47:46 -0400
Subject: [PATCH 1/3] Precommit test showing PeepholeOptimizer folding missing
zero-extension
---
.../X86/peephole-fold-or32-zero-extend.ll | 72 +++++++++++++++++++
1 file changed, 72 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
diff --git a/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll b/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
new file mode 100644
index 0000000000000..2ce27650edbd8
--- /dev/null
+++ b/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
@@ -0,0 +1,72 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu | FileCheck %s
+
+; Verify that the PeepholeOptimizer does not fold a 32-bit identity operation
+; (like OR32rr with 0) into a COPY when the source register's upper 32 bits
+; may contain garbage. On x86-64, 32-bit ALU instructions implicitly zero the
+; upper 32 bits; a COPY does not. Dropping the zero-extension breaks
+; SUBREG_TO_REG and causes miscompilation (e.g., bad jump table indices).
+
+define i64 @test_peephole_or32_fold(i64 %j) {
+; CHECK-LABEL: test_peephole_or32_fold:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: movl $2, %eax
+; CHECK-NEXT: movl %edi, %ecx
+; CHECK-NEXT: orl $-50, %ecx
+; CHECK-NEXT: movl $2863311531, %edx # imm = 0xAAAAAAAB
+; CHECK-NEXT: imulq %rcx, %rdx
+; CHECK-NEXT: shrq $34, %rdx
+; CHECK-NEXT: addl %edx, %edx
+; CHECK-NEXT: leal (%rdx,%rdx,2), %edx
+; CHECK-NEXT: subl %edx, %ecx
+; CHECK-NEXT: jmpq *.LJTI0_0(,%rcx,8)
+; CHECK-NEXT: .p2align 4
+; CHECK-NEXT: .LBB0_3: # %sw.epilog
+; CHECK-NEXT: # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: decq %rax
+; CHECK-NEXT: je .LBB0_4
+; CHECK-NEXT: # %bb.1: # %for.body
+; CHECK-NEXT: # in Loop: Header=BB0_3 Depth=1
+; CHECK-NEXT: movl $2863311531, %ecx # imm = 0xAAAAAAAB
+; CHECK-NEXT: imulq %rdi, %rcx
+; CHECK-NEXT: shrq $34, %rcx
+; CHECK-NEXT: addl %ecx, %ecx
+; CHECK-NEXT: leal (%rcx,%rcx,2), %ecx
+; CHECK-NEXT: movl %edi, %edx
+; CHECK-NEXT: subl %ecx, %edx
+; CHECK-NEXT: jmpq *.LJTI0_0(,%rdx,8)
+; CHECK-NEXT: .LBB0_4: # %common.ret
+; CHECK-NEXT: xorl %eax, %eax
+; CHECK-NEXT: retq
+; CHECK-NEXT: .LBB0_2: # %default.unreachable48
+entry:
+ br label %for.body
+
+common.ret: ; preds = %sw.epilog, %for.body
+ ret i64 0
+
+for.body: ; preds = %sw.epilog, %entry
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %sw.epilog ]
+ %add = phi i32 [ -50, %entry ], [ 0, %sw.epilog ]
+ %conv = trunc i64 %j to i32
+ %sub = or i32 %add, %conv
+ %rem = urem i32 %sub, 6
+ switch i32 %rem, label %default.unreachable48 [
+ i32 3, label %sw.epilog
+ i32 0, label %common.ret
+ i32 4, label %sw.bb15
+ i32 2, label %sw.bb15
+ ]
+
+sw.bb15: ; preds = %for.body, %for.body
+ %xor16 = xor i32 0, 0
+ br label %sw.epilog
+
+default.unreachable48: ; preds = %for.body
+ unreachable
+
+sw.epilog: ; preds = %sw.bb15, %for.body
+ %indvars.iv.next = add i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv, 1
+ br i1 %exitcond.not, label %common.ret, label %for.body
+}
>From 37cbba60ee9fd5d516f4279dfe609a617c4934cd Mon Sep 17 00:00:00 2001
From: AZero13 <gfunni234 at gmail.com>
Date: Mon, 7 Sep 2026 22:25:17 -0400
Subject: [PATCH 2/3] [X86] Use MOV32rr instead of COPY when folding 32-bit
identity ops on 64-bit targets
Converting a 32-bit ALU instruction into a COPY completely destroys the hardware's implicit zero-extending property. This causes subsequent SUBREG_TO_REG instructions to silently operate on garbage upper bits when they are coalesced.
Fixes #213979
---
llvm/lib/Target/X86/X86InstrInfo.cpp | 39 ++++++++++++++-----
.../X86/peephole-fold-or32-zero-extend.ll | 16 ++++----
2 files changed, 37 insertions(+), 18 deletions(-)
diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 11a37efd28948..675379d401f74 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -5870,14 +5870,18 @@ static bool canConvert2Copy(unsigned Opc) {
switch (Opc) {
default:
return false;
- CASE_ND(ADD64ri32)
- CASE_ND(SUB64ri32)
- CASE_ND(OR64ri32)
- CASE_ND(XOR64ri32)
- CASE_ND(ADD32ri)
- CASE_ND(SUB32ri)
- CASE_ND(OR32ri)
- CASE_ND(XOR32ri)
+ CASE_ND(ADD64ri32)
+ CASE_ND(SUB64ri32)
+ CASE_ND(OR64ri32)
+ CASE_ND(XOR64ri32)
+ CASE_ND(ADD32ri)
+ CASE_ND(SUB32ri)
+ CASE_ND(OR32ri)
+ CASE_ND(XOR32ri)
+ CASE_ND(ADD32rr)
+ CASE_ND(SUB32rr)
+ CASE_ND(OR32rr)
+ CASE_ND(XOR32rr)
return true;
}
}
@@ -6072,8 +6076,23 @@ bool X86InstrInfo::foldImmediateImpl(MachineInstr &UseMI, MachineInstr *DefMI,
UseMI.registerDefIsDead(X86::EFLAGS, /*TRI=*/nullptr)) {
// %100 = add %101, 0
// ==>
- // %100 = COPY %101
- UseMI.setDesc(get(TargetOpcode::COPY));
+ // %100 = COPY %101 (or MOV32rr on 64-bit targets)
+ unsigned CopyOpc = TargetOpcode::COPY;
+ if (Subtarget.is64Bit()) {
+ switch (NewOpc) {
+ case X86::ADD32ri:
+ case X86::SUB32ri:
+ case X86::OR32ri:
+ case X86::XOR32ri:
+ case X86::ADD32rr:
+ case X86::SUB32rr:
+ case X86::OR32rr:
+ case X86::XOR32rr:
+ CopyOpc = X86::MOV32rr;
+ break;
+ }
+ }
+ UseMI.setDesc(get(CopyOpc));
UseMI.removeOperand(
UseMI.findRegisterUseOperandIdx(Reg, /*TRI=*/nullptr));
UseMI.removeOperand(
diff --git a/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll b/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
index 2ce27650edbd8..a8537990f5a65 100644
--- a/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
+++ b/llvm/test/CodeGen/X86/peephole-fold-or32-zero-extend.ll
@@ -27,14 +27,14 @@ define i64 @test_peephole_or32_fold(i64 %j) {
; CHECK-NEXT: je .LBB0_4
; CHECK-NEXT: # %bb.1: # %for.body
; CHECK-NEXT: # in Loop: Header=BB0_3 Depth=1
-; CHECK-NEXT: movl $2863311531, %ecx # imm = 0xAAAAAAAB
-; CHECK-NEXT: imulq %rdi, %rcx
-; CHECK-NEXT: shrq $34, %rcx
-; CHECK-NEXT: addl %ecx, %ecx
-; CHECK-NEXT: leal (%rcx,%rcx,2), %ecx
-; CHECK-NEXT: movl %edi, %edx
-; CHECK-NEXT: subl %ecx, %edx
-; CHECK-NEXT: jmpq *.LJTI0_0(,%rdx,8)
+; CHECK-NEXT: movl %edi, %ecx
+; CHECK-NEXT: movl $2863311531, %edx # imm = 0xAAAAAAAB
+; CHECK-NEXT: imulq %rcx, %rdx
+; CHECK-NEXT: shrq $34, %rdx
+; CHECK-NEXT: addl %edx, %edx
+; CHECK-NEXT: leal (%rdx,%rdx,2), %edx
+; CHECK-NEXT: subl %edx, %ecx
+; CHECK-NEXT: jmpq *.LJTI0_0(,%rcx,8)
; CHECK-NEXT: .LBB0_4: # %common.ret
; CHECK-NEXT: xorl %eax, %eax
; CHECK-NEXT: retq
>From 886c803917ead39f504d5ba1253e7bf518ed90d7 Mon Sep 17 00:00:00 2001
From: AZero13 <gfunni234 at gmail.com>
Date: Mon, 7 Sep 2026 22:49:39 -0400
Subject: [PATCH 3/3] [X86] Fold OR32ri/XOR32ri with 0 into OR32rr when EFLAGS
are live
If a 32-bit (or 64/16/8-bit) OR or XOR with an immediate 0 is encountered,
and EFLAGS are required by a downstream instruction, this patch folds it
into an ORrr with the same source register.
For example, becomes . This saves 1
byte of instruction encoding size (e.g. 3 bytes to 2 bytes) while setting
the condition codes identically. (Note: we cannot do this for ADD/SUB,
because they explicitly set AF=0, while OR leaves AF undefined).
Suggested by Sei K.
---
llvm/lib/Target/X86/X86InstrInfo.cpp | 20 ++++++++++++++++++--
1 file changed, 18 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 675379d401f74..de32cc20bbbda 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -6114,8 +6114,24 @@ bool X86InstrInfo::foldImmediateImpl(MachineInstr &UseMI, MachineInstr *DefMI,
commuteInstruction(UseMI);
assert(UseMI.getOperand(ImmOpNum).getReg() == Reg);
- UseMI.setDesc(get(NewOpc));
- UseMI.getOperand(ImmOpNum).ChangeToImmediate(ImmVal);
+
+ unsigned NewOpcrr = 0;
+ if (ImmVal == 0) {
+ switch (NewOpc) {
+ case X86::OR64ri32: case X86::XOR64ri32: NewOpcrr = X86::OR64rr; break;
+ case X86::OR32ri: case X86::XOR32ri: NewOpcrr = X86::OR32rr; break;
+ case X86::OR16ri: case X86::XOR16ri: NewOpcrr = X86::OR16rr; break;
+ case X86::OR8ri: case X86::XOR8ri: NewOpcrr = X86::OR8rr; break;
+ }
+ }
+
+ if (NewOpcrr) {
+ UseMI.setDesc(get(NewOpcrr));
+ UseMI.getOperand(ImmOpNum).ChangeToRegister(UseMI.getOperand(1).getReg(), false);
+ } else {
+ UseMI.setDesc(get(NewOpc));
+ UseMI.getOperand(ImmOpNum).ChangeToImmediate(ImmVal);
+ }
}
}
More information about the llvm-commits
mailing list