[llvm] [X86] Remove `def32` and drop redundant zero extensions after `isel` (PR #225462)
Akash Manna via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 26 23:25:34 PDT 2026
https://github.com/akash-manna-sky updated https://github.com/llvm/llvm-project/pull/225462
>From f1fd1e9c35dfc32a851c2a6eed4917f3e8953ba8 Mon Sep 17 00:00:00 2001
From: Akash Manna <akash.manna.mymail at gmail.com>
Date: Wed, 16 Sep 2026 22:40:06 +0530
Subject: [PATCH 1/3] [X86] Add tests for implicit i32->i64 zero extension
(NFC)
Pre-commit the current codegen for #122104. The pr222714 and pr218382
functions use the index without zeroing its upper 32 bits.
---
llvm/test/CodeGen/X86/pr122104.ll | 128 ++++++++++++++++++++++++++++++
1 file changed, 128 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/pr122104.ll
diff --git a/llvm/test/CodeGen/X86/pr122104.ll b/llvm/test/CodeGen/X86/pr122104.ll
new file mode 100644
index 0000000000000..3d4b20b789acd
--- /dev/null
+++ b/llvm/test/CodeGen/X86/pr122104.ll
@@ -0,0 +1,128 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu | FileCheck %s
+
+; The zero_extend of the index is selected before its 'and' operand, which
+; shrinkAndImmediate then replaces with the truncated cmov result. The upper
+; 32 bits of that value are not zero, so the index must still be zero-extended.
+define i64 @pr222714(i64 %x, ptr %p) {
+; CHECK-LABEL: pr222714:
+; CHECK: # %bb.0:
+; CHECK-NEXT: shlq $30, %rdi
+; CHECK-NEXT: movabsq $17179869183, %rcx # imm = 0x3FFFFFFFF
+; CHECK-NEXT: movabsq $-4294967295, %rax # imm = 0xFFFFFFFF00000001
+; CHECK-NEXT: addq %rax, %rcx
+; CHECK-NEXT: testq %rdi, %rdi
+; CHECK-NEXT: cmovneq %rax, %rcx
+; CHECK-NEXT: movl %ecx, %eax
+; CHECK-NEXT: addq (%rsi,%rcx,8), %rax
+; CHECK-NEXT: retq
+ %m = and i64 %x, 17179869183
+ %c = icmp eq i64 %m, 0
+ %s = select i1 %c, i64 17179869183, i64 0
+ %a = add i64 %s, -4294967295
+ %idx = and i64 %a, 1
+ %gep = getelementptr inbounds i64, ptr %p, i64 %idx
+ %ld = load i64, ptr %gep, align 8
+ %lo = and i64 %a, 255
+ %r = add i64 %ld, %lo
+ ret i64 %r
+}
+
+; Same through a vector round trip: bit 32 of the extracted lane is set.
+define <4 x i32> @pr218382(i32 %x) "target-features"="+avx2,+popcnt" {
+; CHECK-LABEL: pr218382:
+; CHECK: # %bb.0:
+; CHECK-NEXT: andl $1, %edi
+; CHECK-NEXT: popcntl %edi, %eax
+; CHECK-NEXT: vpbroadcastd {{.*#+}} xmm0 = [1,1,1,1]
+; CHECK-NEXT: vpinsrd $0, %eax, %xmm0, %xmm0
+; CHECK-NEXT: vmovq %xmm0, %rax
+; CHECK-NEXT: vpxor %xmm0, %xmm0, %xmm0
+; CHECK-NEXT: vmovdqa %xmm0, -{{[0-9]+}}(%rsp)
+; CHECK-NEXT: movl $0, -24(%rsp,%rax,4)
+; CHECK-NEXT: vmovaps -{{[0-9]+}}(%rsp), %xmm0
+; CHECK-NEXT: retq
+ %a = and i32 %x, 1
+ %c = call i32 @llvm.ctpop.i32(i32 %a)
+ %v = insertelement <4 x i32> splat (i32 1), i32 %c, i64 0
+ %b = bitcast <4 x i32> %v to <2 x i64>
+ %e = extractelement <2 x i64> %b, i64 0
+ %idx = trunc i64 %e to i8
+ %r = insertelement <4 x i32> zeroinitializer, i32 0, i8 %idx
+ ret <4 x i32> %r
+}
+
+define i64 @zext_trunc(i64 %x) {
+; CHECK-LABEL: zext_trunc:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movl %edi, %eax
+; CHECK-NEXT: retq
+ %t = trunc i64 %x to i32
+ %z = zext i32 %t to i64
+ ret i64 %z
+}
+
+; 32-bit instructions zero the upper bits: no movl below.
+
+define i64 @zext_add(i32 %a, i32 %b) {
+; CHECK-LABEL: zext_add:
+; CHECK: # %bb.0:
+; CHECK-NEXT: # kill: def $esi killed $esi def $rsi
+; CHECK-NEXT: # kill: def $edi killed $edi def $rdi
+; CHECK-NEXT: leal (%rdi,%rsi), %eax
+; CHECK-NEXT: retq
+ %s = add i32 %a, %b
+ %z = zext i32 %s to i64
+ ret i64 %z
+}
+
+define i64 @zext_and(i32 %a) {
+; CHECK-LABEL: zext_and:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movzbl %dil, %eax
+; CHECK-NEXT: retq
+ %m = and i32 %a, 255
+ %z = zext i32 %m to i64
+ ret i64 %z
+}
+
+define i64 @zext_popcnt(i32 %a) "target-features"="+popcnt" {
+; CHECK-LABEL: zext_popcnt:
+; CHECK: # %bb.0:
+; CHECK-NEXT: popcntl %edi, %eax
+; CHECK-NEXT: retq
+ %c = call i32 @llvm.ctpop.i32(i32 %a)
+ %z = zext i32 %c to i64
+ ret i64 %z
+}
+
+define i64 @zext_cttz(i32 %a) {
+; CHECK-LABEL: zext_cttz:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movl $32, %eax
+; CHECK-NEXT: rep bsfl %edi, %eax
+; CHECK-NEXT: retq
+ %c = call i32 @llvm.cttz.i32(i32 %a, i1 false)
+ %z = zext i32 %c to i64
+ ret i64 %z
+}
+
+define i64 @zext_insert_byte(i32 %x, i32 %y, ptr %p) {
+; CHECK-LABEL: zext_insert_byte:
+; CHECK: # %bb.0:
+; CHECK-NEXT: # kill: def $esi killed $esi def $rsi
+; CHECK-NEXT: # kill: def $edi killed $edi def $rdi
+; CHECK-NEXT: leal (%rdi,%rsi), %eax
+; CHECK-NEXT: movb (%rdx), %al
+; CHECK-NEXT: retq
+ %s = add i32 %x, %y
+ %hi = and i32 %s, -256
+ %b = load i8, ptr %p
+ %zb = zext i8 %b to i32
+ %o = or i32 %hi, %zb
+ %z = zext i32 %o to i64
+ ret i64 %z
+}
+
+declare i32 @llvm.ctpop.i32(i32)
+declare i32 @llvm.cttz.i32(i32, i1)
>From 2ca7eaf9e074419c025ff3448b1f4e9795fffb9f Mon Sep 17 00:00:00 2001
From: Akash Manna <akash.manna.mymail at gmail.com>
Date: Wed, 16 Sep 2026 22:50:10 +0530
Subject: [PATCH 2/3] [X86] Remove def32 and drop redundant zero extensions
after isel
def32 decided from the unselected operand of a zero_extend that the
upper 32 bits would be zero. Isel selects users before operands, so a
later transform on that operand (shrinkAndImmediate replacing an 'and'
with its truncate) could break the assumption after the SUBREG_TO_REG
was emitted. Always select the zero extension with an explicit MOV32rr
and remove it in PostprocessISelDAG when the final machine node is a
real 32-bit instruction.
Fixes #122104
---
llvm/lib/Target/X86/X86ISelDAGToDAG.cpp | 97 +++++++++++++++++++++++-
llvm/lib/Target/X86/X86InstrCompiler.td | 44 +++--------
llvm/lib/Target/X86/X86InstrExtension.td | 9 +--
llvm/lib/Target/X86/X86InstrFragments.td | 19 -----
llvm/test/CodeGen/X86/insert.ll | 4 +-
llvm/test/CodeGen/X86/pr122104.ll | 13 ++--
6 files changed, 117 insertions(+), 69 deletions(-)
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index 4478930016f63..dda4cabc7345f 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -535,6 +535,11 @@ namespace {
return Mask.countr_one() >= Width;
}
+ /// Return true if the selected i32 value \p V is guaranteed to have the
+ /// upper 32 bits of its 64-bit super-register zeroed. Only valid once
+ /// every node has been selected.
+ bool isDef32(SDValue V) const;
+
/// Return an SDNode that returns the value of the global base register.
/// Output instructions required to initialize the global base register,
/// if necessary.
@@ -1623,6 +1628,76 @@ bool X86DAGToDAGISel::tryOptimizeRem8Extend(SDNode *N) {
return true;
}
+// Any x86-64 instruction writing a 32-bit register zeroes the upper 32 bits of
+// the 64-bit register. This is only decided after selection, on the final
+// machine node: isel selects users before operands, so a predicate on the
+// unselected operand of a zero_extend can be invalidated by a transform that
+// later replaces that operand (e.g. shrinkAndImmediate replacing a redundant
+// 'and' with its truncate operand).
+bool X86DAGToDAGISel::isDef32(SDValue V) const {
+ const X86RegisterInfo *TRI = Subtarget->getRegisterInfo();
+ while (V.getValueType() == MVT::i32 && V.isMachineOpcode()) {
+ unsigned Opc = V.getMachineOpcode();
+ switch (Opc) {
+ case TargetOpcode::COPY:
+ case TargetOpcode::COPY_TO_REGCLASS:
+ V = V.getOperand(0);
+ continue;
+ case TargetOpcode::INSERT_SUBREG: {
+ // The upper 32 bits are those of the base register.
+ unsigned SubIdx = V.getConstantOperandVal(2);
+ if (SubIdx != X86::sub_8bit && SubIdx != X86::sub_8bit_hi &&
+ SubIdx != X86::sub_16bit)
+ return false;
+ V = V.getOperand(0);
+ continue;
+ }
+ case X86::BSF32rr:
+ case X86::BSF32rm:
+ case X86::BSR32rr:
+ case X86::BSR32rm: {
+ // A zero source leaves the tied destination untouched, i.e. the
+ // fallback value. An IMPLICIT_DEF fallback means the result is poison.
+ SDValue Fallback = V.getOperand(0);
+ if (Fallback.isMachineOpcode() &&
+ Fallback.getMachineOpcode() == TargetOpcode::IMPLICIT_DEF)
+ return true;
+ V = Fallback;
+ continue;
+ }
+ // Pseudos that expand to a single instruction writing the whole register.
+ case X86::MOV32r0:
+ case X86::MOV32r1:
+ case X86::MOV32r_1:
+ case X86::SETB_C32r:
+ case X86::ADD32rr_DB:
+ case X86::ADD32ri_DB:
+ return true;
+ default:
+ break;
+ }
+
+ // EXTRACT_SUBREG, IMPLICIT_DEF and other pseudos (e.g. CMOV_GR32, which
+ // expands to a PHI) provide no guarantee.
+ if (Opc <= TargetOpcode::GENERIC_OP_END)
+ return false;
+ const MCInstrDesc &Desc = getInstrInfo()->get(Opc);
+ if (Desc.isPseudo())
+ return false;
+
+ // A real instruction writes all 32 bits of an explicit GR32 definition.
+ unsigned ResNo = V.getResNo();
+ if (ResNo >= Desc.getNumDefs())
+ return false;
+ const MCOperandInfo &OpInfo = Desc.operands()[ResNo];
+ if (OpInfo.OperandType != MCOI::OPERAND_REGISTER ||
+ OpInfo.isLookupRegClassByHwMode())
+ return false;
+ return X86::GR32RegClass.hasSubClassEq(TRI->getRegClass(OpInfo.RegClass));
+ }
+ return false;
+}
+
void X86DAGToDAGISel::PostprocessISelDAG() {
// Skip peepholes at -O0.
if (TM.getOptLevel() == CodeGenOptLevel::None)
@@ -1772,16 +1847,30 @@ void X86DAGToDAGISel::PostprocessISelDAG() {
MadeChange = true;
continue;
}
- // Attempt to remove vectors moves that were inserted to zero upper bits.
+ // Attempt to remove moves that were inserted to zero upper bits.
case TargetOpcode::SUBREG_TO_REG: {
unsigned SubRegIdx = N->getConstantOperandVal(1);
- if (SubRegIdx != X86::sub_xmm && SubRegIdx != X86::sub_ymm)
- continue;
-
SDValue Move = N->getOperand(0);
if (!Move.isMachineOpcode())
continue;
+ // (SUBREG_TO_REG (MOV32rr X)) from the i32->i64 zero_extend patterns:
+ // drop the MOV32rr if X already zeroes the upper 32 bits.
+ if (SubRegIdx == X86::sub_32bit) {
+ if (Move.getMachineOpcode() != X86::MOV32rr ||
+ !isDef32(Move.getOperand(0)))
+ continue;
+ SDNode *Res = CurDAG->UpdateNodeOperands(N, Move.getOperand(0),
+ N->getOperand(1));
+ if (Res != N)
+ ReplaceUses(N, Res);
+ MadeChange = true;
+ continue;
+ }
+
+ if (SubRegIdx != X86::sub_xmm && SubRegIdx != X86::sub_ymm)
+ continue;
+
// Make sure its one of the move opcodes we recognize.
switch (Move.getMachineOpcode()) {
default:
diff --git a/llvm/lib/Target/X86/X86InstrCompiler.td b/llvm/lib/Target/X86/X86InstrCompiler.td
index 45619663c45cb..f66d4c4a10111 100644
--- a/llvm/lib/Target/X86/X86InstrCompiler.td
+++ b/llvm/lib/Target/X86/X86InstrCompiler.td
@@ -1555,12 +1555,9 @@ def : Pat<(i64 (anyext GR32:$src)),
def : Pat<(i32 (anyext_sdiv GR8:$src)), (MOVSX32rr8 GR8:$src)>;
-// In the case of a 32-bit def that is known to implicitly zero-extend,
-// we can use a SUBREG_TO_REG.
-def : Pat<(i64 (zext def32:$src)),
- (SUBREG_TO_REG GR32:$src, sub_32bit)>;
-def : Pat<(i64 (and (anyext def32:$src), 0x00000000FFFFFFFF)),
- (SUBREG_TO_REG GR32:$src, sub_32bit)>;
+// Select like (i64 (zext GR32:$src)), see X86InstrExtension.td.
+def : Pat<(i64 (and (anyext GR32:$src), 0x00000000FFFFFFFF)),
+ (SUBREG_TO_REG (MOV32rr GR32:$src), sub_32bit)>;
//===----------------------------------------------------------------------===//
// Pattern match OR as ADD
@@ -1707,34 +1704,15 @@ def : Pat<(or (and GR64:$dst, -65536),
(i64 (zextloadi16 addr:$src))),
(INSERT_SUBREG (i64 (COPY $dst)), (MOV16rm i16mem:$src), sub_16bit)>;
-let Predicates = [Not64BitMode] in {
- def : Pat<(or (and GR32:$dst, -256),
- (i32 (zextloadi8 addr:$src))),
- (INSERT_SUBREG (i32 (COPY $dst)), (MOV8rm i8mem:$src), sub_8bit)>;
+// The upper 32 bits of the result are those of $dst; a zero extension of the
+// result only drops its MOV32rr if X86DAGToDAGISel::isDef32 holds for $dst.
+def : Pat<(or (and GR32:$dst, -256),
+ (i32 (zextloadi8 addr:$src))),
+ (INSERT_SUBREG (i32 (COPY $dst)), (MOV8rm i8mem:$src), sub_8bit)>;
- def : Pat<(or (and GR32:$dst, -65536),
- (i32 (zextloadi16 addr:$src))),
- (INSERT_SUBREG (i32 (COPY $dst)), (MOV16rm i16mem:$src), sub_16bit)>;
-}
-
-let Predicates = [In64BitMode] in {
- def : Pat<(or (and def32:$dst, -256),
- (i32 (zextloadi8 addr:$src))),
- (INSERT_SUBREG (i32 (COPY $dst)), (MOV8rm i8mem:$src), sub_8bit)>;
-
- def : Pat<(or (and def32:$dst, -65536),
- (i32 (zextloadi16 addr:$src))),
- (INSERT_SUBREG (i32 (COPY $dst)), (MOV16rm i16mem:$src), sub_16bit)>;
-
- // These patterns use MOV32rr since GR32 could have junk in the upper 32-bits.
- def : Pat<(or (and GR32:$dst, -256),
- (i32 (zextloadi8 addr:$src))),
- (MOV32rr (INSERT_SUBREG (i32 (COPY $dst)), (MOV8rm i8mem:$src), sub_8bit))>;
-
- def : Pat<(or (and GR32:$dst, -65536),
- (i32 (zextloadi16 addr:$src))),
- (MOV32rr (INSERT_SUBREG (i32 (COPY $dst)), (MOV16rm i16mem:$src), sub_16bit))>;
-}
+def : Pat<(or (and GR32:$dst, -65536),
+ (i32 (zextloadi16 addr:$src))),
+ (INSERT_SUBREG (i32 (COPY $dst)), (MOV16rm i16mem:$src), sub_16bit)>;
// To avoid needing to materialize an immediate in a register, use a 32-bit and
// with implicit zero-extension instead of a 64-bit and if the immediate has at
diff --git a/llvm/lib/Target/X86/X86InstrExtension.td b/llvm/lib/Target/X86/X86InstrExtension.td
index 7bf0cb1d64755..72b4ad3d31450 100644
--- a/llvm/lib/Target/X86/X86InstrExtension.td
+++ b/llvm/lib/Target/X86/X86InstrExtension.td
@@ -211,11 +211,10 @@ def : Pat<(i64 (zext GR16:$src)),
def : Pat<(zextloadi64i16 addr:$src),
(SUBREG_TO_REG (MOVZX32rm16 addr:$src), sub_32bit)>;
-// The preferred way to do 32-bit-to-64-bit zero extension on x86-64 is to use a
-// SUBREG_TO_REG to utilize implicit zero-extension, however this isn't possible
-// when the 32-bit value is defined by a truncate or is copied from something
-// where the high bits aren't necessarily all zero. In such cases, we fall back
-// to these explicit zext instructions.
+// 32-bit-to-64-bit zero extension always uses an explicit MOV32rr. Whether the
+// instruction that ends up defining $src already zeroes the upper 32 bits is
+// only known once every node has been selected, so the MOV32rr is removed in
+// X86DAGToDAGISel::PostprocessISelDAG instead (see isDef32).
def : Pat<(i64 (zext GR32:$src)),
(SUBREG_TO_REG (MOV32rr GR32:$src), sub_32bit)>;
def : Pat<(i64 (zextloadi64i32 addr:$src)),
diff --git a/llvm/lib/Target/X86/X86InstrFragments.td b/llvm/lib/Target/X86/X86InstrFragments.td
index 4a32284e79207..88dbbe5b599e5 100644
--- a/llvm/lib/Target/X86/X86InstrFragments.td
+++ b/llvm/lib/Target/X86/X86InstrFragments.td
@@ -841,25 +841,6 @@ def anyext_sdiv : PatFrag<(ops node:$lhs), (anyext node:$lhs),[{
N->getOperand(0).getResNo() == 1);
}]>;
-// Any instruction that defines a 32-bit result leaves the high half of the
-// register. Truncate can be lowered to EXTRACT_SUBREG. CopyFromReg may
-// be copying from a truncate. AssertSext/AssertZext/AssertAlign aren't saying
-// anything about the upper 32 bits, they're probably just qualifying a
-// CopyFromReg. FREEZE may be coming from a a truncate. BitScan fall through
-// values may not zero the upper bits correctly.
-// Any other 32-bit operation will zero-extend up to 64 bits.
-def def32 : PatLeaf<(i32 GR32:$src), [{
- return N->getOpcode() != ISD::TRUNCATE &&
- N->getOpcode() != TargetOpcode::EXTRACT_SUBREG &&
- N->getOpcode() != ISD::CopyFromReg &&
- N->getOpcode() != ISD::AssertSext &&
- N->getOpcode() != ISD::AssertZext &&
- N->getOpcode() != ISD::AssertAlign &&
- N->getOpcode() != ISD::FREEZE &&
- !((N->getOpcode() == X86ISD::BSF || N->getOpcode() == X86ISD::BSR) &&
- (!N->getOperand(0).isUndef() && !isa<ConstantSDNode>(N->getOperand(0))));
-}]>;
-
def shiftMask8 : PatFrag<(ops node:$lhs), (and node:$lhs, imm), [{
return isUnneededShiftMask(N, 3);
}]>;
diff --git a/llvm/test/CodeGen/X86/insert.ll b/llvm/test/CodeGen/X86/insert.ll
index 53d0040fa2464..f8b44881287de 100644
--- a/llvm/test/CodeGen/X86/insert.ll
+++ b/llvm/test/CodeGen/X86/insert.ll
@@ -58,8 +58,8 @@ define i32 @sub8_32(i32 noundef %res, ptr %byte) {
;
; X64-LABEL: sub8_32:
; X64: # %bb.0: # %entry
-; X64-NEXT: movb (%rsi), %dil
; X64-NEXT: movl %edi, %eax
+; X64-NEXT: movb (%rsi), %al
; X64-NEXT: retq
entry:
%and = and i32 %res, -256
@@ -81,8 +81,8 @@ define i32 @sub16_32(i32 noundef %res, ptr %byte) {
;
; X64-LABEL: sub16_32:
; X64: # %bb.0: # %entry
-; X64-NEXT: movw (%rsi), %di
; X64-NEXT: movl %edi, %eax
+; X64-NEXT: movw (%rsi), %ax
; X64-NEXT: retq
entry:
%and = and i32 %res, -65536
diff --git a/llvm/test/CodeGen/X86/pr122104.ll b/llvm/test/CodeGen/X86/pr122104.ll
index 3d4b20b789acd..2ee36a7bb64b7 100644
--- a/llvm/test/CodeGen/X86/pr122104.ll
+++ b/llvm/test/CodeGen/X86/pr122104.ll
@@ -8,13 +8,13 @@ define i64 @pr222714(i64 %x, ptr %p) {
; CHECK-LABEL: pr222714:
; CHECK: # %bb.0:
; CHECK-NEXT: shlq $30, %rdi
-; CHECK-NEXT: movabsq $17179869183, %rcx # imm = 0x3FFFFFFFF
-; CHECK-NEXT: movabsq $-4294967295, %rax # imm = 0xFFFFFFFF00000001
-; CHECK-NEXT: addq %rax, %rcx
+; CHECK-NEXT: movabsq $17179869183, %rax # imm = 0x3FFFFFFFF
+; CHECK-NEXT: movabsq $-4294967295, %rcx # imm = 0xFFFFFFFF00000001
+; CHECK-NEXT: addq %rcx, %rax
; CHECK-NEXT: testq %rdi, %rdi
-; CHECK-NEXT: cmovneq %rax, %rcx
-; CHECK-NEXT: movl %ecx, %eax
-; CHECK-NEXT: addq (%rsi,%rcx,8), %rax
+; CHECK-NEXT: cmovneq %rcx, %rax
+; CHECK-NEXT: movl %eax, %eax
+; CHECK-NEXT: addq (%rsi,%rax,8), %rax
; CHECK-NEXT: retq
%m = and i64 %x, 17179869183
%c = icmp eq i64 %m, 0
@@ -39,6 +39,7 @@ define <4 x i32> @pr218382(i32 %x) "target-features"="+avx2,+popcnt" {
; CHECK-NEXT: vmovq %xmm0, %rax
; CHECK-NEXT: vpxor %xmm0, %xmm0, %xmm0
; CHECK-NEXT: vmovdqa %xmm0, -{{[0-9]+}}(%rsp)
+; CHECK-NEXT: movl %eax, %eax
; CHECK-NEXT: movl $0, -24(%rsp,%rax,4)
; CHECK-NEXT: vmovaps -{{[0-9]+}}(%rsp), %xmm0
; CHECK-NEXT: retq
>From 864218afe8267800a2845bf8503639b1be4dc3d5 Mon Sep 17 00:00:00 2001
From: Akash Manna <akash.manna.mymail at gmail.com>
Date: Sun, 27 Sep 2026 11:54:40 +0530
Subject: [PATCH 3/3] [X86] Keep 32-bit div/mul results and -O0 zero extensions
free
The quotient and remainder of a 32-bit div/idiv, and the halves of a
32-bit mul/imul, reach the zero extension as EAX/EDX copies. These
instructions always write both registers, so accept them in isDef32,
together with the MULX32H pseudos. cmpxchg only writes EAX on failure
and stays rejected.
The old def32 pattern also applied at -O0, so run the MOV32rr removal
at every optimization level.
---
llvm/lib/Target/X86/X86ISelDAGToDAG.cpp | 100 ++++++++++++++++++------
llvm/test/CodeGen/X86/pr122104.ll | 27 +++++++
2 files changed, 103 insertions(+), 24 deletions(-)
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index dda4cabc7345f..c8964d44a9aeb 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -614,6 +614,7 @@ namespace {
SDValue &InGlue);
bool tryOptimizeRem8Extend(SDNode *N);
+ bool tryRemoveRedundantZExtMove(SDNode *N);
bool onlyUsesZeroFlag(SDValue Flags) const;
bool hasNoSignFlagUses(SDValue Flags) const;
@@ -1628,6 +1629,23 @@ bool X86DAGToDAGISel::tryOptimizeRem8Extend(SDNode *N) {
return true;
}
+// The one-operand 32-bit multiplies and divides always write EAX and EDX.
+static bool isMulDiv32(unsigned Opc) {
+ switch (Opc) {
+ case X86::DIV32r:
+ case X86::DIV32m:
+ case X86::IDIV32r:
+ case X86::IDIV32m:
+ case X86::MUL32r:
+ case X86::MUL32m:
+ case X86::IMUL32r:
+ case X86::IMUL32m:
+ return true;
+ default:
+ return false;
+ }
+}
+
// Any x86-64 instruction writing a 32-bit register zeroes the upper 32 bits of
// the 64-bit register. This is only decided after selection, on the final
// machine node: isel selects users before operands, so a predicate on the
@@ -1636,7 +1654,23 @@ bool X86DAGToDAGISel::tryOptimizeRem8Extend(SDNode *N) {
// 'and' with its truncate operand).
bool X86DAGToDAGISel::isDef32(SDValue V) const {
const X86RegisterInfo *TRI = Subtarget->getRegisterInfo();
- while (V.getValueType() == MVT::i32 && V.isMachineOpcode()) {
+ while (V.getValueType() == MVT::i32) {
+ // A copy out of EAX/EDX glued, possibly through other such copies, to the
+ // mul/div that wrote it. Other writers (e.g. cmpxchg) may leave it as is.
+ if (V.getOpcode() == ISD::CopyFromReg) {
+ Register Reg = cast<RegisterSDNode>(V.getOperand(1))->getReg();
+ if (V.getResNo() != 0 || (Reg != X86::EAX && Reg != X86::EDX))
+ return false;
+ SDValue Glue = V;
+ while (Glue.getOpcode() == ISD::CopyFromReg) {
+ if (Glue.getNumOperands() != 3)
+ return false;
+ Glue = Glue.getOperand(2);
+ }
+ return Glue.isMachineOpcode() && isMulDiv32(Glue.getMachineOpcode());
+ }
+ if (!V.isMachineOpcode())
+ return false;
unsigned Opc = V.getMachineOpcode();
switch (Opc) {
case TargetOpcode::COPY:
@@ -1672,6 +1706,8 @@ bool X86DAGToDAGISel::isDef32(SDValue V) const {
case X86::SETB_C32r:
case X86::ADD32rr_DB:
case X86::ADD32ri_DB:
+ case X86::MULX32Hrr:
+ case X86::MULX32Hrm:
return true;
default:
break;
@@ -1687,8 +1723,13 @@ bool X86DAGToDAGISel::isDef32(SDValue V) const {
// A real instruction writes all 32 bits of an explicit GR32 definition.
unsigned ResNo = V.getResNo();
- if (ResNo >= Desc.getNumDefs())
- return false;
+ if (ResNo >= Desc.getNumDefs()) {
+ // An implicit EAX/EDX result, e.g. of the MUL32r for X86ISD::UMUL.
+ unsigned Idx = ResNo - Desc.getNumDefs();
+ ArrayRef<MCPhysReg> ImpDefs = Desc.implicit_defs();
+ return isMulDiv32(Opc) && Idx < ImpDefs.size() &&
+ (ImpDefs[Idx] == X86::EAX || ImpDefs[Idx] == X86::EDX);
+ }
const MCOperandInfo &OpInfo = Desc.operands()[ResNo];
if (OpInfo.OperandType != MCOI::OPERAND_REGISTER ||
OpInfo.isLookupRegClassByHwMode())
@@ -1698,10 +1739,27 @@ bool X86DAGToDAGISel::isDef32(SDValue V) const {
return false;
}
+// The i32->i64 zero_extend patterns always emit (SUBREG_TO_REG (MOV32rr X)).
+// Drop the MOV32rr if X already zeroes the upper 32 bits.
+bool X86DAGToDAGISel::tryRemoveRedundantZExtMove(SDNode *N) {
+ if (N->getMachineOpcode() != TargetOpcode::SUBREG_TO_REG ||
+ N->getConstantOperandVal(1) != X86::sub_32bit)
+ return false;
+ SDValue Move = N->getOperand(0);
+ if (!Move.isMachineOpcode() || Move.getMachineOpcode() != X86::MOV32rr ||
+ !isDef32(Move.getOperand(0)))
+ return false;
+ SDNode *Res =
+ CurDAG->UpdateNodeOperands(N, Move.getOperand(0), N->getOperand(1));
+ if (Res != N)
+ ReplaceUses(N, Res);
+ return true;
+}
+
void X86DAGToDAGISel::PostprocessISelDAG() {
- // Skip peepholes at -O0.
- if (TM.getOptLevel() == CodeGenOptLevel::None)
- return;
+ // The zero_extend patterns rely on the MOV32rr removal at every opt level.
+ // Skip the other peepholes at -O0.
+ bool OptNone = TM.getOptLevel() == CodeGenOptLevel::None;
SelectionDAG::allnodes_iterator Position = CurDAG->allnodes_end();
@@ -1712,6 +1770,14 @@ void X86DAGToDAGISel::PostprocessISelDAG() {
if (N->use_empty() || !N->isMachineOpcode())
continue;
+ if (tryRemoveRedundantZExtMove(N)) {
+ MadeChange = true;
+ continue;
+ }
+
+ if (OptNone)
+ continue;
+
if (tryOptimizeRem8Extend(N)) {
MadeChange = true;
continue;
@@ -1847,28 +1913,14 @@ void X86DAGToDAGISel::PostprocessISelDAG() {
MadeChange = true;
continue;
}
- // Attempt to remove moves that were inserted to zero upper bits.
+ // Attempt to remove vectors moves that were inserted to zero upper bits.
case TargetOpcode::SUBREG_TO_REG: {
unsigned SubRegIdx = N->getConstantOperandVal(1);
- SDValue Move = N->getOperand(0);
- if (!Move.isMachineOpcode())
+ if (SubRegIdx != X86::sub_xmm && SubRegIdx != X86::sub_ymm)
continue;
- // (SUBREG_TO_REG (MOV32rr X)) from the i32->i64 zero_extend patterns:
- // drop the MOV32rr if X already zeroes the upper 32 bits.
- if (SubRegIdx == X86::sub_32bit) {
- if (Move.getMachineOpcode() != X86::MOV32rr ||
- !isDef32(Move.getOperand(0)))
- continue;
- SDNode *Res = CurDAG->UpdateNodeOperands(N, Move.getOperand(0),
- N->getOperand(1));
- if (Res != N)
- ReplaceUses(N, Res);
- MadeChange = true;
- continue;
- }
-
- if (SubRegIdx != X86::sub_xmm && SubRegIdx != X86::sub_ymm)
+ SDValue Move = N->getOperand(0);
+ if (!Move.isMachineOpcode())
continue;
// Make sure its one of the move opcodes we recognize.
diff --git a/llvm/test/CodeGen/X86/pr122104.ll b/llvm/test/CodeGen/X86/pr122104.ll
index bdf1a85e74375..4fbc15797a9cf 100644
--- a/llvm/test/CodeGen/X86/pr122104.ll
+++ b/llvm/test/CodeGen/X86/pr122104.ll
@@ -38,6 +38,20 @@ define i64 @zext_trunc(i64 %x) {
ret i64 %z
}
+; cmpxchg only writes EAX when the comparison fails.
+define i64 @zext_cmpxchg(ptr %p, i32 %cmp, i32 %new) {
+; CHECK-LABEL: zext_cmpxchg:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movl %esi, %eax
+; CHECK-NEXT: lock cmpxchgl %edx, (%rdi)
+; CHECK-NEXT: movl %eax, %eax
+; CHECK-NEXT: retq
+ %r = cmpxchg ptr %p, i32 %cmp, i32 %new seq_cst seq_cst
+ %v = extractvalue { i32, i1 } %r, 0
+ %z = zext i32 %v to i64
+ ret i64 %z
+}
+
; 32-bit instructions zero the upper bits: no movl below.
define i64 @zext_add(i32 %a, i32 %b) {
@@ -72,6 +86,19 @@ define i64 @zext_popcnt(i32 %a) "target-features"="+popcnt" {
ret i64 %z
}
+define i64 @zext_udiv(i32 %a, i32 %b) {
+; CHECK-LABEL: zext_udiv:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movl %edi, %eax
+; CHECK-NEXT: xorl %edx, %edx
+; CHECK-NEXT: divl %esi
+; CHECK-NEXT: # kill: def $eax killed $eax def $rax
+; CHECK-NEXT: retq
+ %d = udiv i32 %a, %b
+ %z = zext i32 %d to i64
+ ret i64 %z
+}
+
define i64 @zext_cttz(i32 %a) {
; CHECK-LABEL: zext_cttz:
; CHECK: # %bb.0:
More information about the llvm-commits
mailing list