[llvm] [DAG][X86] Bitfield insertion can combine into SHL + SHRD (PR #220182)

Pratyay Pande via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 1 00:55:05 PDT 2026


https://github.com/pratyaypande created https://github.com/llvm/llvm-project/pull/220182

This change combines the following pattern:

```llvm
define dso_local noundef i64 @updateTop10Bits(unsigned long, unsigned long)(i64 noundef %A, i64 noundef %B) local_unnamed_addr {
entry:
  %and = and i64 %A, 18014398509481983
  %shl = shl i64 %B, 54
  %or = or disjoint i64 %shl, %and
  ret i64 %or
}
```

Into an `shrd` usage instead of `bhziq`.

This change is intended to fix issue #112488 

>From 41fe1c2b7923511710c8758a78d347ba63146b17 Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Fri, 28 Aug 2026 01:25:02 +0530
Subject: [PATCH 01/11] Initial Combine implementation

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 49 +++++++++++++++++++++++++
 1 file changed, 49 insertions(+)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 3a66310ff1b3c..4e67db90b5b3d 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53768,6 +53768,52 @@ static SDValue combineOrXorWithSETCC(unsigned Opc, const SDLoc &DL, EVT VT,
   return SDValue();
 }
 
+// Matches the following pattern:
+//
+//   (or (and X, HighBitsMask(C)), (srl Y, C)) --> (fshl (srl X, BW-C), Y, BW-C)
+static SDValue combineDisjointORToSHLD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
+										const X86Subtarget &Subtarget) {
+	using namespace SDPatternMatch;
+
+	N->dump();
+
+	// Bail if the OR is not disjoint or if SHLD is slow
+	if (!N->getFlags().hasDisjoint() || Subtarget.isSHLDSlow())
+		return SDValue();
+
+	EVT VT = N->getValueType(0);
+	APInt Mask;
+	uint64_t ShiftAmount;
+	SDValue X, Y;
+
+	bool Match = sd_match(N, 
+			m_Or( m_OneUse(m_And( m_Value(X), m_ConstInt(Mask))), 
+					 m_OneUse(m_Shl( m_Value(Y), m_ConstInt(ShiftAmount)))));
+
+	// Max bit-width of operands
+	uint64_t MaxMaskBitWidth = VT.getSizeInBits();
+
+	// Check for Mask and ShiftAmount
+	//
+	// (shl Y, ShiftAmount) fills the top (MaxMaskBitWidth - ShiftAmount) bits,
+	// so X must keep exactly the low ShiftAmount.
+	APInt ExpectedMask = APInt::getLowBitsSet(MaxMaskBitWidth, ShiftAmount);
+
+	bool Applicable = Match && (ShiftAmount > 0) 
+							&& (ShiftAmount < MaxMaskBitWidth)
+							&& (Mask == ExpectedMask);
+
+	if (!Applicable)
+		return SDValue();
+
+	uint64_t InvShAmt = MaxMaskBitWidth - ShiftAmount;
+	SDValue ShVal = DAG.getShiftAmountConstant(InvShAmt, VT, DL);
+	SDValue SHLVal = DAG.getNode(ISD::SHL, DL, VT, X, ShVal);
+	SDValue fin = DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShVal);
+
+	return fin;
+}
+
 static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
                          TargetLowering::DAGCombinerInfo &DCI,
                          const X86Subtarget &Subtarget) {
@@ -53830,6 +53876,9 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
   if (SDValue R = combineOrWithGF2P8AFFINEQB(N, dl, DAG, VT))
     return R;
 
+  if (SDValue R = combineDisjointORToSHLD(N, dl, DAG, Subtarget))
+  	  return R;
+
   if (DCI.isBeforeLegalizeOps())
     return SDValue();
 

>From f80151951e0b5b242762f776749e5bf3f9fb0030 Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Mon, 31 Aug 2026 18:07:42 +0530
Subject: [PATCH 02/11] [DAG][X86] Do not bail if OR is not disjoint Removed
 debug statements

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 6 ++----
 1 file changed, 2 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 4e67db90b5b3d..728abe31e3ccb 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53775,10 +53775,8 @@ static SDValue combineDisjointORToSHLD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 										const X86Subtarget &Subtarget) {
 	using namespace SDPatternMatch;
 
-	N->dump();
-
-	// Bail if the OR is not disjoint or if SHLD is slow
-	if (!N->getFlags().hasDisjoint() || Subtarget.isSHLDSlow())
+	// Bail if SHLD is slow
+	if (Subtarget.isSHLDSlow())
 		return SDValue();
 
 	EVT VT = N->getValueType(0);

>From 94fba6273c53c94ae3231583df5cb9279d92b69d Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Mon, 31 Aug 2026 18:21:34 +0530
Subject: [PATCH 03/11] [DAG][X86] Bail early if pattern match fails Added
 descriptive comments

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 10 +++++++---
 1 file changed, 7 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 728abe31e3ccb..f05341162a0ca 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53784,10 +53784,15 @@ static SDValue combineDisjointORToSHLD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 	uint64_t ShiftAmount;
 	SDValue X, Y;
 
+	// Check for the following pattern:
+	//   (or (and X, HighBitsMask(C)), (srl Y, C))
 	bool Match = sd_match(N, 
 			m_Or( m_OneUse(m_And( m_Value(X), m_ConstInt(Mask))), 
 					 m_OneUse(m_Shl( m_Value(Y), m_ConstInt(ShiftAmount)))));
 
+	if (!Match)
+		return SDValue();
+
 	// Max bit-width of operands
 	uint64_t MaxMaskBitWidth = VT.getSizeInBits();
 
@@ -53797,9 +53802,8 @@ static SDValue combineDisjointORToSHLD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 	// so X must keep exactly the low ShiftAmount.
 	APInt ExpectedMask = APInt::getLowBitsSet(MaxMaskBitWidth, ShiftAmount);
 
-	bool Applicable = Match && (ShiftAmount > 0) 
-							&& (ShiftAmount < MaxMaskBitWidth)
-							&& (Mask == ExpectedMask);
+	bool Applicable = (ShiftAmount > 0) && (ShiftAmount < MaxMaskBitWidth)
+										&& (Mask == ExpectedMask);
 
 	if (!Applicable)
 		return SDValue();

>From 29f7e7311a1edfdb0e88146b1285bba61f1c9438 Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Mon, 31 Aug 2026 18:23:02 +0530
Subject: [PATCH 04/11] [DAG][X86][NFC] Cleaned up return statement

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index f05341162a0ca..bf93ad4c64313 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53808,12 +53808,12 @@ static SDValue combineDisjointORToSHLD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 	if (!Applicable)
 		return SDValue();
 
+	LLVM_DEBUG(dbgs() << "The optimization is applicable.\n");
+
 	uint64_t InvShAmt = MaxMaskBitWidth - ShiftAmount;
 	SDValue ShVal = DAG.getShiftAmountConstant(InvShAmt, VT, DL);
 	SDValue SHLVal = DAG.getNode(ISD::SHL, DL, VT, X, ShVal);
-	SDValue fin = DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShVal);
-
-	return fin;
+	return DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShVal);
 }
 
 static SDValue combineOr(SDNode *N, SelectionDAG &DAG,

>From 9afdc1138d7e1cf5e358824410e2c98eee73b810 Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Mon, 31 Aug 2026 21:16:19 +0530
Subject: [PATCH 05/11] [DAG][X86] SHRD does not support 8--bit operands

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 5 +++++
 1 file changed, 5 insertions(+)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index bf93ad4c64313..b487e3a38dad3 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53780,6 +53780,11 @@ static SDValue combineDisjointORToSHLD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 		return SDValue();
 
 	EVT VT = N->getValueType(0);
+
+	// SHRD does not suppport 8-bit operands
+	if (VT == MVT::i8)
+		return SDValue();
+
 	APInt Mask;
 	uint64_t ShiftAmount;
 	SDValue X, Y;

>From fec97cbf838a7d04d840638dcd6e51a25298c3c7 Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Tue, 1 Sep 2026 11:03:32 +0530
Subject: [PATCH 06/11] [DAG][X86] Renamed implementation to `combineORToSHRD`
 Added an assertion to check for `ISD::OR`. Updated comments to be a bit more
 descriptive Renamed variables

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 17 +++++++++--------
 1 file changed, 9 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index b487e3a38dad3..8946904f9a92b 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53768,12 +53768,13 @@ static SDValue combineOrXorWithSETCC(unsigned Opc, const SDLoc &DL, EVT VT,
   return SDValue();
 }
 
-// Matches the following pattern:
-//
-//   (or (and X, HighBitsMask(C)), (srl Y, C)) --> (fshl (srl X, BW-C), Y, BW-C)
-static SDValue combineDisjointORToSHLD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
+ 
+// Fold an OR with a masked destination and a left-shifted
+// source into a shift + double-precision shift (SHRD):
+static SDValue combineORToSHRD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 										const X86Subtarget &Subtarget) {
 	using namespace SDPatternMatch;
+	assert(N->getOpcode() == ISD::OR && "Invalid Node. Expected OR.");
 
 	// Bail if SHLD is slow
 	if (Subtarget.isSHLDSlow())
@@ -53816,9 +53817,9 @@ static SDValue combineDisjointORToSHLD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 	LLVM_DEBUG(dbgs() << "The optimization is applicable.\n");
 
 	uint64_t InvShAmt = MaxMaskBitWidth - ShiftAmount;
-	SDValue ShVal = DAG.getShiftAmountConstant(InvShAmt, VT, DL);
-	SDValue SHLVal = DAG.getNode(ISD::SHL, DL, VT, X, ShVal);
-	return DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShVal);
+	SDValue ShAConst = DAG.getShiftAmountConstant(InvShAmt, VT, DL);
+	SDValue SHLVal = DAG.getNode(ISD::SHL, DL, VT, X, ShAConst);
+	return DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShAConst);
 }
 
 static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
@@ -53883,7 +53884,7 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
   if (SDValue R = combineOrWithGF2P8AFFINEQB(N, dl, DAG, VT))
     return R;
 
-  if (SDValue R = combineDisjointORToSHLD(N, dl, DAG, Subtarget))
+  if (SDValue R = combineORToSHRD(N, dl, DAG, Subtarget))
   	  return R;
 
   if (DCI.isBeforeLegalizeOps())

>From 82068c80b065acf674c360bdd70c32a274a79d7c Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Tue, 1 Sep 2026 11:38:13 +0530
Subject: [PATCH 07/11] [DAG][X86][NFC] Rename impl. method to something more
 descriptive

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 8946904f9a92b..5e80c125b9757 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53771,7 +53771,7 @@ static SDValue combineOrXorWithSETCC(unsigned Opc, const SDLoc &DL, EVT VT,
  
 // Fold an OR with a masked destination and a left-shifted
 // source into a shift + double-precision shift (SHRD):
-static SDValue combineORToSHRD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
+static SDValue combineOrOnSHLToSHRD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 										const X86Subtarget &Subtarget) {
 	using namespace SDPatternMatch;
 	assert(N->getOpcode() == ISD::OR && "Invalid Node. Expected OR.");
@@ -53884,7 +53884,7 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
   if (SDValue R = combineOrWithGF2P8AFFINEQB(N, dl, DAG, VT))
     return R;
 
-  if (SDValue R = combineORToSHRD(N, dl, DAG, Subtarget))
+  if (SDValue R = combineOrOnSHLToSHRD(N, dl, DAG, Subtarget))
   	  return R;
 
   if (DCI.isBeforeLegalizeOps())

>From cf1bf80e32b0635e144a7e673194875ff1b0a155 Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Tue, 1 Sep 2026 11:50:55 +0530
Subject: [PATCH 08/11] [DAG][X86] Tests for `combineOrOnSHLToSHRD` in
 `X86ISelLowering.cpp`

---
 llvm/test/CodeGen/X86/insert-bitfield.ll | 349 +++++++++++++++++++++++
 1 file changed, 349 insertions(+)
 create mode 100644 llvm/test/CodeGen/X86/insert-bitfield.ll

diff --git a/llvm/test/CodeGen/X86/insert-bitfield.ll b/llvm/test/CodeGen/X86/insert-bitfield.ll
new file mode 100644
index 0000000000000..c51c0c2a2878f
--- /dev/null
+++ b/llvm/test/CodeGen/X86/insert-bitfield.ll
@@ -0,0 +1,349 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-- | FileCheck %s --check-prefixes=X64
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+slow-shld | FileCheck %s --check-prefixes=X64-SLOW
+
+define i64 @insert_10_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i64:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+define i64 @insert_10_i64_disjoint(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64_disjoint:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i64_disjoint:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or disjoint i64 %shl, %and
+  ret i64 %or
+}
+
+; Commuted operands.
+define i64 @insert_10_i64_commute(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64_commute:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i64_commute:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or i64 %and, %shl
+  ret i64 %or
+}
+
+define i64 @insert_33_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_33_i64:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $33, %rax
+; X64-NEXT:    shrdq $33, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_33_i64:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    andl $2147483647, %edi # imm = 0x7FFFFFFF
+; X64-SLOW-NEXT:    shlq $31, %rsi
+; X64-SLOW-NEXT:    leaq (%rsi,%rdi), %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 2147483647
+  %shl = shl i64 %b, 31
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+define i64 @insert_1_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_1_i64:
+; X64:       # %bb.0:
+; X64-NEXT:    leaq (%rdi,%rdi), %rax
+; X64-NEXT:    shrdq $1, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_1_i64:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $9223372036854775807, %rax # imm = 0x7FFFFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $63, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 9223372036854775807
+  %shl = shl i64 %b, 63
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; i32 inputs and mask
+define i32 @insert_10_i32(i32 %a, i32 %b) nounwind {
+; X64-LABEL: insert_10_i32:
+; X64:       # %bb.0:
+; X64-NEXT:    movl %edi, %eax
+; X64-NEXT:    shll $10, %eax
+; X64-NEXT:    shrdl $10, %esi, %eax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i32:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT:    andl $4194303, %edi # imm = 0x3FFFFF
+; X64-SLOW-NEXT:    shll $22, %esi
+; X64-SLOW-NEXT:    leal (%rsi,%rdi), %eax
+; X64-SLOW-NEXT:    retq
+  %and = and i32 %a, 4194303
+  %shl = shl i32 %b, 22
+  %or = or i32 %shl, %and
+  ret i32 %or
+}
+
+; i16 inputs and mask
+define i16 @insert_6_i16(i16 %a, i16 %b) nounwind {
+; X64-LABEL: insert_6_i16:
+; X64:       # %bb.0:
+; X64-NEXT:    movl %edi, %eax
+; X64-NEXT:    shll $6, %eax
+; X64-NEXT:    shrdw $6, %si, %ax
+; X64-NEXT:    # kill: def $ax killed $ax killed $eax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_6_i16:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT:    andl $1023, %edi # imm = 0x3FF
+; X64-SLOW-NEXT:    shll $10, %esi
+; X64-SLOW-NEXT:    leal (%rsi,%rdi), %eax
+; X64-SLOW-NEXT:    # kill: def $ax killed $ax killed $eax
+; X64-SLOW-NEXT:    retq
+  %and = and i16 %a, 1023
+  %shl = shl i16 %b, 10
+  %or = or i16 %shl, %and
+  ret i16 %or
+}
+
+; Negative test for i8: SHRD does not accept an 8-bit operand
+define i8 @insert_3_i8(i8 %a, i8 %b) nounwind {
+; X64-LABEL: insert_3_i8:
+; X64:       # %bb.0:
+; X64-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-NEXT:    andb $31, %dil
+; X64-NEXT:    shlb $5, %sil
+; X64-NEXT:    leal (%rsi,%rdi), %eax
+; X64-NEXT:    # kill: def $al killed $al killed $eax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_3_i8:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT:    andb $31, %dil
+; X64-SLOW-NEXT:    shlb $5, %sil
+; X64-SLOW-NEXT:    leal (%rsi,%rdi), %eax
+; X64-SLOW-NEXT:    # kill: def $al killed $al killed $eax
+; X64-SLOW-NEXT:    retq
+  %and = and i8 %a, 31
+  %shl = shl i8 %b, 5
+  %or = or i8 %shl, %and
+  ret i8 %or
+}
+
+; Negative test: the mask and the shift amount describe different split points,
+; so the result is not a plain concatenation of A and B.
+define i64 @mask_shift_mismatch(i64 %a, i64 %b) nounwind {
+; X64-LABEL: mask_shift_mismatch:
+; X64:       # %bb.0:
+; X64-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-NEXT:    andq %rdi, %rax
+; X64-NEXT:    shlq $55, %rsi
+; X64-NEXT:    orq %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: mask_shift_mismatch:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $55, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 55
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; Negative test: the mask keeps too many bits, so the shifted-in value would be
+; OR'd on top of bits of A instead of replacing them.
+define i64 @mask_too_wide(i64 %a, i64 %b) nounwind {
+; X64-LABEL: mask_too_wide:
+; X64:       # %bb.0:
+; X64-NEXT:    movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-NEXT:    andq %rdi, %rax
+; X64-NEXT:    shlq $54, %rsi
+; X64-NEXT:    orq %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: mask_too_wide:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 36028797018963967
+  %shl = shl i64 %b, 54
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; Negative test: extra use of the AND means the mask has to be materialized
+; anyway, so folding would only add instructions.
+define i64 @multi_use_and(i64 %a, i64 %b, ptr %p) nounwind {
+; X64-LABEL: multi_use_and:
+; X64:       # %bb.0:
+; X64-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-NEXT:    andq %rdi, %rax
+; X64-NEXT:    movq %rax, (%rdx)
+; X64-NEXT:    shlq $54, %rsi
+; X64-NEXT:    orq %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: multi_use_and:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    movq %rax, (%rdx)
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  store i64 %and, ptr %p
+  %shl = shl i64 %b, 54
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; Negative test: extra use of the SHL.
+define i64 @multi_use_shl(i64 %a, i64 %b, ptr %p) nounwind {
+; X64-LABEL: multi_use_shl:
+; X64:       # %bb.0:
+; X64-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-NEXT:    andq %rdi, %rax
+; X64-NEXT:    shlq $54, %rsi
+; X64-NEXT:    movq %rsi, (%rdx)
+; X64-NEXT:    orq %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: multi_use_shl:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    movq %rsi, (%rdx)
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  store i64 %shl, ptr %p
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; Slow-SHLD targets still fold when optimizing for size.
+define i64 @insert_10_i64_optsize(i64 %a, i64 %b) nounwind optsize {
+; X64-LABEL: insert_10_i64_optsize:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i64_optsize:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+define i64 @insert_10_i64_minsize(i64 %a, i64 %b) nounwind minsize {
+; X64-LABEL: insert_10_i64_minsize:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i64_minsize:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; A loaded destination operand.
+define i64 @insert_10_i64_load(ptr %p, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64_load:
+; X64:       # %bb.0:
+; X64-NEXT:    movq (%rdi), %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+; 
+; X64-SLOW-LABEL: insert_10_i64_load:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq (%rdi), %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+  %a = load i64, ptr %p
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or i64 %shl, %and
+  ret i64 %or
+}

>From 04a111d653895c01a6eb05c082be6ca16637b2ab Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Tue, 1 Sep 2026 12:40:49 +0530
Subject: [PATCH 09/11] [NFC] Remove extra line

---
 llvm/test/CodeGen/X86/insert-bitfield.ll | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/test/CodeGen/X86/insert-bitfield.ll b/llvm/test/CodeGen/X86/insert-bitfield.ll
index c51c0c2a2878f..54e6027c6a105 100644
--- a/llvm/test/CodeGen/X86/insert-bitfield.ll
+++ b/llvm/test/CodeGen/X86/insert-bitfield.ll
@@ -333,7 +333,7 @@ define i64 @insert_10_i64_load(ptr %p, i64 %b) nounwind {
 ; X64-NEXT:    shlq $10, %rax
 ; X64-NEXT:    shrdq $10, %rsi, %rax
 ; X64-NEXT:    retq
-; 
+;
 ; X64-SLOW-LABEL: insert_10_i64_load:
 ; X64-SLOW:       # %bb.0:
 ; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF

>From 8c2aa26aeaecf065d985e77bec436a29af567ab0 Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Tue, 1 Sep 2026 12:43:56 +0530
Subject: [PATCH 10/11] [DAG][X86] Do not combine multi-use AND and SHL They do
 not result in more performant code

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 6 ++++--
 1 file changed, 4 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 5e80c125b9757..e93bafb546b12 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53792,8 +53792,10 @@ static SDValue combineOrOnSHLToSHRD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
 
 	// Check for the following pattern:
 	//   (or (and X, HighBitsMask(C)), (srl Y, C))
-	bool Match = sd_match(N, 
-			m_Or( m_OneUse(m_And( m_Value(X), m_ConstInt(Mask))), 
+	// Do not combine if there are multi-use AND and OR.
+	// It does not result in more performant code.
+	bool Match = sd_match(N,
+			m_Or( m_OneUse(m_And( m_Value(X), m_ConstInt(Mask))),
 					 m_OneUse(m_Shl( m_Value(Y), m_ConstInt(ShiftAmount)))));
 
 	if (!Match)

>From 920683bab4c6cec85dbb6262db0ec271133f6d99 Mon Sep 17 00:00:00 2001
From: Pratyay Pande <pratyaypande at outlook.com>
Date: Tue, 1 Sep 2026 12:58:34 +0530
Subject: [PATCH 11/11] [DAG][X86] clang-format and remove logging

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 77 ++++++++++++-------------
 1 file changed, 37 insertions(+), 40 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 74f79b448ebc4..da1030fc57394 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53859,60 +53859,57 @@ static SDValue combineOrXorWithSETCC(unsigned Opc, const SDLoc &DL, EVT VT,
   return SDValue();
 }
 
- 
 // Fold an OR with a masked destination and a left-shifted
 // source into a shift + double-precision shift (SHRD):
 static SDValue combineOrOnSHLToSHRD(SDNode *N, SDLoc &DL, SelectionDAG &DAG,
-										const X86Subtarget &Subtarget) {
-	using namespace SDPatternMatch;
-	assert(N->getOpcode() == ISD::OR && "Invalid Node. Expected OR.");
-
-	// Bail if SHLD is slow
-	if (Subtarget.isSHLDSlow())
-		return SDValue();
+                                    const X86Subtarget &Subtarget) {
+  using namespace SDPatternMatch;
+  assert(N->getOpcode() == ISD::OR && "Invalid Node. Expected OR.");
 
-	EVT VT = N->getValueType(0);
+  // Bail if SHLD is slow
+  if (Subtarget.isSHLDSlow())
+    return SDValue();
 
-	// SHRD does not suppport 8-bit operands
-	if (VT == MVT::i8)
-		return SDValue();
+  EVT VT = N->getValueType(0);
 
-	APInt Mask;
-	uint64_t ShiftAmount;
-	SDValue X, Y;
+  // SHRD does not suppport 8-bit operands
+  if (VT == MVT::i8)
+    return SDValue();
 
-	// Check for the following pattern:
-	//   (or (and X, HighBitsMask(C)), (srl Y, C))
-	// Do not combine if there are multi-use AND and OR.
-	// It does not result in more performant code.
-	bool Match = sd_match(N,
-			m_Or( m_OneUse(m_And( m_Value(X), m_ConstInt(Mask))),
-					 m_OneUse(m_Shl( m_Value(Y), m_ConstInt(ShiftAmount)))));
+  APInt Mask;
+  uint64_t ShiftAmount;
+  SDValue X, Y;
 
-	if (!Match)
-		return SDValue();
+  // Check for the following pattern:
+  //   (or (and X, HighBitsMask(C)), (srl Y, C))
+  // Do not combine if there are multi-use AND and OR.
+  // It does not result in more performant code.
+  bool Match =
+      sd_match(N, m_Or(m_OneUse(m_And(m_Value(X), m_ConstInt(Mask))),
+                       m_OneUse(m_Shl(m_Value(Y), m_ConstInt(ShiftAmount)))));
 
-	// Max bit-width of operands
-	uint64_t MaxMaskBitWidth = VT.getSizeInBits();
+  if (!Match)
+    return SDValue();
 
-	// Check for Mask and ShiftAmount
-	//
-	// (shl Y, ShiftAmount) fills the top (MaxMaskBitWidth - ShiftAmount) bits,
-	// so X must keep exactly the low ShiftAmount.
-	APInt ExpectedMask = APInt::getLowBitsSet(MaxMaskBitWidth, ShiftAmount);
+  // Max bit-width of operands
+  uint64_t MaxMaskBitWidth = VT.getSizeInBits();
 
-	bool Applicable = (ShiftAmount > 0) && (ShiftAmount < MaxMaskBitWidth)
-										&& (Mask == ExpectedMask);
+  // Check for Mask and ShiftAmount
+  //
+  // (shl Y, ShiftAmount) fills the top (MaxMaskBitWidth - ShiftAmount) bits,
+  // so X must keep exactly the low ShiftAmount.
+  APInt ExpectedMask = APInt::getLowBitsSet(MaxMaskBitWidth, ShiftAmount);
 
-	if (!Applicable)
-		return SDValue();
+  bool Applicable = (ShiftAmount > 0) && (ShiftAmount < MaxMaskBitWidth) &&
+                    (Mask == ExpectedMask);
 
-	LLVM_DEBUG(dbgs() << "The optimization is applicable.\n");
+  if (!Applicable)
+    return SDValue();
 
-	uint64_t InvShAmt = MaxMaskBitWidth - ShiftAmount;
-	SDValue ShAConst = DAG.getShiftAmountConstant(InvShAmt, VT, DL);
-	SDValue SHLVal = DAG.getNode(ISD::SHL, DL, VT, X, ShAConst);
-	return DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShAConst);
+  uint64_t InvShAmt = MaxMaskBitWidth - ShiftAmount;
+  SDValue ShAConst = DAG.getShiftAmountConstant(InvShAmt, VT, DL);
+  SDValue SHLVal = DAG.getNode(ISD::SHL, DL, VT, X, ShAConst);
+  return DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShAConst);
 }
 
 static SDValue combineOr(SDNode *N, SelectionDAG &DAG,



More information about the llvm-commits mailing list