[llvm] [Hexagon] Handle truncating COPY from DoubleRegs to IntRegs (PR #181360)

Brian Cain via llvm-commits llvm-commits at lists.llvm.org
Fri Feb 13 05:40:35 PST 2026


https://github.com/androm3da created https://github.com/llvm/llvm-project/pull/181360

ISel can generate truncating COPYs from DoubleRegs to IntRegs when a 64-bit C2_mask result is used in a context that only needs 32 bits (e.g., V6_vinsertwr). Three passes crashed on this pattern:

BitTracker asserted WD >= WS for COPY instructions. Fix by handling the WD < WS case: extract the low WD bits from the source.

HexagonTfrCleanup asserted that source and destination register sizes match. Fix by returning false (skipping the optimization) when sizes don't match.

HexagonInstrInfo::copyPhysReg had no case for IntRegs <- DoubleRegs. Fix by extracting the isub_lo sub-register and using A2_tfr.

>From 7fc0466628bb9573088d2a39aa60469936d0cc9b Mon Sep 17 00:00:00 2001
From: Brian Cain <brian.cain at oss.qualcomm.com>
Date: Thu, 12 Feb 2026 22:09:42 -0800
Subject: [PATCH] [Hexagon] Handle truncating COPY from DoubleRegs to IntRegs

ISel can generate truncating COPYs from DoubleRegs to IntRegs when a
64-bit C2_mask result is used in a context that only needs 32 bits
(e.g., V6_vinsertwr). Three passes crashed on this pattern:

BitTracker asserted WD >= WS for COPY instructions. Fix by handling
the WD < WS case: extract the low WD bits from the source.

HexagonTfrCleanup asserted that source and destination register
sizes match. Fix by returning false (skipping the optimization)
when sizes don't match.

HexagonInstrInfo::copyPhysReg had no case for IntRegs <- DoubleRegs.
Fix by extracting the isub_lo sub-register and using A2_tfr.
---
 llvm/lib/Target/Hexagon/BitTracker.cpp        | 14 +++++++++---
 llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp  |  7 ++++++
 llvm/lib/Target/Hexagon/HexagonTfrCleanup.cpp |  3 ++-
 .../Hexagon/truncating-copy-double-to-int.ll  | 22 +++++++++++++++++++
 4 files changed, 42 insertions(+), 4 deletions(-)
 create mode 100644 llvm/test/CodeGen/Hexagon/truncating-copy-double-to-int.ll

diff --git a/llvm/lib/Target/Hexagon/BitTracker.cpp b/llvm/lib/Target/Hexagon/BitTracker.cpp
index 80eedabb0d038..193eadb8fb4d4 100644
--- a/llvm/lib/Target/Hexagon/BitTracker.cpp
+++ b/llvm/lib/Target/Hexagon/BitTracker.cpp
@@ -736,16 +736,24 @@ bool BT::MachineEvaluator::evaluate(const MachineInstr &MI,
     case TargetOpcode::COPY: {
       // COPY can transfer a smaller register into a wider one.
       // If that is the case, fill the remaining high bits with 0.
+      // COPY can also transfer a wider register into a narrower one,
+      // in which case the high bits are simply truncated.
       RegisterRef RD = MI.getOperand(0);
       RegisterRef RS = MI.getOperand(1);
       assert(RD.Sub == 0);
       uint16_t WD = getRegBitWidth(RD);
       uint16_t WS = getRegBitWidth(RS);
-      assert(WD >= WS);
       RegisterCell Src = getCell(RS, Inputs);
       RegisterCell Res(WD);
-      Res.insert(Src, BitMask(0, WS-1));
-      Res.fill(WS, WD, BitValue::Zero);
+      if (WD <= WS) {
+        // Truncating copy: extract low WD bits from source.
+        RegisterCell Trunc = Src.extract(BitMask(0, WD - 1));
+        Res.insert(Trunc, BitMask(0, WD - 1));
+      } else {
+        // Widening copy: insert all source bits, zero-fill high bits.
+        Res.insert(Src, BitMask(0, WS - 1));
+        Res.fill(WS, WD, BitValue::Zero);
+      }
       putCell(RD, Res, Outputs);
       break;
     }
diff --git a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
index 9f511efdfb461..212629141ccc6 100644
--- a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
@@ -882,6 +882,13 @@ void HexagonInstrInfo::copyPhysReg(MachineBasicBlock &MBB,
       .addReg(SrcReg).addReg(SrcReg, KillFlag);
     return;
   }
+  if (Hexagon::IntRegsRegClass.contains(DestReg) &&
+      Hexagon::DoubleRegsRegClass.contains(SrcReg)) {
+    // Truncating copy: extract the low 32-bit sub-register.
+    Register SrcLo = HRI.getSubReg(SrcReg, Hexagon::isub_lo);
+    BuildMI(MBB, I, DL, get(Hexagon::A2_tfr), DestReg).addReg(SrcLo, KillFlag);
+    return;
+  }
   if (Hexagon::CtrRegsRegClass.contains(DestReg) &&
       Hexagon::IntRegsRegClass.contains(SrcReg)) {
     BuildMI(MBB, I, DL, get(Hexagon::A2_tfrrcr), DestReg)
diff --git a/llvm/lib/Target/Hexagon/HexagonTfrCleanup.cpp b/llvm/lib/Target/Hexagon/HexagonTfrCleanup.cpp
index 5a85f348fdaf7..b97671cf13529 100644
--- a/llvm/lib/Target/Hexagon/HexagonTfrCleanup.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonTfrCleanup.cpp
@@ -186,7 +186,8 @@ bool HexagonTfrCleanup::rewriteIfImm(MachineInstr *MI, ImmediateMap &IMap,
   bool Tmp, Is32;
   if (!isIntReg(DstR, Is32) || !isIntReg(SrcR, Tmp))
     return false;
-  assert(Tmp == Is32 && "Register size mismatch");
+  if (Tmp != Is32)
+    return false;
   uint64_t Val;
   bool Found = getReg(SrcR, Val, IMap);
   if (!Found)
diff --git a/llvm/test/CodeGen/Hexagon/truncating-copy-double-to-int.ll b/llvm/test/CodeGen/Hexagon/truncating-copy-double-to-int.ll
new file mode 100644
index 0000000000000..762163890a6b3
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/truncating-copy-double-to-int.ll
@@ -0,0 +1,22 @@
+; RUN: llc -mtriple=hexagon -mcpu=hexagonv73 -mattr=+hvxv73,+hvx-length128b \
+; RUN:   < %s | FileCheck %s
+;
+; Check that a truncating copy from DoubleRegs to IntRegs (generated by
+; C2_mask + truncation) does not crash in BitTracker, HexagonTfrCleanup,
+; or ExpandPostRAPseudo.
+
+target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048"
+target triple = "hexagon-unknown-linux-musl"
+
+; CHECK-LABEL: endgame:
+; CHECK:       mask(p0)
+; CHECK:       popcount
+; CHECK:       jumpr r31
+define i16 @endgame(<8 x i32> %0) {
+entry:
+  %1 = icmp eq <8 x i32> %0, zeroinitializer
+  %rdx.op150 = shufflevector <8 x i1> zeroinitializer, <8 x i1> %1, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+  %2 = bitcast <16 x i1> %rdx.op150 to i16
+  %3 = call i16 @llvm.ctpop.i16(i16 %2)
+  ret i16 %3
+}



More information about the llvm-commits mailing list