[llvm] [AMDGPU] Emit the Y component's source location for VOPD instructions (PR #228397)
Daniel Hernandez-Juarez via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 04:20:47 PDT 2026
https://github.com/dhernandez0 updated https://github.com/llvm/llvm-project/pull/228397
>From 8601f4a612ce84bc2bd8fedeaa470b1d388cd3e4 Mon Sep 17 00:00:00 2001
From: Daniel Hernandez <danherna at amd.com>
Date: Fri, 2 Oct 2026 11:10:54 +0000
Subject: [PATCH] [AMDGPU] Emit the Y component's source location for VOPD
instructions
GCNCreateVOPD built the fused instruction with only X's debug location, so
the Y component was attributed to X's source line, or to no line if X had
none.
Keep X's location on the instruction, or take Y's if X has no line. If both
have different locations, record Y's in SIMachineFunctionInfo and emit it
from a new AMDGPUDwarfDebug as an extra line-table row right before the
instruction's own row. Both source lines are then attributed to the
instruction's address, and X's row comes last, so address lookups still
resolve to X. A merged location (see "When to merge instruction locations"
in HowToUpdateDebugInfo.md) would be line 0 and lose both.
The scalar move that materializes a pair's shared literal now takes the
pair's line, without its source atom, instead of no location.
Limitations:
- With Key Instructions, Y's row is not marked is_stmt.
- The recorded locations are not serialized to MIR.
Assisted-by: Cursor (Claude)
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 5 +
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h | 2 +
llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.cpp | 56 +++++
llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.h | 38 ++++
llvm/lib/Target/AMDGPU/CMakeLists.txt | 1 +
llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp | 40 +++-
.../Target/AMDGPU/SIMachineFunctionInfo.cpp | 5 +-
.../lib/Target/AMDGPU/SIMachineFunctionInfo.h | 17 ++
llvm/test/CodeGen/AMDGPU/vopd-debug-loc.ll | 200 ++++++++++++++++++
9 files changed, 358 insertions(+), 6 deletions(-)
create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.cpp
create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.h
create mode 100644 llvm/test/CodeGen/AMDGPU/vopd-debug-loc.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 06acc7613d66083..c548ddac45ca78b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -17,6 +17,7 @@
#include "AMDGPUAsmPrinter.h"
#include "AMDGPU.h"
+#include "AMDGPUDwarfDebug.h"
#include "AMDGPUHSAMetadataStreamer.h"
#include "AMDGPUMCResourceInfo.h"
#include "AMDGPUResourceUsageAnalysis.h"
@@ -1961,6 +1962,10 @@ void AMDGPUAsmPrinter::getAnalysisUsage(AnalysisUsage &AU) const {
AsmPrinter::getAnalysisUsage(AU);
}
+DwarfDebug *AMDGPUAsmPrinter::createDwarfDebug() {
+ return new AMDGPUDwarfDebug(this);
+}
+
void AMDGPUAsmPrinter::emitResourceUsageRemarks(
const MachineFunction &MF, const SIProgramInfo &CurrentProgramInfo,
bool isModuleEntryFunction, bool hasMAIInsts) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 4394cde308665dc..7cb7d366a4658be 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -159,6 +159,8 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
protected:
void getAnalysisUsage(AnalysisUsage &AU) const override;
+ DwarfDebug *createDwarfDebug() override;
+
std::vector<std::string> DisasmLines, HexLines;
size_t DisasmLineMaxLen;
bool IsTargetStreamerInitialized;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.cpp b/llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.cpp
new file mode 100644
index 000000000000000..84a862657172c32
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.cpp
@@ -0,0 +1,56 @@
+//===-- AMDGPUDwarfDebug.cpp - AMDGPU DwarfDebug Implementation -----------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPUDwarfDebug.h"
+#include "SIMachineFunctionInfo.h"
+#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/CodeGen/AsmPrinter.h"
+#include "llvm/CodeGen/MachineFunction.h"
+#include "llvm/CodeGen/MachineInstr.h"
+#include "llvm/IR/DebugInfoMetadata.h"
+#include "llvm/IR/Function.h"
+#include "llvm/MC/MCDwarf.h"
+
+using namespace llvm;
+
+void AMDGPUDwarfDebug::beginInstruction(const MachineInstr *MI) {
+ recordFusedSourceLine(*MI);
+ DwarfDebug::beginInstruction(MI);
+}
+
+void AMDGPUDwarfDebug::recordFusedSourceLine(const MachineInstr &MI) {
+ if (!Asm->hasDebugInfo() || MI.getFlag(MachineInstr::FrameSetup))
+ return;
+
+ const MachineFunction &MF = *MI.getMF();
+ const DISubprogram *SP = MF.getFunction().getSubprogram();
+ if (!SP || SP->getUnit()->getEmissionKind() == DICompileUnit::NoDebug)
+ return;
+
+ DebugLoc DL = MF.getInfo<SIMachineFunctionInfo>()->getFusedDebugLoc(MI);
+ if (!DL || !AMDGPU::isVOPD(MI.getOpcode()))
+ return;
+
+ // Same rule as the non-Key-Instructions case of
+ // DwarfDebug::beginInstruction: a new line is a new statement.
+ // FIXME: With Key Instructions, is_stmt is decided per instruction, and the
+ // instruction fused into MI was erased before the key instructions were
+ // computed. If it was the key instruction of its atom, that atom gets no
+ // is_stmt.
+ unsigned Flags = 0;
+ if (!DL->getScope()->getSubprogram()->getKeyInstructionsEnabled() &&
+ (!PrevInstLoc || PrevInstLoc.getLine() != DL.getLine()))
+ Flags |= DWARF2_FLAG_IS_STMT;
+
+ recordTargetSourceLine(DL, Flags);
+
+ // DwarfDebug::beginInstruction compares against this, so the instruction's
+ // own location is emitted after this row even if it matches the location of
+ // the previous instruction.
+ PrevInstLoc = DL;
+}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.h b/llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.h
new file mode 100644
index 000000000000000..e6f701c7d528df0
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUDwarfDebug.h
@@ -0,0 +1,38 @@
+//===-- AMDGPUDwarfDebug.h - AMDGPU DwarfDebug Implementation ---*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// \file
+/// AMDGPU-specific subclass of DwarfDebug. It emits the source locations of
+/// instructions that were fused into another instruction (e.g. the Y component
+/// of a VOPD) to the line table.
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUDWARFDEBUG_H
+#define LLVM_LIB_TARGET_AMDGPU_AMDGPUDWARFDEBUG_H
+
+#include "../../CodeGen/AsmPrinter/DwarfDebug.h"
+
+namespace llvm {
+
+class AMDGPUDwarfDebug : public DwarfDebug {
+public:
+ explicit AMDGPUDwarfDebug(AsmPrinter *A) : DwarfDebug(A) {}
+
+ void beginInstruction(const MachineInstr *MI) override;
+
+private:
+ /// Emit a line-table row for the location of the instruction that was fused
+ /// into \p MI, if there is one. It precedes the row of \p MI's own location,
+ /// so the latter stays the location of the instruction's address.
+ void recordFusedSourceLine(const MachineInstr &MI);
+};
+
+} // end namespace llvm
+
+#endif // LLVM_LIB_TARGET_AMDGPU_AMDGPUDWARFDEBUG_H
diff --git a/llvm/lib/Target/AMDGPU/CMakeLists.txt b/llvm/lib/Target/AMDGPU/CMakeLists.txt
index 4a5f77d55afa5fb..ab1bfc8ffb9a731 100644
--- a/llvm/lib/Target/AMDGPU/CMakeLists.txt
+++ b/llvm/lib/Target/AMDGPU/CMakeLists.txt
@@ -54,6 +54,7 @@ add_llvm_target(AMDGPUCodeGen
AMDGPUCodeGenPrepare.cpp
AMDGPUCombinerHelper.cpp
AMDGPUCtorDtorLowering.cpp
+ AMDGPUDwarfDebug.cpp
AMDGPUExportClustering.cpp
AMDGPUExportKernelRuntimeHandles.cpp
AMDGPUFrameLowering.cpp
diff --git a/llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp b/llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp
index 2ca4fa50f7f2919..f47effcc33feab9 100644
--- a/llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp
@@ -28,6 +28,7 @@
#include "GCNSubtarget.h"
#include "GCNVOPDUtils.h"
#include "SIInstrInfo.h"
+#include "SIMachineFunctionInfo.h"
#include "Utils/AMDGPUBaseInfo.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/SmallBitVector.h"
@@ -104,6 +105,17 @@ static Register takeFreeSGPR(const GCNSubtarget &ST,
return Register();
}
+static bool hasLine(const DebugLoc &DL) { return DL && DL.getLine() != 0; }
+
+/// Return the location of the VOPD instruction fusing \p MIX and \p MIY. It can
+/// carry only one: X's, unless X has no line and Y has one.
+static const DebugLoc &getVOPDDebugLoc(const MachineInstr &MIX,
+ const MachineInstr &MIY) {
+ const DebugLoc &XLoc = MIX.getDebugLoc();
+ const DebugLoc &YLoc = MIY.getDebugLoc();
+ return hasLine(XLoc) || !hasLine(YLoc) ? XLoc : YLoc;
+}
+
namespace {
class GCNCreateVOPD {
@@ -208,9 +220,16 @@ class GCNCreateVOPD {
return Fixup.Imm == Imm;
}));
+ // The move takes the pair's line but not its source atom, which stays with
+ // the VOPD instruction.
+ DebugLoc DL =
+ getVOPDDebugLoc(*Candidate.Match.getMIX(), *Candidate.Match.getMIY());
+ if (DL)
+ DL = DebugLoc(DL->getWithoutAtom());
+
MachineInstr *InsertPt = Candidate.Match.InOrder[0];
- BuildMI(*InsertPt->getParent(), InsertPt, DebugLoc(),
- TII.get(AMDGPU::S_MOV_B32), Candidate.MaterializationReg)
+ BuildMI(*InsertPt->getParent(), InsertPt, DL, TII.get(AMDGPU::S_MOV_B32),
+ Candidate.MaterializationReg)
.addImm(Fixups.front().Imm);
++NumLiteralsMaterialized;
@@ -237,9 +256,9 @@ class GCNCreateVOPD {
assert(NewOpcode != -1 &&
"Should have previously determined this as a possible VOPD\n");
- auto VOPDInst =
- BuildMI(*MIX->getParent(), MIX, MIX->getDebugLoc(), SII->get(NewOpcode))
- .setMIFlags(MIX->getFlags() | MIY->getFlags());
+ auto VOPDInst = BuildMI(*MIX->getParent(), MIX, getVOPDDebugLoc(*MIX, *MIY),
+ SII->get(NewOpcode))
+ .setMIFlags(MIX->getFlags() | MIY->getFlags());
namespace VOPD = AMDGPU::VOPD;
MachineInstr *MI[] = {MIX, MIY};
@@ -287,6 +306,17 @@ class GCNCreateVOPD {
for (auto CompIdx : VOPD::COMPONENTS)
VOPDInst.copyImplicitOps(*MI[CompIdx]);
+ // If X and Y have different locations, Y's is kept as well and emitted to
+ // the line table before the instruction's own. A merged location (see
+ // "When to merge instruction locations" in HowToUpdateDebugInfo.md) would
+ // be line 0 and lose both.
+ const DebugLoc &XLoc = MIX->getDebugLoc();
+ const DebugLoc &YLoc = MIY->getDebugLoc();
+ if (hasLine(XLoc) && hasLine(YLoc) && !YLoc.isSameSourceLocation(XLoc)) {
+ MachineFunction &MF = *MIX->getMF();
+ MF.getInfo<SIMachineFunctionInfo>()->setFusedDebugLoc(*VOPDInst, YLoc);
+ }
+
LLVM_DEBUG(dbgs() << "VOPD Fused: " << *VOPDInst << " from\tX: " << *MIX
<< "\tY: " << *MIY << "\n");
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
index d1df50d26a83275..db942247c2a9b8e 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
@@ -191,7 +191,10 @@ MachineFunctionInfo *SIMachineFunctionInfo::clone(
BumpPtrAllocator &Allocator, MachineFunction &DestMF,
const DenseMap<MachineBasicBlock *, MachineBasicBlock *> &Src2DstMBB)
const {
- return DestMF.cloneInfo<SIMachineFunctionInfo>(*this);
+ auto *MFI = DestMF.cloneInfo<SIMachineFunctionInfo>(*this);
+ // The keys point into the source function.
+ MFI->FusedDebugLocs.clear();
+ return MFI;
}
void SIMachineFunctionInfo::limitOccupancy(const MachineFunction &MF) {
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 0207c728ea9b80d..51f757b836a4b18 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -626,6 +626,15 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
// load/store is enabled.
IndexedMap<uint32_t, VGPRBlock2IndexFunctor> MaskForVGPRBlockOps;
+ // Source locations of instructions fused into another instruction (the Y
+ // component of a VOPD), keyed by the fused instruction. AMDGPUDwarfDebug
+ // emits them to the line table before the fused instruction's own location.
+ // Entries are not removed when an instruction is erased, so a key can be
+ // reused by a new instruction. AMDGPUDwarfDebug only looks up VOPD
+ // instructions, so passes after GCNCreateVOPD must not create VOPD
+ // instructions without updating this map. Not serialized to MIR.
+ DenseMap<const MachineInstr *, DebugLoc> FusedDebugLocs;
+
private:
Register VGPRForAGPRCopy;
@@ -659,6 +668,14 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
return MaskForVGPRBlockOps.inBounds(RegisterBlock);
}
+ void setFusedDebugLoc(const MachineInstr &MI, const DebugLoc &DL) {
+ FusedDebugLocs[&MI] = DL;
+ }
+
+ DebugLoc getFusedDebugLoc(const MachineInstr &MI) const {
+ return FusedDebugLocs.lookup(&MI);
+ }
+
public:
SIMachineFunctionInfo(const SIMachineFunctionInfo &MFI) = default;
SIMachineFunctionInfo(const Function &F, const GCNSubtarget *STI);
diff --git a/llvm/test/CodeGen/AMDGPU/vopd-debug-loc.ll b/llvm/test/CodeGen/AMDGPU/vopd-debug-loc.ll
new file mode 100644
index 000000000000000..ebe70b2f8a6e6c0
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/vopd-debug-loc.ll
@@ -0,0 +1,200 @@
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -verify-machineinstrs < %s | FileCheck %s
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -filetype=obj < %s \
+; RUN: | llvm-dwarfdump --debug-line - | FileCheck %s --check-prefix=LINES
+
+; A VOPD instruction has the location of its X component, or of its Y component
+; if X has none. If both have different locations, Y's is emitted to the line
+; table right before the instruction's own, so both source lines are attributed
+; to the instruction's address and X's row comes last.
+
+; CHECK-LABEL: different_lines:
+; CHECK: .loc 0 20 5
+; CHECK-NEXT: {{^}}.Ltmp{{[0-9]+}}:
+; CHECK-NEXT: .loc 0 10 3 prologue_end
+; CHECK-NEXT: v_dual_mul_f32 {{.*}} :: v_dual_add_f32
+
+; LINES: [[ADDR:0x[0-9a-f]+]] 20 5 0 0 0 0 is_stmt{{$}}
+; LINES-NEXT: [[ADDR]] 10 3 0 0 0 0 is_stmt prologue_end
+define <2 x float> @different_lines(float %a, float %b, float %c, float %d) !dbg !5 {
+ %x = fmul float %a, %b, !dbg !8
+ %y = fadd float %c, %d, !dbg !9
+ %v0 = insertelement <2 x float> poison, float %x, i32 0, !dbg !9
+ %v1 = insertelement <2 x float> %v0, float %y, i32 1, !dbg !9
+ ret <2 x float> %v1, !dbg !9
+}
+
+; Both components have the same location; nothing extra is emitted.
+; CHECK-LABEL: same_line:
+; CHECK: .loc 0 30 0
+; CHECK-NOT: .loc
+; CHECK: .loc 0 31 7 prologue_end
+; CHECK-NEXT: v_dual_mul_f32 {{.*}} :: v_dual_add_f32
+define <2 x float> @same_line(float %a, float %b, float %c, float %d) !dbg !10 {
+ %x = fmul float %a, %b, !dbg !11
+ %y = fadd float %c, %d, !dbg !11
+ %v0 = insertelement <2 x float> poison, float %x, i32 0, !dbg !11
+ %v1 = insertelement <2 x float> %v0, float %y, i32 1, !dbg !11
+ ret <2 x float> %v1, !dbg !11
+}
+
+; X has no location, so the VOPD takes Y's location.
+; CHECK-LABEL: x_without_loc:
+; CHECK: .loc 0 42 9
+; CHECK-NEXT: v_dual_mul_f32 {{.*}} :: v_dual_add_f32
+define <2 x float> @x_without_loc(float %a, float %b, float %c, float %d) !dbg !12 {
+ %x = fmul float %a, %b
+ %y = fadd float %c, %d, !dbg !13
+ %v0 = insertelement <2 x float> poison, float %x, i32 0
+ %v1 = insertelement <2 x float> %v0, float %y, i32 1
+ ret <2 x float> %v1, !dbg !14
+}
+
+; X's location has line 0, which attributes it to no line, so the VOPD takes
+; Y's location.
+; CHECK-LABEL: x_line_zero:
+; CHECK: .loc 0 102 5
+; CHECK-NEXT: v_dual_mul_f32 {{.*}} :: v_dual_add_f32
+define <2 x float> @x_line_zero(float %a, float %b, float %c, float %d) !dbg !70 {
+ %x = fmul float %a, %b, !dbg !71
+ %y = fadd float %c, %d, !dbg !72
+ %v0 = insertelement <2 x float> poison, float %x, i32 0, !dbg !73
+ %v1 = insertelement <2 x float> %v0, float %y, i32 1, !dbg !73
+ ret <2 x float> %v1, !dbg !73
+}
+
+; X is on the same line as the previous instruction. Its location is emitted
+; again after Y's, and the next instruction on X's line gets no new row.
+; CHECK-LABEL: prev_same_as_x:
+; CHECK: .loc 0 51 3 prologue_end
+; CHECK-NEXT: v_mul_f32_e32
+; CHECK-NOT: .loc
+; CHECK: .loc 0 52 5
+; CHECK-NEXT: .loc 0 51 3
+; CHECK-NEXT: v_dual_mul_f32 {{.*}} :: v_dual_add_f32
+; CHECK-NEXT: v_mul_f32_e32
+; CHECK-NEXT: .loc 0 53 1
+
+; LINES: [[ADDR:0x[0-9a-f]+]] 52 5 0 0 0 0 is_stmt{{$}}
+; LINES-NEXT: [[ADDR]] 51 3 0 0 0 0 is_stmt{{$}}
+define float @prev_same_as_x(float %a, float %b, float %c, float %d) !dbg !20 {
+ %t = fmul float %a, %b, !dbg !21
+ %x = fmul float %t, %c, !dbg !21
+ %y = fadd float %t, %d, !dbg !22
+ %z = fmul float %x, %y, !dbg !21
+ ret float %z, !dbg !23
+}
+
+; Y is on the same line as the previous instruction. Its location is emitted
+; again, so that both lines precede the VOPD, but not as a new statement.
+; CHECK-LABEL: prev_same_as_y:
+; CHECK: .loc 0 82 5 prologue_end
+; CHECK-NEXT: v_add_f32_e32
+; CHECK-NOT: .loc
+; CHECK: .loc 0 82 5 is_stmt 0
+; CHECK-NEXT: .loc 0 81 3 is_stmt 1
+; CHECK-NEXT: v_dual_mul_f32 {{.*}} :: v_dual_add_f32
+define float @prev_same_as_y(float %a, float %b, float %c, float %d) !dbg !50 {
+ %t = fadd float %a, %b, !dbg !52
+ %x = fmul float %t, %c, !dbg !51
+ %y = fadd float %t, %d, !dbg !52
+ %z = fmul float %x, %y, !dbg !53
+ ret float %z, !dbg !53
+}
+
+; X has no location and the VOPD starts a block, where an instruction without a
+; location would get a line-0 row. The VOPD takes Y's location instead.
+; CHECK-LABEL: x_without_loc_block_start:
+; CHECK: %then
+; CHECK-NEXT: .loc 0 62 5
+; CHECK-NEXT: v_dual_mul_f32 {{.*}} :: v_dual_add_f32
+define void @x_without_loc_block_start(ptr addrspace(1) %p, float %a, float %b, float %c, float %d, i32 inreg %n) !dbg !30 {
+entry:
+ %cond = icmp eq i32 %n, 0, !dbg !31
+ br i1 %cond, label %then, label %exit, !dbg !31
+
+then:
+ %x = fmul float %a, %b
+ %y = fadd float %c, %d, !dbg !32
+ %v0 = insertelement <2 x float> poison, float %x, i32 0, !dbg !33
+ %v1 = insertelement <2 x float> %v0, float %y, i32 1, !dbg !33
+ store <2 x float> %v1, ptr addrspace(1) %p, align 8, !dbg !33
+ br label %exit, !dbg !33
+
+exit:
+ ret void, !dbg !34
+}
+
+; The scalar move materializing the pair's shared literal takes the pair's
+; location.
+; CHECK-LABEL: literal_move:
+; CHECK: .loc 0 91 3 prologue_end
+; CHECK-NEXT: s_mov_b32 [[SREG:s[0-9]+]], 0xffff0000
+; CHECK-NOT: .loc
+; CHECK: .loc 0 92 5
+; CHECK-NEXT: .loc 0 91 3
+; CHECK-NEXT: v_dual_lshlrev_b32 {{.*}} :: v_dual_bitop2_b32 {{.*}}, [[SREG]],
+define <2 x i32> @literal_move(i32 %a, i32 %b) !dbg !60 {
+ %x = shl i32 %a, 16, !dbg !61
+ %y = and i32 %b, -65536, !dbg !62
+ %v0 = insertelement <2 x i32> poison, i32 %x, i32 0, !dbg !63
+ %v1 = insertelement <2 x i32> %v0, i32 %y, i32 1, !dbg !63
+ ret <2 x i32> %v1, !dbg !63
+}
+
+; With Key Instructions, Y's row is not a statement even though Y was the key
+; instruction of its atom.
+; FIXME: Y's row should have is_stmt.
+; CHECK-LABEL: key_instructions:
+; CHECK: .loc 0 72 5 is_stmt 0
+; CHECK-NEXT: {{^}}.Ltmp{{[0-9]+}}:
+; CHECK-NEXT: .loc 0 71 3 prologue_end is_stmt 1
+; CHECK-NEXT: v_dual_mul_f32 {{.*}} :: v_dual_add_f32
+define <2 x float> @key_instructions(float %a, float %b, float %c, float %d) !dbg !40 {
+ %x = fmul float %a, %b, !dbg !41
+ %y = fadd float %c, %d, !dbg !42
+ %v0 = insertelement <2 x float> poison, float %x, i32 0, !dbg !43
+ %v1 = insertelement <2 x float> %v0, float %y, i32 1, !dbg !43
+ ret <2 x float> %v1, !dbg !43
+}
+
+!llvm.dbg.cu = !{!0}
+!llvm.module.flags = !{!2, !3}
+
+!0 = distinct !DICompileUnit(language: DW_LANG_C99, file: !1, producer: "clang", isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly)
+!1 = !DIFile(filename: "t.c", directory: "/")
+!2 = !{i32 2, !"Debug Info Version", i32 3}
+!3 = !{i32 7, !"Dwarf Version", i32 5}
+!4 = !DISubroutineType(types: !{})
+!5 = distinct !DISubprogram(name: "different_lines", scope: !1, file: !1, line: 1, type: !4, scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!8 = !DILocation(line: 10, column: 3, scope: !5)
+!9 = !DILocation(line: 20, column: 5, scope: !5)
+!10 = distinct !DISubprogram(name: "same_line", scope: !1, file: !1, line: 30, type: !4, scopeLine: 30, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!11 = !DILocation(line: 31, column: 7, scope: !10)
+!12 = distinct !DISubprogram(name: "x_without_loc", scope: !1, file: !1, line: 40, type: !4, scopeLine: 40, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!13 = !DILocation(line: 42, column: 9, scope: !12)
+!14 = !DILocation(line: 43, column: 1, scope: !12)
+!20 = distinct !DISubprogram(name: "prev_same_as_x", scope: !1, file: !1, line: 50, type: !4, scopeLine: 50, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!21 = !DILocation(line: 51, column: 3, scope: !20)
+!22 = !DILocation(line: 52, column: 5, scope: !20)
+!23 = !DILocation(line: 53, column: 1, scope: !20)
+!30 = distinct !DISubprogram(name: "x_without_loc_block_start", scope: !1, file: !1, line: 60, type: !4, scopeLine: 60, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!31 = !DILocation(line: 61, column: 3, scope: !30)
+!32 = !DILocation(line: 62, column: 5, scope: !30)
+!33 = !DILocation(line: 63, column: 7, scope: !30)
+!34 = !DILocation(line: 64, column: 1, scope: !30)
+!40 = distinct !DISubprogram(name: "key_instructions", scope: !1, file: !1, line: 70, type: !4, scopeLine: 70, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0, keyInstructions: true)
+!41 = !DILocation(line: 71, column: 3, scope: !40, atomGroup: 1, atomRank: 1)
+!42 = !DILocation(line: 72, column: 5, scope: !40, atomGroup: 2, atomRank: 1)
+!43 = !DILocation(line: 73, column: 1, scope: !40, atomGroup: 3, atomRank: 1)
+!50 = distinct !DISubprogram(name: "prev_same_as_y", scope: !1, file: !1, line: 80, type: !4, scopeLine: 80, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!51 = !DILocation(line: 81, column: 3, scope: !50)
+!52 = !DILocation(line: 82, column: 5, scope: !50)
+!53 = !DILocation(line: 83, column: 1, scope: !50)
+!60 = distinct !DISubprogram(name: "literal_move", scope: !1, file: !1, line: 90, type: !4, scopeLine: 90, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!61 = !DILocation(line: 91, column: 3, scope: !60)
+!62 = !DILocation(line: 92, column: 5, scope: !60)
+!63 = !DILocation(line: 93, column: 1, scope: !60)
+!70 = distinct !DISubprogram(name: "x_line_zero", scope: !1, file: !1, line: 100, type: !4, scopeLine: 100, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!71 = !DILocation(line: 0, scope: !70)
+!72 = !DILocation(line: 102, column: 5, scope: !70)
+!73 = !DILocation(line: 103, column: 1, scope: !70)
More information about the llvm-commits
mailing list