[llvm] [X86] Ignore debug instructions when picking block entry vertex in LVI hardening (PR #225994)
Demetrios Chiuratto Agourakis via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 18:54:46 PDT 2026
https://github.com/agourakis82 created https://github.com/llvm/llvm-project/pull/225994
Fixes #224484.
X86LoadValueInjectionLoadHardeningImpl::getGadgetGraph() previously used
MBB->begin() as the block representative vertex in the gadget graph. When
a block begins with debug instructions (such as DBG_VALUE or DBG_INSTR_REF),
the debug instruction became an ordinary graph vertex, altering the ingress
and egress edge costs and moving LFENCE placement between ordinary instructions.
Use MBB->getFirstNonDebugInstr(/*SkipPseudoOp=*/false) instead so that leading
debug records are skipped, preserving pseudo probes as graph vertices while
ensuring debug information does not affect fence placement.
>From 00b855d91167b3298dc9f13af5219a301fbe2ab4 Mon Sep 17 00:00:00 2001
From: Demetrios Chiuratto Agourakis <agourakis82 at gmail.com>
Date: Wed, 23 Sep 2026 18:41:49 +0000
Subject: [PATCH 1/2] [SelectionDAG] Fix result index and vector width in
unrollExpandedOp
Fixes #224127.
### Summary
In `DAGTypeLegalizer::WidenVectorResult`, `unrollExpandedOp` is a helper lambda
that unrolls vector operations whose wide vector forms would expand to scalar
libcalls anyway (e.g. `ISD::FFREXP`, `ISD::FSINCOS`).
When widening a multi-result node where the results have different element types
(such as `ISD::FFREXP`, where result 0 is floating-point like `<2 x double>` and
result 1 is integer like `<2 x i32>`):
1. `unrollExpandedOp` previously computed `WideVecVT` using `N->getValueType(0)`
instead of `N->getValueType(ResNo)`. When `ResNo == 1`, `N->getValueType(0)`
(`<2 x double>`) may already be a legal 128-bit vector with 2 elements on
SSE2, so `WideVecVT.getVectorNumElements()` was 2 instead of the widened count
4 for `<2 x i32>`.
2. `Res` was set to `DAG.UnrollVectorOp(...)`, which for multiple results returns
a `MERGE_VALUES` node. Assigning `Res = Unrolled` meant `Res` had value index 0
(`SDValue(MergeNode, 0)`), which is result 0 (`<2 x double>`), rather than
the requested `ResNo` (`SDValue(MergeNode, ResNo)`).
3. Consequently, `SetWidenedVector(SDValue(N, ResNo), Res)` asserted:
`Result.getValueType() == TLI.getTypeToTransformTo(Op.getValueType())`
due to mismatched types (`<2 x double>` vs `<4 x i32>`).
This patch fixes `unrollExpandedOp` to:
- Use `N->getValueType(ResNo)` to determine `WideVecVT` and the unroll count.
- Select `SDValue(Unrolled.getNode(), ResNo)` for `Res`.
- Pass `Unrolled.getNode()` to `ReplaceOtherWidenResults`.
### Test plan
- Uncommented and updated `llvm.frexp.v2f64.v2i32` tests in `llvm/test/CodeGen/X86/llvm.frexp.ll`.
---
.../SelectionDAG/LegalizeVectorTypes.cpp | 12 +-
llvm/test/CodeGen/X86/llvm.frexp.ll | 144 ++++++++++++++++--
2 files changed, 136 insertions(+), 20 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 56ad7c65b48d1..7b521cfff9134 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -5234,13 +5234,15 @@ void DAGTypeLegalizer::WidenVectorResult(SDNode *N, unsigned ResNo) {
// elements. If the wide vector op is eventually going to be expanded to
// scalar libcalls, then unroll into scalar ops now to avoid unnecessary
// libcalls on the undef elements.
- EVT VT = N->getValueType(0);
- EVT WideVecVT = TLI.getTypeToTransformTo(*DAG.getContext(), VT);
+ EVT ResVT = N->getValueType(ResNo);
+ EVT WideVecVT = TLI.getTypeToTransformTo(*DAG.getContext(), ResVT);
+ EVT VT0 = N->getValueType(0);
if (!TLI.isOperationLegalOrCustomOrPromote(N->getOpcode(), WideVecVT) &&
- TLI.isOperationExpandOrLibCall(N->getOpcode(), VT.getScalarType())) {
- Res = DAG.UnrollVectorOp(N, WideVecVT.getVectorNumElements());
+ TLI.isOperationExpandOrLibCall(N->getOpcode(), VT0.getScalarType())) {
+ SDValue Unrolled = DAG.UnrollVectorOp(N, WideVecVT.getVectorNumElements());
+ Res = SDValue(Unrolled.getNode(), ResNo);
if (N->getNumValues() > 1)
- ReplaceOtherWidenResults(N, Res.getNode(), ResNo);
+ ReplaceOtherWidenResults(N, Unrolled.getNode(), ResNo);
return true;
}
return false;
diff --git a/llvm/test/CodeGen/X86/llvm.frexp.ll b/llvm/test/CodeGen/X86/llvm.frexp.ll
index fb778add339fb..079c1504ca50f 100644
--- a/llvm/test/CodeGen/X86/llvm.frexp.ll
+++ b/llvm/test/CodeGen/X86/llvm.frexp.ll
@@ -588,22 +588,136 @@ define { float, i32 } @pr160981() {
ret { float, i32 } %ret
}
-; FIXME: Widen vector result
-; define { <2 x double>, <2 x i32> } @test_frexp_v2f64_v2i32(<2 x double> %a) nounwind {
-; %result = call { <2 x double>, <2 x i32> } @llvm.frexp.v2f64.v2i32(<2 x double> %a)
-; ret { <2 x double>, <2 x i32> } %result
-; }
+define { <2 x double>, <2 x i32> } @test_frexp_v2f64_v2i32(<2 x double> %a) nounwind {
+; X64-LABEL: test_frexp_v2f64_v2i32:
+; X64: # %bb.0:
+; X64-NEXT: subq $56, %rsp
+; X64-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; X64-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
+; X64-NEXT: callq frexp at PLT
+; X64-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; X64-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload
+; X64-NEXT: movhlps {{.*#+}} xmm0 = xmm0[1,1]
+; X64-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
+; X64-NEXT: callq frexp at PLT
+; X64-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm2 # 16-byte Reload
+; X64-NEXT: movlhps {{.*#+}} xmm2 = xmm2[0],xmm0[0]
+; X64-NEXT: movss {{.*#+}} xmm1 = mem[0],zero,zero,zero
+; X64-NEXT: movss {{.*#+}} xmm0 = mem[0],zero,zero,zero
+; X64-NEXT: unpcklps {{.*#+}} xmm1 = xmm1[0],xmm0[0],xmm1[1],xmm0[1]
+; X64-NEXT: movaps %xmm2, %xmm0
+; X64-NEXT: addq $56, %rsp
+; X64-NEXT: retq
+;
+; WIN32-LABEL: test_frexp_v2f64_v2i32:
+; WIN32: # %bb.0:
+; WIN32-NEXT: subl $36, %esp
+; WIN32-NEXT: fldl {{[0-9]+}}(%esp)
+; WIN32-NEXT: fstpl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Spill
+; WIN32-NEXT: fldl {{[0-9]+}}(%esp)
+; WIN32-NEXT: leal {{[0-9]+}}(%esp), %eax
+; WIN32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; WIN32-NEXT: fstpl (%esp)
+; WIN32-NEXT: calll _frexp
+; WIN32-NEXT: fstpl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Spill
+; WIN32-NEXT: leal {{[0-9]+}}(%esp), %eax
+; WIN32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; WIN32-NEXT: fldl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Reload
+; WIN32-NEXT: fstpl (%esp)
+; WIN32-NEXT: calll _frexp
+; WIN32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; WIN32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; WIN32-NEXT: fldl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Reload
+; WIN32-NEXT: addl $36, %esp
+; WIN32-NEXT: retl
+ %result = call { <2 x double>, <2 x i32> } @llvm.frexp.v2f64.v2i32(<2 x double> %a)
+ ret { <2 x double>, <2 x i32> } %result
+}
-; define <2 x double> @test_frexp_v2f64_v2i32_only_use_fract(<2 x double> %a) nounwind {
-; %result = call { <2 x double>, <2 x i32> } @llvm.frexp.v2f64.v2i32(<2 x double> %a)
-; %result.0 = extractvalue { <2 x double>, <2 x i32> } %result, 0
-; ret <2 x double> %result.0
-; }
+define <2 x double> @test_frexp_v2f64_v2i32_only_use_fract(<2 x double> %a) nounwind {
+; X64-LABEL: test_frexp_v2f64_v2i32_only_use_fract:
+; X64: # %bb.0:
+; X64-NEXT: subq $56, %rsp
+; X64-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; X64-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
+; X64-NEXT: callq frexp at PLT
+; X64-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; X64-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload
+; X64-NEXT: movhlps {{.*#+}} xmm0 = xmm0[1,1]
+; X64-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
+; X64-NEXT: callq frexp at PLT
+; X64-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm1 # 16-byte Reload
+; X64-NEXT: movlhps {{.*#+}} xmm1 = xmm1[0],xmm0[0]
+; X64-NEXT: movaps %xmm1, %xmm0
+; X64-NEXT: addq $56, %rsp
+; X64-NEXT: retq
+;
+; WIN32-LABEL: test_frexp_v2f64_v2i32_only_use_fract:
+; WIN32: # %bb.0:
+; WIN32-NEXT: subl $36, %esp
+; WIN32-NEXT: fldl {{[0-9]+}}(%esp)
+; WIN32-NEXT: fstpl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Spill
+; WIN32-NEXT: fldl {{[0-9]+}}(%esp)
+; WIN32-NEXT: leal {{[0-9]+}}(%esp), %eax
+; WIN32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; WIN32-NEXT: fstpl (%esp)
+; WIN32-NEXT: calll _frexp
+; WIN32-NEXT: fstpl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Spill
+; WIN32-NEXT: leal {{[0-9]+}}(%esp), %eax
+; WIN32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; WIN32-NEXT: fldl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Reload
+; WIN32-NEXT: fstpl (%esp)
+; WIN32-NEXT: calll _frexp
+; WIN32-NEXT: fldl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Reload
+; WIN32-NEXT: fxch %st(1)
+; WIN32-NEXT: addl $36, %esp
+; WIN32-NEXT: retl
+ %result = call { <2 x double>, <2 x i32> } @llvm.frexp.v2f64.v2i32(<2 x double> %a)
+ %result.0 = extractvalue { <2 x double>, <2 x i32> } %result, 0
+ ret <2 x double> %result.0
+}
-; define <2 x i32> @test_frexp_v2f64_v2i32_only_use_exp(<2 x double> %a) nounwind {
-; %result = call { <2 x double>, <2 x i32> } @llvm.frexp.v2f64.v2i32(<2 x double> %a)
-; %result.1 = extractvalue { <2 x double>, <2 x i32> } %result, 1
-; ret <2 x i32> %result.1
-; }
+define <2 x i32> @test_frexp_v2f64_v2i32_only_use_exp(<2 x double> %a) nounwind {
+; X64-LABEL: test_frexp_v2f64_v2i32_only_use_exp:
+; X64: # %bb.0:
+; X64-NEXT: subq $40, %rsp
+; X64-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill
+; X64-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
+; X64-NEXT: callq frexp at PLT
+; X64-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload
+; X64-NEXT: movhlps {{.*#+}} xmm0 = xmm0[1,1]
+; X64-NEXT: leaq {{[0-9]+}}(%rsp), %rdi
+; X64-NEXT: callq frexp at PLT
+; X64-NEXT: movss {{.*#+}} xmm0 = mem[0],zero,zero,zero
+; X64-NEXT: movss {{.*#+}} xmm1 = mem[0],zero,zero,zero
+; X64-NEXT: unpcklps {{.*#+}} xmm0 = xmm0[0],xmm1[0],xmm0[1],xmm1[1]
+; X64-NEXT: addq $40, %rsp
+; X64-NEXT: retq
+;
+; WIN32-LABEL: test_frexp_v2f64_v2i32_only_use_exp:
+; WIN32: # %bb.0:
+; WIN32-NEXT: subl $28, %esp
+; WIN32-NEXT: fldl {{[0-9]+}}(%esp)
+; WIN32-NEXT: fstpl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Spill
+; WIN32-NEXT: fldl {{[0-9]+}}(%esp)
+; WIN32-NEXT: leal {{[0-9]+}}(%esp), %eax
+; WIN32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; WIN32-NEXT: fstpl (%esp)
+; WIN32-NEXT: calll _frexp
+; WIN32-NEXT: fstp %st(0)
+; WIN32-NEXT: leal {{[0-9]+}}(%esp), %eax
+; WIN32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; WIN32-NEXT: fldl {{[-0-9]+}}(%e{{[sb]}}p) # 8-byte Folded Reload
+; WIN32-NEXT: fstpl (%esp)
+; WIN32-NEXT: calll _frexp
+; WIN32-NEXT: fstp %st(0)
+; WIN32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; WIN32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; WIN32-NEXT: addl $28, %esp
+; WIN32-NEXT: retl
+ %result = call { <2 x double>, <2 x i32> } @llvm.frexp.v2f64.v2i32(<2 x double> %a)
+ %result.1 = extractvalue { <2 x double>, <2 x i32> } %result, 1
+ ret <2 x i32> %result.1
+}
attributes #0 = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
>From bcd28e3dd774581508f5095759a588903282166f Mon Sep 17 00:00:00 2001
From: Demetrios Chiuratto Agourakis <agourakis82 at gmail.com>
Date: Thu, 24 Sep 2026 01:53:56 +0000
Subject: [PATCH 2/2] [X86] Ignore debug instructions when picking block entry
vertex in LVI hardening
Fixes #224484.
X86LoadValueInjectionLoadHardeningImpl::getGadgetGraph() previously used
MBB->begin() as the block representative vertex in the gadget graph. When
a block begins with debug instructions (such as DBG_VALUE or DBG_INSTR_REF),
the debug instruction became an ordinary graph vertex, altering the ingress
and egress edge costs and moving LFENCE placement between ordinary instructions.
Use MBB->getFirstNonDebugInstr(/*SkipPseudoOp=*/false) instead so that leading
debug records are skipped, preserving pseudo probes as graph vertices while
ensuring debug information does not affect fence placement.
---
.../X86LoadValueInjectionLoadHardening.cpp | 4 +-
.../X86/lvi-hardening-debug-invariance.ll | 78 +++++++++++++++++++
2 files changed, 81 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/CodeGen/X86/lvi-hardening-debug-invariance.ll
diff --git a/llvm/lib/Target/X86/X86LoadValueInjectionLoadHardening.cpp b/llvm/lib/Target/X86/X86LoadValueInjectionLoadHardening.cpp
index 871081e328724..a4e18a8a6aca8 100644
--- a/llvm/lib/Target/X86/X86LoadValueInjectionLoadHardening.cpp
+++ b/llvm/lib/Target/X86/X86LoadValueInjectionLoadHardening.cpp
@@ -495,7 +495,9 @@ X86LoadValueInjectionLoadHardeningImpl::getGadgetGraph(
unsigned LoopDepth = MLI.getLoopDepth(MBB);
if (!MBB->empty()) {
// Always add the first instruction in each block
- auto NI = MBB->begin();
+ auto NI = MBB->getFirstNonDebugInstr(/*SkipPseudoOp=*/false);
+ if (NI == MBB->end())
+ NI = MBB->begin();
auto BeginBB = MaybeAddNode(&*NI);
Builder.addEdge(ParentDepth, GI, BeginBB.first);
if (!BlocksVisited.insert(MBB).second)
diff --git a/llvm/test/CodeGen/X86/lvi-hardening-debug-invariance.ll b/llvm/test/CodeGen/X86/lvi-hardening-debug-invariance.ll
new file mode 100644
index 0000000000000..7d3b1c902e1a2
--- /dev/null
+++ b/llvm/test/CodeGen/X86/lvi-hardening-debug-invariance.ll
@@ -0,0 +1,78 @@
+; RUN: llc -verify-machineinstrs -mtriple=x86_64-unknown-linux-gnu -x86-lvi-load-no-cbranch < %s | FileCheck %s
+
+; PR224484: A leading debug instruction (such as DBG_VALUE or DBG_INSTR_REF) at
+; the start of a basic block must not become an ordinary vertex in the LVI gadget
+; graph. When getGadgetGraph adds the first instruction in each block, it should
+; skip debug-only instructions so that debug info does not alter fence placement.
+
+define i32 @nested_lvi(ptr %pointer_slot, ptr %control, i32 %limit) #0 !dbg !4 {
+; CHECK-LABEL: nested_lvi:
+; CHECK: # %bb.0:
+; CHECK: lfence
+; CHECK: .LBB0_2: # %inner.header
+; CHECK: movq (%rdi), %r8
+; CHECK-NEXT: lfence
+; CHECK-NOT: .LBB0_{{[0-9]+}}: # %inner.debug.join
+; CHECK-NOT: lfence
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.i = phi i32 [ 0, %entry ], [ %outer.next, %outer.latch ]
+ %sum = phi i32 [ 0, %entry ], [ %sum.next, %outer.latch ]
+ %outer.more = icmp slt i32 %outer.i, %limit
+ br i1 %outer.more, label %inner.header, label %exit
+
+inner.header:
+ %loaded.pointer = load ptr, ptr %pointer_slot, align 8
+ %first.control = load volatile i32, ptr %control, align 4
+ %take.first = icmp eq i32 %first.control, 1
+ br i1 %take.first, label %inner.exit.a, label %inner.check
+
+inner.check:
+ %second.control = load volatile i32, ptr %control, align 4
+ %take.second = icmp eq i32 %second.control, 2
+ br i1 %take.second, label %inner.exit.b, label %inner.latch
+
+inner.latch:
+ call void asm sideeffect "", "~{memory}"()
+ br label %inner.header
+
+inner.exit.a:
+ br label %inner.debug.join
+
+inner.exit.b:
+ br label %inner.debug.join
+
+inner.debug.join:
+ call void @llvm.dbg.value(metadata ptr %loaded.pointer, metadata !6, metadata !DIExpression()), !dbg !7
+ br label %inner.merge
+
+inner.merge:
+ %value = load i32, ptr %loaded.pointer, align 4
+ %sum.next = add i32 %sum, %value
+ br label %outer.latch
+
+outer.latch:
+ %outer.next = add nuw nsw i32 %outer.i, 1
+ br label %outer.header
+
+exit:
+ ret i32 %sum
+}
+
+attributes #0 = { "target-features"="+lvi-load-hardening" }
+
+declare void @llvm.dbg.value(metadata, metadata, metadata)
+
+!llvm.dbg.cu = !{!0}
+!llvm.module.flags = !{!3}
+
+!0 = distinct !DICompileUnit(language: DW_LANG_C99, file: !1, emissionKind: FullDebug)
+!1 = !DIFile(filename: "lvi-debug-empty.c", directory: "/")
+!2 = !{}
+!3 = !{i32 2, !"Debug Info Version", i32 3}
+!4 = distinct !DISubprogram(name: "nested_lvi", scope: !1, file: !1, line: 1, type: !5, unit: !0, retainedNodes: !2)
+!5 = !DISubroutineType(types: !2)
+!6 = !DILocalVariable(name: "ghost", scope: !4, file: !1, line: 1)
+!7 = !DILocation(line: 1, scope: !4)
More information about the llvm-commits
mailing list