[llvm] 9d9c442 - [ARM][FastISel] Fix soft-float f64 register pair order on big endian (#227172)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 03:37:01 PDT 2026
Author: dong jianqiang
Date: 2026-10-01T18:36:52+08:00
New Revision: 9d9c4426fb0e0f0a43b283678acaafd22d6dfe8a
URL: https://github.com/llvm/llvm-project/commit/9d9c4426fb0e0f0a43b283678acaafd22d6dfe8a
DIFF: https://github.com/llvm/llvm-project/commit/9d9c4426fb0e0f0a43b283678acaafd22d6dfe8a.diff
LOG: [ARM][FastISel] Fix soft-float f64 register pair order on big endian (#227172)
Under the soft-float ABI an f64 argument or return value is passed in a
core register pair which is ordered like any other 64-bit value in core
registers: on big-endian targets the register holding the most
significant word comes first.
FastISel materialized the pair with VMOVRRD/VMOVDRR in little-endian
order on both endiannesses, disagreeing with the SelectionDAG ISel paths
and miscompiling soft-float f64 argument passing and returns for armebv7
targets at -O0.
VMOVRRD moves Dn[31:0] to Rt and Dn[63:32] to Rt2, and VMOVDRR reads
them symmetrically, so swap the destination/source pair when the data
layout is big-endian, mirroring the big-endian handling already present
in ARMISelLowering.
Fixes #227171
Added:
llvm/test/CodeGen/ARM/fast-isel-call-be-softfp.ll
Modified:
llvm/lib/Target/ARM/ARMFastISel.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Target/ARM/ARMFastISel.cpp b/llvm/lib/Target/ARM/ARMFastISel.cpp
index 88ee3156ee339..20e08d0df7ec6 100644
--- a/llvm/lib/Target/ARM/ARMFastISel.cpp
+++ b/llvm/lib/Target/ARM/ARMFastISel.cpp
@@ -2057,10 +2057,16 @@ bool ARMFastISel::ProcessCallArgs(SmallVectorImpl<Value*> &Args,
assert(VA.isRegLoc() && NextVA.isRegLoc() &&
"We only handle register args!");
+ // VMOVRRD moves the low word to Rt and the high word to Rt2; on
+ // big-endian targets the high word goes in the first register.
+ Register Lo = VA.getLocReg();
+ Register Hi = NextVA.getLocReg();
+ if (DL.isBigEndian())
+ std::swap(Lo, Hi);
AddOptionalDefs(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
- TII.get(ARM::VMOVRRD), VA.getLocReg())
- .addReg(NextVA.getLocReg(), RegState::Define)
- .addReg(Arg));
+ TII.get(ARM::VMOVRRD), Lo)
+ .addReg(Hi, RegState::Define)
+ .addReg(Arg));
RegArgs.push_back(VA.getLocReg());
RegArgs.push_back(NextVA.getLocReg());
} else {
@@ -2107,10 +2113,16 @@ bool ARMFastISel::FinishCall(MVT RetVT, SmallVectorImpl<Register> &UsedRegs,
MVT DestVT = RVLocs[0].getValVT();
const TargetRegisterClass* DstRC = TLI.getRegClassFor(DestVT);
Register ResultReg = createResultReg(DstRC);
+ // VMOVDRR takes the low word from Rt and the high word from Rt2; on
+ // big-endian targets the first register holds the high word.
+ Register Lo = RVLocs[0].getLocReg();
+ Register Hi = RVLocs[1].getLocReg();
+ if (DL.isBigEndian())
+ std::swap(Lo, Hi);
AddOptionalDefs(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
TII.get(ARM::VMOVDRR), ResultReg)
- .addReg(RVLocs[0].getLocReg())
- .addReg(RVLocs[1].getLocReg()));
+ .addReg(Lo)
+ .addReg(Hi));
UsedRegs.push_back(RVLocs[0].getLocReg());
UsedRegs.push_back(RVLocs[1].getLocReg());
diff --git a/llvm/test/CodeGen/ARM/fast-isel-call-be-softfp.ll b/llvm/test/CodeGen/ARM/fast-isel-call-be-softfp.ll
new file mode 100644
index 0000000000000..2d929d83a9401
--- /dev/null
+++ b/llvm/test/CodeGen/ARM/fast-isel-call-be-softfp.ll
@@ -0,0 +1,77 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -O0 -verify-machineinstrs -relocation-model=dynamic-no-pic -mtriple=armebv7-linux-gnueabi -mattr=+vfp3 -fast-isel=1 | FileCheck %s --check-prefix=FAST
+; RUN: llc < %s -O0 -verify-machineinstrs -relocation-model=dynamic-no-pic -mtriple=armebv7-linux-gnueabi -mattr=+vfp3 -fast-isel=0 | FileCheck %s --check-prefix=SDAG
+
+; Both ISel paths must pass soft-float f64 call arguments and results in core
+; register pairs ordered like other 64-bit values: the most significant word
+; in the first register on big-endian targets. The FAST output additionally
+; sets up a frame pointer for these frameless functions; the pair order is
+; the same in both outputs.
+
+define void @f64_arg() {
+; FAST-LABEL: f64_arg:
+; FAST: @ %bb.0: @ %entry
+; FAST-NEXT: .save {r11, lr}
+; FAST-NEXT: push {r11, lr}
+; FAST-NEXT: .setfp r11, sp
+; FAST-NEXT: mov r11, sp
+; FAST-NEXT: vldr d16, .LCPI0_0
+; FAST-NEXT: vmov r1, r0, d16
+; FAST-NEXT: bl void_callee
+; FAST-NEXT: pop {r11, pc}
+; FAST-NEXT: .p2align 3
+; FAST-NEXT: @ %bb.1:
+; FAST-NEXT: .LCPI0_0:
+; FAST-NEXT: .long 1070176665 @ double 0.20000000298023224
+; FAST-NEXT: .long 2684354560
+;
+; SDAG-LABEL: f64_arg:
+; SDAG: @ %bb.0: @ %entry
+; SDAG-NEXT: .save {r11, lr}
+; SDAG-NEXT: push {r11, lr}
+; SDAG-NEXT: vldr d16, .LCPI0_0
+; SDAG-NEXT: vmov r1, r0, d16
+; SDAG-NEXT: bl void_callee
+; SDAG-NEXT: pop {r11, pc}
+; SDAG-NEXT: .p2align 3
+; SDAG-NEXT: @ %bb.1:
+; SDAG-NEXT: .LCPI0_0:
+; SDAG-NEXT: .long 1070176665 @ double 0.20000000298023224
+; SDAG-NEXT: .long 2684354560
+entry:
+ call void @void_callee(double 0x3FC99999A0000000)
+ ret void
+}
+
+define double @f64_ret() {
+; FAST-LABEL: f64_ret:
+; FAST: @ %bb.0: @ %entry
+; FAST-NEXT: .save {r11, lr}
+; FAST-NEXT: push {r11, lr}
+; FAST-NEXT: .setfp r11, sp
+; FAST-NEXT: mov r11, sp
+; FAST-NEXT: bl f64_callee
+; FAST-NEXT: vmov d16, r1, r0
+; FAST-NEXT: vmov.f64 d17, #1.000000e+00
+; FAST-NEXT: vadd.f64 d16, d16, d17
+; FAST-NEXT: vmov r1, r0, d16
+; FAST-NEXT: pop {r11, pc}
+;
+; SDAG-LABEL: f64_ret:
+; SDAG: @ %bb.0: @ %entry
+; SDAG-NEXT: .save {r11, lr}
+; SDAG-NEXT: push {r11, lr}
+; SDAG-NEXT: bl f64_callee
+; SDAG-NEXT: vmov d16, r1, r0
+; SDAG-NEXT: vmov.f64 d17, #1.000000e+00
+; SDAG-NEXT: vadd.f64 d16, d16, d17
+; SDAG-NEXT: vmov r1, r0, d16
+; SDAG-NEXT: pop {r11, pc}
+entry:
+ %r = call double @f64_callee()
+ %s = fadd double %r, 1.000000e+00
+ ret double %s
+}
+
+declare void @void_callee(double)
+declare double @f64_callee()
More information about the llvm-commits
mailing list