[llvm] [AMDGPU] Enable scalar subword loads for plain 16-bit loads (PR #225097)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 03:27:57 PDT 2026
github-actions[bot] wrote:
<!--LLVM CODE FORMAT COMMENT: {clang-format}-->
:warning: C/C++ code formatter, clang-format found issues in your code. :warning:
<details>
<summary>
You can test this locally with the following command:
</summary>
``````````bash
git-clang-format --diff origin/main HEAD --extensions h,cpp -- llvm/include/llvm/CodeGen/TargetLowering.h llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp llvm/lib/Target/AMDGPU/SIISelLowering.cpp llvm/lib/Target/AMDGPU/SIISelLowering.h --diff_from_common_commit
``````````
:warning:
The reproduction instructions above might return results for more than one PR
in a stack if you are using a stacked PR workflow. You can limit the results by
changing `origin/main` to the base branch/commit you want to compare against.
:warning:
</details>
<details>
<summary>
View the diff from clang-format here.
</summary>
``````````diff
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index e30b8baf0..e222bf66d 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -17972,7 +17972,8 @@ SDValue DAGCombiner::visitTRUNCATE(SDNode *N) {
// fold (truncate (load x)) -> (smaller load x)
// fold (truncate (srl (load x), c)) -> (smaller load (x+c/evtbits))
- if (!LegalTypes || TLI.isTypeDesirableForOp(N0.getOpcode(), VT, N0.getNode())) {
+ if (!LegalTypes ||
+ TLI.isTypeDesirableForOp(N0.getOpcode(), VT, N0.getNode())) {
if (SDValue Reduced = reduceLoadWidth(N))
return Reduced;
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index d76d08a62..ce0acb755 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -2511,8 +2511,8 @@ bool SITargetLowering::isTypeDesirableForOp(unsigned Op, EVT VT,
// Uniform 16-bit loads are legalized to i16 = trunc (zextload i16->i32)
// to match subword load patterns. Allowing conversion back to a 16-bit
// load would create an infinite loop.
- if (Subtarget->hasScalarSubwordLoads() && Op == ISD::LOAD &&
- !VT.isVector() && VT.getSizeInBits() == 16) {
+ if (Subtarget->hasScalarSubwordLoads() && Op == ISD::LOAD && !VT.isVector() &&
+ VT.getSizeInBits() == 16) {
auto *Load = dyn_cast<LoadSDNode>(N);
if (Load && Load->getValueType(0) == MVT::i32 && !Load->isDivergent() &&
AMDGPU::isUniformMMO(Load->getMemOperand())) {
@@ -13635,8 +13635,9 @@ SDValue SITargetLowering::LowerLOAD(SDValue Op, SelectionDAG &DAG) const {
SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, NewLD);
// For f16/bf16, bitcast from i16 to the original fp type
- SDValue Result = (MemVT == MVT::i16) ? Trunc :
- DAG.getNode(ISD::BITCAST, DL, MemVT, Trunc);
+ SDValue Result = (MemVT == MVT::i16)
+ ? Trunc
+ : DAG.getNode(ISD::BITCAST, DL, MemVT, Trunc);
SDValue Ops[] = {Result, NewLD.getValue(1)};
return DAG.getMergeValues(Ops, DL);
``````````
</details>
https://github.com/llvm/llvm-project/pull/225097
More information about the llvm-commits
mailing list