[llvm] [SystemZTTI][CostModel] Improve SystemZ cost model for scalar Read-Modify-Write Sequence, Fix #189183 (PR #190350)

via llvm-commits llvm-commits at lists.llvm.org
Thu Apr 9 17:19:39 PDT 2026


https://github.com/anoopkg6 updated https://github.com/llvm/llvm-project/pull/190350

>From fd7d3787ffad17eef957a46a0cebefa8988026ec Mon Sep 17 00:00:00 2001
From: anoopkg6 <anoopkg6 at github.com>
Date: Fri, 3 Apr 2026 16:45:30 +0200
Subject: [PATCH 1/3] [SystemZTTI][CostModel] Improve SystemZ cost model for
 scalar Read-Modify-Write Sequence, Fixes #189183.

 Refine the SystemZ cost model to identify Read-Modify-Write sequences
 that can be folded into a single instruction (e.g., ASI, NI, OI).
 If a load, a scalar arithmetic operation (ADD, SUB, AND, OR, XOR) with an
 immediate, and a store all target the same memory location and have no
 external uses, the cost of the arithmetic and store insn should bw 0.
 This implementation does not include TTI::TCK_RecipThroughput CostKind, as
 it causes regression in non-power-2-subvector-extract.ll.
---
 .../SystemZ/SystemZTargetTransformInfo.cpp    |  70 ++++++++-
 .../CostModel/SystemZ/fold-rmw-cost.ll        | 147 ++++++++++++++++++
 2 files changed, 215 insertions(+), 2 deletions(-)
 create mode 100644 llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll

diff --git a/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp b/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
index 4322773f4afd6..b453a6bb25d41 100644
--- a/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
@@ -539,15 +539,44 @@ static unsigned getNumVectorRegs(Type *Ty) {
   return ((WideBits % 128U) ? ((WideBits / 128U) + 1) : (WideBits / 128U));
 }
 
+static bool isArithmeticFoldableRMW(const Instruction *I) {
+  if (!I || !isa<BinaryOperator>(I) || !I->hasOneUse())
+    return false;
+
+  Value *Op0 = I->getOperand(0), *Op1 = I->getOperand(1);
+  if (!isa<ConstantInt>(Op0) && !isa<ConstantInt>(Op1))
+    return false;
+
+  Value *V = nullptr;
+  unsigned Opcode = I->getOpcode();
+  if (Opcode == Instruction::And || Opcode == Instruction::Or ||
+      Opcode == Instruction::Xor || Opcode == Instruction::Add)
+    V = isa<ConstantInt>(Op0) ? Op1 : (isa<ConstantInt>(Op1) ? Op0 : nullptr);
+  else if (Opcode == Instruction::Sub && isa<ConstantInt>(Op1))
+    V = Op0;
+  if (!V)
+    return false;
+
+  auto *LI = dyn_cast_or_null<LoadInst>(V);
+  auto *SI = dyn_cast<StoreInst>(*I->users().begin());
+
+  return LI && SI && !LI->isVolatile() && !SI->isVolatile() &&
+         LI->hasOneUse() && LI->getPointerOperand() == SI->getPointerOperand();
+}
+
 InstructionCost SystemZTTIImpl::getArithmeticInstrCost(
     unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
     TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
     ArrayRef<const Value *> Args, const Instruction *CxtI) const {
 
   // TODO: Handle more cost kinds.
-  if (CostKind != TTI::TCK_RecipThroughput)
+  if (CostKind != TTI::TCK_RecipThroughput) {
+    if (CxtI && Ty && !Ty->isVectorTy() && isArithmeticFoldableRMW(CxtI))
+      return TTI::TCC_Free;
+
     return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info,
                                          Op2Info, Args, CxtI);
+  }
 
   // TODO: return a good value for BB-VECTORIZER that includes the
   // immediate loads, which we do not want to count for the loop
@@ -1298,6 +1327,35 @@ static bool isBswapIntrinsicCall(const Value *V) {
   return false;
 }
 
+static bool isStoreFoldableRMW(const Instruction *I) {
+  auto *SI = dyn_cast_or_null<StoreInst>(I);
+  if (!SI || SI->isVolatile())
+    return false;
+
+  auto *BI = dyn_cast<BinaryOperator>(SI->getValueOperand());
+  if (!BI || !BI->hasOneUse())
+    return false;
+
+  Value *Op0 = BI->getOperand(0), *Op1 = BI->getOperand(1);
+  if (!isa<ConstantInt>(Op0) && !isa<ConstantInt>(Op1))
+    return false;
+
+  unsigned Opcode = BI->getOpcode();
+  Value *V = nullptr;
+  if (Opcode == Instruction::And || Opcode == Instruction::Or ||
+      Opcode == Instruction::Xor || Opcode == Instruction::Add)
+    V = isa<ConstantInt>(Op0) ? Op1 : (isa<ConstantInt>(Op1) ? Op0 : nullptr);
+  else if (Opcode == Instruction::Sub && isa<ConstantInt>(Op1))
+    V = Op0;
+
+  if (!V)
+    return false;
+
+  auto *LI = dyn_cast_or_null<LoadInst>(V);
+  return LI && !LI->isVolatile() &&
+         LI->getPointerOperand() == SI->getPointerOperand();
+}
+
 InstructionCost SystemZTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src,
                                                 Align Alignment,
                                                 unsigned AddressSpace,
@@ -1307,8 +1365,16 @@ InstructionCost SystemZTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src,
   assert(!Src->isVoidTy() && "Invalid type");
 
   // TODO: Handle other cost kinds.
-  if (CostKind != TTI::TCK_RecipThroughput)
+  if (CostKind != TTI::TCK_RecipThroughput) {
+    // TCK_RecipThroughput causes regressions in the SLP Vectorizer test
+    // non-power-2-subvector-extract.ll. It seems profit analysis does not
+    // favor vectorization by making scalar cost cheaper.
+    if (I && Opcode == Instruction::Store && !Src->isVectorTy()) {
+      if (isStoreFoldableRMW(I))
+        return TTI::TCC_Free;
+    }
     return 1;
+  }
 
   if (!Src->isVectorTy() && Opcode == Instruction::Load && I != nullptr) {
     // Store the load or its truncated or extended value in FoldedValue.
diff --git a/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll b/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll
new file mode 100644
index 0000000000000..12199505f2fab
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll
@@ -0,0 +1,147 @@
+; RUN: opt -S -mtriple=s390x-unknown-linux -mcpu=z17 -passes='print<cost-model>' \
+; RUN:  -cost-kind=code-size -disable-output %s 2>&1 | FileCheck %s
+;
+; Test the SystemZ cost model for scalar Read-Modify-Write (RMW) folding.
+; A load, a scalar arithmetic operation (ADD, SUB, AND, OR, XOR) with
+; an immediate, and a store all target the same memory location and have
+; no external uses, the cost of the arithmetic and store insn should be 0.
+
+define void @test_and(ptr %p) {
+; CHECK-LABEL: 'test_and'
+; CHECK: cost of 0 {{.*}} and i32
+; CHECK: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = and i32 %v, 1
+  store i32 %res, ptr %p
+  ret void
+}
+
+define void @test_or(ptr %p) {
+; CHECK-LABEL: 'test_or'
+; CHECK: cost of 0 {{.*}} or i32
+; CHECK: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = or i32 %v, 1
+  store i32 %res, ptr %p
+  ret void
+}
+
+define void @test_xor(ptr %p) {
+; CHECK-LABEL: 'test_xor'
+; CHECK: cost of 0 {{.*}} xor i32
+; CHECK: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = xor i32 %v, 1
+  store i32 %res, ptr %p
+  ret void
+}
+
+define void @test_add(ptr %p) {
+; CHECK-LABEL: 'test_add'
+; CHECK: cost of 0 {{.*}} add i32
+; CHECK: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = add i32 %v, 1
+  store i32 %res, ptr %p
+  ret void
+}
+
+define void @test_sub(ptr %p) {
+; CHECK-LABEL: 'test_sub'
+; CHECK: cost of 0 {{.*}} sub i32
+; CHECK: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = sub i32 %v, 1
+  store i32 %res, ptr %p
+  ret void
+}
+
+; Subtraction is not commutative.
+define void @test_sub_neg_imm_lhs(ptr %p) {
+; CHECK-LABEL: 'test_sub_neg_imm_lhs'
+; CHECK-NOT: cost of 0 {{.*}} sub i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = sub i32 1, %v
+  store i32 %res, ptr %p
+  ret void
+}
+
+; Different Addresses.
+define void @test_diff_addr(ptr %p, ptr %q) {
+; CHECK-LABEL: 'test_diff_addr'
+; CHECK-NOT: cost of 0 {{.*}} add i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = add i32 %v, 1
+  store i32 %res, ptr %q
+  ret void
+}
+
+; Multi-use of Arithmetic Result.
+define i32 @test_multi_use_arith(ptr %p) {
+; CHECK-LABEL: 'test_multi_use_arith'
+; CHECK-NOT: cost of 0 {{.*}} add i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = add i32 %v, 1
+  store i32 %res, ptr %p
+  ret i32 %res
+}
+
+; Multi-use of Load Result.
+define i32 @test_multi_use_load(ptr %p) {
+; CHECK-LABEL: 'test_multi_use_load'
+; CHECK-NOT: cost of 0 {{.*}} add i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = add i32 %v, 1
+  store i32 %res, ptr %p
+  %use2 = add i32 %v, %res
+  ret i32 %use2
+}
+
+; Neither Operand is Immediate.
+define void @test_no_immediate(ptr %p, i32 %reg) {
+; CHECK-LABEL: 'test_no_immediate'
+; CHECK-NOT: cost of 0 {{.*}} add i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = add i32 %v, %reg
+  store i32 %res, ptr %p
+  ret void
+}
+
+; Both Operands are Immediate.
+define void @test_both_immediate(ptr %p) {
+; CHECK-LABEL: 'test_both_immediate'
+; CHECK-NOT: cost of 0 {{.*}} add i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = add i32 5, 10
+  store i32 %res, ptr %p
+  ret void
+}
+
+; Volatile Load/Store.
+define void @test_volatile(ptr %p) {
+; CHECK-LABEL: 'test_volatile'
+; CHECK-NOT: cost of 0 {{.*}} add i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load volatile i32, ptr %p
+  %res = add i32 %v, 1
+  store volatile i32 %res, ptr %p
+  ret void
+}
+
+; Vector types cost should not be changed.
+define void @test_vector_no_fold(<4 x i32> %val, ptr %p) {
+; CHECK-LABEL: 'test_vector_no_fold'
+; CHECK-NOT: cost of 0 {{.*}} add <4 x i32>
+; CHECK-NOT: cost of 0 {{.*}} store <4 x i32>
+  %v = load <4 x i32>, ptr %p
+  %res = add <4 x i32> %v, %val
+  store <4 x i32> %res, ptr %p
+  ret void
+}
+

>From 294608a71eae12819353f8119bf4c00e5873a40a Mon Sep 17 00:00:00 2001
From: anoopkg6 <anoopkg6 at github.com>
Date: Thu, 9 Apr 2026 04:50:41 +0200
Subject: [PATCH 2/3] [SystemZTTI][CostModel] Improve SystemZ cost model for
 scalar Read-Modify-Write  Sequence only for immediate-to-memory arithmatic
 operation for i8 types  (Only for CostKind of TTI::TCK_RecipThroughput).

---
 .../SystemZ/SystemZTargetTransformInfo.cpp    |  27 ++--
 .../CostModel/SystemZ/fold-rmw-cost.ll        | 146 +++++++++---------
 2 files changed, 86 insertions(+), 87 deletions(-)

diff --git a/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp b/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
index b453a6bb25d41..edc2dc90eb9c0 100644
--- a/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
@@ -570,14 +570,15 @@ InstructionCost SystemZTTIImpl::getArithmeticInstrCost(
     ArrayRef<const Value *> Args, const Instruction *CxtI) const {
 
   // TODO: Handle more cost kinds.
-  if (CostKind != TTI::TCK_RecipThroughput) {
-    if (CxtI && Ty && !Ty->isVectorTy() && isArithmeticFoldableRMW(CxtI))
-      return TTI::TCC_Free;
-
+  if (CostKind != TTI::TCK_RecipThroughput)
     return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info,
                                          Op2Info, Args, CxtI);
-  }
-
+  // Supporting only i8 type immediate-to-memory arithmatic operation until
+  // backend start folding it for i32 type if constant only has bits in one
+  // single byte set.
+  if (CxtI && Ty && !Ty->isVectorTy() && Ty->isIntegerTy(8) &&
+      isArithmeticFoldableRMW(CxtI))
+    return TTI::TCC_Free;
   // TODO: return a good value for BB-VECTORIZER that includes the
   // immediate loads, which we do not want to count for the loop
   // vectorizer, since they are hopefully hoisted out of the loop. This
@@ -1365,15 +1366,13 @@ InstructionCost SystemZTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src,
   assert(!Src->isVoidTy() && "Invalid type");
 
   // TODO: Handle other cost kinds.
-  if (CostKind != TTI::TCK_RecipThroughput) {
-    // TCK_RecipThroughput causes regressions in the SLP Vectorizer test
-    // non-power-2-subvector-extract.ll. It seems profit analysis does not
-    // favor vectorization by making scalar cost cheaper.
-    if (I && Opcode == Instruction::Store && !Src->isVectorTy()) {
-      if (isStoreFoldableRMW(I))
-        return TTI::TCC_Free;
-    }
+  if (CostKind != TTI::TCK_RecipThroughput)
     return 1;
+
+  if (I && Opcode == Instruction::Store && !Src->isVectorTy() &&
+      Src->isIntegerTy(8)) {
+    if (isStoreFoldableRMW(I))
+      return TTI::TCC_Free;
   }
 
   if (!Src->isVectorTy() && Opcode == Instruction::Load && I != nullptr) {
diff --git a/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll b/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll
index 12199505f2fab..1ba1b31171669 100644
--- a/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll
+++ b/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll
@@ -1,5 +1,5 @@
 ; RUN: opt -S -mtriple=s390x-unknown-linux -mcpu=z17 -passes='print<cost-model>' \
-; RUN:  -cost-kind=code-size -disable-output %s 2>&1 | FileCheck %s
+; RUN:  -disable-output %s 2>&1 | FileCheck %s
 ;
 ; Test the SystemZ cost model for scalar Read-Modify-Write (RMW) folding.
 ; A load, a scalar arithmetic operation (ADD, SUB, AND, OR, XOR) with
@@ -8,140 +8,140 @@
 
 define void @test_and(ptr %p) {
 ; CHECK-LABEL: 'test_and'
-; CHECK: cost of 0 {{.*}} and i32
-; CHECK: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = and i32 %v, 1
-  store i32 %res, ptr %p
+; CHECK: cost of 0 {{.*}} and i8
+; CHECK: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = and i8 %v, 1
+  store i8 %res, ptr %p
   ret void
 }
 
 define void @test_or(ptr %p) {
 ; CHECK-LABEL: 'test_or'
-; CHECK: cost of 0 {{.*}} or i32
-; CHECK: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = or i32 %v, 1
-  store i32 %res, ptr %p
+; CHECK: cost of 0 {{.*}} or i8
+; CHECK: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = or i8 %v, 1
+  store i8 %res, ptr %p
   ret void
 }
 
 define void @test_xor(ptr %p) {
 ; CHECK-LABEL: 'test_xor'
-; CHECK: cost of 0 {{.*}} xor i32
-; CHECK: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = xor i32 %v, 1
-  store i32 %res, ptr %p
+; CHECK: cost of 0 {{.*}} xor i8
+; CHECK: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = xor i8 %v, 1
+  store i8 %res, ptr %p
   ret void
 }
 
 define void @test_add(ptr %p) {
 ; CHECK-LABEL: 'test_add'
-; CHECK: cost of 0 {{.*}} add i32
-; CHECK: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = add i32 %v, 1
-  store i32 %res, ptr %p
+; CHECK: cost of 0 {{.*}} add i8
+; CHECK: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = add i8 %v, 1
+  store i8 %res, ptr %p
   ret void
 }
 
 define void @test_sub(ptr %p) {
 ; CHECK-LABEL: 'test_sub'
-; CHECK: cost of 0 {{.*}} sub i32
-; CHECK: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = sub i32 %v, 1
-  store i32 %res, ptr %p
+; CHECK: cost of 0 {{.*}} sub i8
+; CHECK: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = sub i8 %v, 1
+  store i8 %res, ptr %p
   ret void
 }
 
 ; Subtraction is not commutative.
 define void @test_sub_neg_imm_lhs(ptr %p) {
 ; CHECK-LABEL: 'test_sub_neg_imm_lhs'
-; CHECK-NOT: cost of 0 {{.*}} sub i32
-; CHECK-NOT: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = sub i32 1, %v
-  store i32 %res, ptr %p
+; CHECK-NOT: cost of 0 {{.*}} sub i8
+; CHECK-NOT: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = sub i8 1, %v
+  store i8 %res, ptr %p
   ret void
 }
 
 ; Different Addresses.
 define void @test_diff_addr(ptr %p, ptr %q) {
 ; CHECK-LABEL: 'test_diff_addr'
-; CHECK-NOT: cost of 0 {{.*}} add i32
-; CHECK-NOT: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = add i32 %v, 1
-  store i32 %res, ptr %q
+; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = add i8 %v, 1
+  store i8 %res, ptr %q
   ret void
 }
 
 ; Multi-use of Arithmetic Result.
-define i32 @test_multi_use_arith(ptr %p) {
+define i8 @test_multi_use_arith(ptr %p) {
 ; CHECK-LABEL: 'test_multi_use_arith'
-; CHECK-NOT: cost of 0 {{.*}} add i32
-; CHECK-NOT: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = add i32 %v, 1
-  store i32 %res, ptr %p
-  ret i32 %res
+; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = add i8 %v, 1
+  store i8 %res, ptr %p
+  ret i8 %res
 }
 
 ; Multi-use of Load Result.
-define i32 @test_multi_use_load(ptr %p) {
+define i8 @test_multi_use_load(ptr %p) {
 ; CHECK-LABEL: 'test_multi_use_load'
-; CHECK-NOT: cost of 0 {{.*}} add i32
-; CHECK-NOT: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = add i32 %v, 1
-  store i32 %res, ptr %p
-  %use2 = add i32 %v, %res
-  ret i32 %use2
+; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = add i8 %v, 1
+  store i8 %res, ptr %p
+  %use2 = add i8 %v, %res
+  ret i8 %use2
 }
 
 ; Neither Operand is Immediate.
-define void @test_no_immediate(ptr %p, i32 %reg) {
+define void @test_no_immediate(ptr %p, i8 %reg) {
 ; CHECK-LABEL: 'test_no_immediate'
-; CHECK-NOT: cost of 0 {{.*}} add i32
-; CHECK-NOT: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = add i32 %v, %reg
-  store i32 %res, ptr %p
+; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = add i8 %v, %reg
+  store i8 %res, ptr %p
   ret void
 }
 
 ; Both Operands are Immediate.
 define void @test_both_immediate(ptr %p) {
 ; CHECK-LABEL: 'test_both_immediate'
-; CHECK-NOT: cost of 0 {{.*}} add i32
-; CHECK-NOT: cost of 0 {{.*}} store i32
-  %v = load i32, ptr %p
-  %res = add i32 5, 10
-  store i32 %res, ptr %p
+; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} store i8
+  %v = load i8, ptr %p
+  %res = add i8 5, 10
+  store i8 %res, ptr %p
   ret void
 }
 
 ; Volatile Load/Store.
 define void @test_volatile(ptr %p) {
 ; CHECK-LABEL: 'test_volatile'
-; CHECK-NOT: cost of 0 {{.*}} add i32
-; CHECK-NOT: cost of 0 {{.*}} store i32
-  %v = load volatile i32, ptr %p
-  %res = add i32 %v, 1
-  store volatile i32 %res, ptr %p
+; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} store i8
+  %v = load volatile i8, ptr %p
+  %res = add i8 %v, 1
+  store volatile i8 %res, ptr %p
   ret void
 }
 
 ; Vector types cost should not be changed.
-define void @test_vector_no_fold(<4 x i32> %val, ptr %p) {
+define void @test_vector_no_fold(<4 x i8> %val, ptr %p) {
 ; CHECK-LABEL: 'test_vector_no_fold'
-; CHECK-NOT: cost of 0 {{.*}} add <4 x i32>
-; CHECK-NOT: cost of 0 {{.*}} store <4 x i32>
-  %v = load <4 x i32>, ptr %p
-  %res = add <4 x i32> %v, %val
-  store <4 x i32> %res, ptr %p
+; CHECK-NOT: cost of 0 {{.*}} add <4 x i8>
+; CHECK-NOT: cost of 0 {{.*}} store <4 x i8>
+  %v = load <4 x i8>, ptr %p
+  %res = add <4 x i8> %v, %val
+  store <4 x i8> %res, ptr %p
   ret void
 }
 

>From 20c325e2a1fd4892be037aaf26e53154528c537e Mon Sep 17 00:00:00 2001
From: anoopkg6 <anoopkg6 at github.com>
Date: Fri, 10 Apr 2026 01:02:23 +0200
Subject: [PATCH 3/3] [SystemZTTI][CostModel] Merged isArithmeticFoldableRMW
 and isStoreFoldableRMW  into one helper isFoldableRMW and type-check i8 for
 bitwise RMW and i32/i64  for arithmatic RMW inside it.

---
 .../SystemZ/SystemZTargetTransformInfo.cpp    | 80 +++++++------------
 .../CostModel/SystemZ/fold-rmw-cost.ll        | 80 +++++++++----------
 2 files changed, 71 insertions(+), 89 deletions(-)

diff --git a/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp b/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
index edc2dc90eb9c0..c3e3be0068a81 100644
--- a/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
+++ b/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp
@@ -539,26 +539,42 @@ static unsigned getNumVectorRegs(Type *Ty) {
   return ((WideBits % 128U) ? ((WideBits / 128U) + 1) : (WideBits / 128U));
 }
 
-static bool isArithmeticFoldableRMW(const Instruction *I) {
-  if (!I || !isa<BinaryOperator>(I) || !I->hasOneUse())
+static bool isFoldableRMW(const Instruction *I, Type *Ty) {
+  auto *BI = dyn_cast_or_null<BinaryOperator>(I);
+  if (!BI || !BI->hasOneUse())
+    return false;
+
+  unsigned Opcode = BI->getOpcode();
+  unsigned BitWidth = Ty->getScalarSizeInBits();
+
+  switch (Opcode) {
+  case Instruction::And:
+  case Instruction::Or:
+  case Instruction::Xor:
+    if (BitWidth != 8)
+      return false;
+    break;
+  case Instruction::Add:
+  case Instruction::Sub:
+    if (BitWidth != 32 && BitWidth != 64)
+      return false;
+    break;
+  default:
     return false;
+  }
 
-  Value *Op0 = I->getOperand(0), *Op1 = I->getOperand(1);
+  Value *Op0 = BI->getOperand(0), *Op1 = BI->getOperand(1);
   if (!isa<ConstantInt>(Op0) && !isa<ConstantInt>(Op1))
     return false;
 
-  Value *V = nullptr;
-  unsigned Opcode = I->getOpcode();
-  if (Opcode == Instruction::And || Opcode == Instruction::Or ||
-      Opcode == Instruction::Xor || Opcode == Instruction::Add)
-    V = isa<ConstantInt>(Op0) ? Op1 : (isa<ConstantInt>(Op1) ? Op0 : nullptr);
-  else if (Opcode == Instruction::Sub && isa<ConstantInt>(Op1))
-    V = Op0;
-  if (!V)
+  Value *V =
+      (Opcode == Instruction::Sub) ? Op0 : (isa<ConstantInt>(Op0) ? Op1 : Op0);
+  if (Opcode == Instruction::Sub && !isa<ConstantInt>(Op1))
     return false;
 
   auto *LI = dyn_cast_or_null<LoadInst>(V);
-  auto *SI = dyn_cast<StoreInst>(*I->users().begin());
+  // Already checked BI hasOneUse.
+  auto *SI = dyn_cast<StoreInst>(BI->user_back());
 
   return LI && SI && !LI->isVolatile() && !SI->isVolatile() &&
          LI->hasOneUse() && LI->getPointerOperand() == SI->getPointerOperand();
@@ -573,11 +589,7 @@ InstructionCost SystemZTTIImpl::getArithmeticInstrCost(
   if (CostKind != TTI::TCK_RecipThroughput)
     return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info,
                                          Op2Info, Args, CxtI);
-  // Supporting only i8 type immediate-to-memory arithmatic operation until
-  // backend start folding it for i32 type if constant only has bits in one
-  // single byte set.
-  if (CxtI && Ty && !Ty->isVectorTy() && Ty->isIntegerTy(8) &&
-      isArithmeticFoldableRMW(CxtI))
+  if (CxtI && Ty && !Ty->isVectorTy() && isFoldableRMW(CxtI, Ty))
     return TTI::TCC_Free;
   // TODO: return a good value for BB-VECTORIZER that includes the
   // immediate loads, which we do not want to count for the loop
@@ -1328,35 +1340,6 @@ static bool isBswapIntrinsicCall(const Value *V) {
   return false;
 }
 
-static bool isStoreFoldableRMW(const Instruction *I) {
-  auto *SI = dyn_cast_or_null<StoreInst>(I);
-  if (!SI || SI->isVolatile())
-    return false;
-
-  auto *BI = dyn_cast<BinaryOperator>(SI->getValueOperand());
-  if (!BI || !BI->hasOneUse())
-    return false;
-
-  Value *Op0 = BI->getOperand(0), *Op1 = BI->getOperand(1);
-  if (!isa<ConstantInt>(Op0) && !isa<ConstantInt>(Op1))
-    return false;
-
-  unsigned Opcode = BI->getOpcode();
-  Value *V = nullptr;
-  if (Opcode == Instruction::And || Opcode == Instruction::Or ||
-      Opcode == Instruction::Xor || Opcode == Instruction::Add)
-    V = isa<ConstantInt>(Op0) ? Op1 : (isa<ConstantInt>(Op1) ? Op0 : nullptr);
-  else if (Opcode == Instruction::Sub && isa<ConstantInt>(Op1))
-    V = Op0;
-
-  if (!V)
-    return false;
-
-  auto *LI = dyn_cast_or_null<LoadInst>(V);
-  return LI && !LI->isVolatile() &&
-         LI->getPointerOperand() == SI->getPointerOperand();
-}
-
 InstructionCost SystemZTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src,
                                                 Align Alignment,
                                                 unsigned AddressSpace,
@@ -1369,9 +1352,8 @@ InstructionCost SystemZTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src,
   if (CostKind != TTI::TCK_RecipThroughput)
     return 1;
 
-  if (I && Opcode == Instruction::Store && !Src->isVectorTy() &&
-      Src->isIntegerTy(8)) {
-    if (isStoreFoldableRMW(I))
+  if (I && Opcode == Instruction::Store && !Src->isVectorTy()) {
+    if (isFoldableRMW(dyn_cast<Instruction>(I->getOperand(0)), Src))
       return TTI::TCC_Free;
   }
 
diff --git a/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll b/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll
index 1ba1b31171669..98aa0cda026d1 100644
--- a/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll
+++ b/llvm/test/Analysis/CostModel/SystemZ/fold-rmw-cost.ll
@@ -38,76 +38,76 @@ define void @test_xor(ptr %p) {
 
 define void @test_add(ptr %p) {
 ; CHECK-LABEL: 'test_add'
-; CHECK: cost of 0 {{.*}} add i8
-; CHECK: cost of 0 {{.*}} store i8
-  %v = load i8, ptr %p
-  %res = add i8 %v, 1
-  store i8 %res, ptr %p
+; CHECK: cost of 0 {{.*}} add i32
+; CHECK: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = add i32 %v, 1
+  store i32 %res, ptr %p
   ret void
 }
 
 define void @test_sub(ptr %p) {
 ; CHECK-LABEL: 'test_sub'
-; CHECK: cost of 0 {{.*}} sub i8
-; CHECK: cost of 0 {{.*}} store i8
-  %v = load i8, ptr %p
-  %res = sub i8 %v, 1
-  store i8 %res, ptr %p
+; CHECK: cost of 0 {{.*}} sub i32
+; CHECK: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = sub i32 %v, 1
+  store i32 %res, ptr %p
   ret void
 }
 
 ; Subtraction is not commutative.
 define void @test_sub_neg_imm_lhs(ptr %p) {
 ; CHECK-LABEL: 'test_sub_neg_imm_lhs'
-; CHECK-NOT: cost of 0 {{.*}} sub i8
-; CHECK-NOT: cost of 0 {{.*}} store i8
-  %v = load i8, ptr %p
-  %res = sub i8 1, %v
-  store i8 %res, ptr %p
+; CHECK-NOT: cost of 0 {{.*}} sub i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = sub i32 1, %v
+  store i32 %res, ptr %p
   ret void
 }
 
 ; Different Addresses.
 define void @test_diff_addr(ptr %p, ptr %q) {
 ; CHECK-LABEL: 'test_diff_addr'
-; CHECK-NOT: cost of 0 {{.*}} add i8
-; CHECK-NOT: cost of 0 {{.*}} store i8
-  %v = load i8, ptr %p
-  %res = add i8 %v, 1
-  store i8 %res, ptr %q
+; CHECK-NOT: cost of 0 {{.*}} add i32
+; CHECK-NOT: cost of 0 {{.*}} store i32
+  %v = load i32, ptr %p
+  %res = add i32 %v, 1
+  store i32 %res, ptr %q
   ret void
 }
 
 ; Multi-use of Arithmetic Result.
-define i8 @test_multi_use_arith(ptr %p) {
+define i32 @test_multi_use_arith(ptr %p) {
 ; CHECK-LABEL: 'test_multi_use_arith'
 ; CHECK-NOT: cost of 0 {{.*}} add i8
 ; CHECK-NOT: cost of 0 {{.*}} store i8
-  %v = load i8, ptr %p
-  %res = add i8 %v, 1
-  store i8 %res, ptr %p
-  ret i8 %res
+  %v = load i32, ptr %p
+  %res = add i32 %v, 1
+  store i32 %res, ptr %p
+  ret i32 %res
 }
 
 ; Multi-use of Load Result.
 define i8 @test_multi_use_load(ptr %p) {
 ; CHECK-LABEL: 'test_multi_use_load'
-; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} and i8
 ; CHECK-NOT: cost of 0 {{.*}} store i8
   %v = load i8, ptr %p
-  %res = add i8 %v, 1
+  %res = and i8 %v, 1
   store i8 %res, ptr %p
-  %use2 = add i8 %v, %res
+  %use2 = and i8 %v, %res
   ret i8 %use2
 }
 
 ; Neither Operand is Immediate.
 define void @test_no_immediate(ptr %p, i8 %reg) {
 ; CHECK-LABEL: 'test_no_immediate'
-; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} and i8
 ; CHECK-NOT: cost of 0 {{.*}} store i8
   %v = load i8, ptr %p
-  %res = add i8 %v, %reg
+  %res = and i8 %v, %reg
   store i8 %res, ptr %p
   ret void
 }
@@ -115,10 +115,10 @@ define void @test_no_immediate(ptr %p, i8 %reg) {
 ; Both Operands are Immediate.
 define void @test_both_immediate(ptr %p) {
 ; CHECK-LABEL: 'test_both_immediate'
-; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} or i8
 ; CHECK-NOT: cost of 0 {{.*}} store i8
   %v = load i8, ptr %p
-  %res = add i8 5, 10
+  %res = or i8 5, 10
   store i8 %res, ptr %p
   ret void
 }
@@ -126,22 +126,22 @@ define void @test_both_immediate(ptr %p) {
 ; Volatile Load/Store.
 define void @test_volatile(ptr %p) {
 ; CHECK-LABEL: 'test_volatile'
-; CHECK-NOT: cost of 0 {{.*}} add i8
+; CHECK-NOT: cost of 0 {{.*}} xor i8
 ; CHECK-NOT: cost of 0 {{.*}} store i8
   %v = load volatile i8, ptr %p
-  %res = add i8 %v, 1
+  %res = xor i8 %v, 1
   store volatile i8 %res, ptr %p
   ret void
 }
 
 ; Vector types cost should not be changed.
-define void @test_vector_no_fold(<4 x i8> %val, ptr %p) {
+define void @test_vector_no_fold(<4 x i32> %val, ptr %p) {
 ; CHECK-LABEL: 'test_vector_no_fold'
-; CHECK-NOT: cost of 0 {{.*}} add <4 x i8>
-; CHECK-NOT: cost of 0 {{.*}} store <4 x i8>
-  %v = load <4 x i8>, ptr %p
-  %res = add <4 x i8> %v, %val
-  store <4 x i8> %res, ptr %p
+; CHECK-NOT: cost of 0 {{.*}} add <4 x i32>
+; CHECK-NOT: cost of 0 {{.*}} store <4 x i32>
+  %v = load <4 x i32>, ptr %p
+  %res = add <4 x i32> %v, %val
+  store <4 x i32> %res, ptr %p
   ret void
 }
 



More information about the llvm-commits mailing list