[llvm] [X86][LV] Prefer tail folding for i64 div/rem loops (PR #227721)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 01:36:53 PDT 2026
Andarwinux wrote:
> This PR probably needs performance numbers to make sure this is always profitable
If iterations are small, tail folding is noticeably faster than no tail folding, and when iterations are large, there is only a slight slowdown. Maybe should wait for #208764 to be merged first?
```c++
#include <cstdint>
#include <chrono>
#include <cstdio>
#include <cstdlib>
#include <limits>
#include <random>
static inline uint64_t foo(uint64_t *__restrict a, uint64_t *__restrict b, int n) {
uint64_t x = 0;
for (int i = 0; i < n; i++) {
x += a[i] / b[i];
}
return x;
}
int main(int argc, char **argv) {
const int n = argc > 1 ? std::atoi(argv[1]) : 1048576;
const int repeat = argc > 2 ? std::atoi(argv[2]) : 1000;
uint64_t *a = new uint64_t[n];
uint64_t *b = new uint64_t[n];
std::mt19937_64 rng(0x12345678u);
std::uniform_int_distribution<uint64_t> dist(UINT32_MAX, UINT64_MAX);
for (int i = 0; i < n; i++) {
a[i] = dist(rng);
b[i] = dist(rng);
}
uint64_t result = 0;
auto t0 = std::chrono::steady_clock::now();
for (int i = 0; i < repeat; i++)
result += foo(a, b, n);
auto t1 = std::chrono::steady_clock::now();
double time = std::chrono::duration<double>(t1 - t0).count();
std::printf("n = %d\n", n);
std::printf("repeat = %d\n", repeat);
std::printf("%.9f\n", time);
std::printf("%lu\n", result);
return 0;
}
```
```
[tmp]$ clang++ -O3 -march=tigerlake -mprefer-vector-width=512 bench.cc -o a
[tmp]$ clang++ -O3 -march=tigerlake -mprefer-vector-width=512 bench.cc -mllvm -tail-folding-policy=prefer-fold-tail -mllvm -prefer-predicated-reduction-select -o b
[tmp]$ ./a 3 10000000
n = 3
repeat = 10000000
0.068204177
830000000
[tmp]$ ./b 3 10000000
n = 3
repeat = 10000000
0.066675285
830000000
[tmp]$ ./a 5 10000000
n = 5
repeat = 10000000
0.065823685
830000000
[tmp]$ ./b 5 10000000
n = 5
repeat = 10000000
0.070479154
830000000
[tmp]$ ./a 7 10000000
n = 7
repeat = 10000000
0.081586868
850000000
[tmp]$ ./b 7 10000000
n = 7
repeat = 10000000
0.066645344
850000000
[tmp]$ ./a 16 10000000
n = 16
repeat = 10000000
0.186788568
910000000
[tmp]$ ./b 16 10000000
n = 16
repeat = 10000000
0.112079637
910000000
[tmp]$ ./a 31 10000000
n = 31
repeat = 10000000
0.343280714
2140000000
[tmp]$ ./b 31 10000000
n = 31
repeat = 10000000
0.203392337
2140000000
[tmp]$ ./a 32 10000000
n = 32
repeat = 10000000
0.198278182
2140000000
[tmp]$ ./b 32 10000000
n = 32
repeat = 10000000
0.201642710
2140000000
[tmp]$ ./a 63 10000000
n = 63
repeat = 10000000
0.540284426
2640000000
[tmp]$ ./b 63 10000000
n = 63
repeat = 10000000
0.391395006
2640000000
[tmp]$ ./a 64 10000000
n = 64
repeat = 10000000
0.371330738
2660000000
[tmp]$ ./b 64 10000000
n = 64
repeat = 10000000
0.387223290
2660000000
[tmp]$ ./a 512 10000000
n = 512
repeat = 10000000
2.799437441
18080000000
[tmp]$ ./b 512 10000000
n = 512
repeat = 10000000
3.005602510
18080000000
[tmp]$ ./a
n = 1048576
repeat = 1000
0.766780443
7611089000
[tmp]$ ./b
n = 1048576
repeat = 1000
0.739762262
7611089000
```
https://github.com/llvm/llvm-project/pull/227721
More information about the llvm-commits
mailing list