[llvm] [LAA] Add stencil group merging to reduce runtime pointer checks (PR #187252)
Igor Kirillov via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 2 07:55:57 PDT 2026
================
@@ -0,0 +1,2951 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes='print<access-info>' -stencil-runtime-check-merge=force -disable-output %s 2>&1 | FileCheck --check-prefixes=CHECK,MERGE %s
+; RUN: opt -passes='print<access-info>' -stencil-runtime-check-merge=off -disable-output %s 2>&1 | FileCheck --check-prefixes=CHECK,NOMERGE %s
+
+;; Test 1: Basic stencil merge with one runtime stride.
+;; 6 loads from %a at byte offsets {-3*cdj, -2*cdj, -cdj, +cdj, +2*cdj, +3*cdj}
+;; relative to base = %a + 8*iv. Store to %out.
+;; With merge: all 6 loads form 1 merged group with bounds spanning [-3*cdj, +3*cdj]
+;; relative to base, producing 1 runtime check and 1 stride predicate (cdj > 0).
+;; Without merge: 6 separate groups (each load alone), 6 checks, no predicates.
+define void @stencil_merge_single_stride(ptr %a, ptr %out, i64 %n, i64 %cdj) {
+; MERGE-LABEL: 'stencil_merge_single_stride'
+; MERGE-NEXT: loop:
+; MERGE-NEXT: Memory dependences are safe with run-time checks
+; MERGE-NEXT: Dependences:
+; MERGE-NEXT: Run-time memory checks:
+; MERGE-NEXT: Check 0:
+; MERGE-NEXT: Comparing group GRP0:
+; MERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; MERGE-NEXT: Against group GRP1:
+; MERGE-NEXT: %p5 = getelementptr inbounds i8, ptr %base, i64 %pos3cdj
+; MERGE-NEXT: %p4 = getelementptr inbounds i8, ptr %base, i64 %pos2cdj
+; MERGE-NEXT: %p3 = getelementptr inbounds i8, ptr %base, i64 %cdj
+; MERGE-NEXT: %p2 = getelementptr inbounds i8, ptr %base, i64 %negcdj
+; MERGE-NEXT: %p1 = getelementptr inbounds i8, ptr %base, i64 %neg2cdj
+; MERGE-NEXT: %p0 = getelementptr inbounds i8, ptr %base, i64 %neg3cdj
+; MERGE-NEXT: Grouped accesses:
+; MERGE-NEXT: Group GRP0:
+; MERGE-NEXT: (Low: (24 + %out) High: (-24 + (8 * %n) + %out))
+; MERGE-NEXT: Member: {(24 + %out),+,8}<nuw><%loop>
+; MERGE-NEXT: Group GRP1:
+; MERGE-NEXT: (Low: (24 + (-3 * %cdj) + %a) High: (-24 + (3 * %cdj) + (8 * %n) + %a))
+; MERGE-NEXT: Member: {(24 + (3 * %cdj) + %a),+,8}<nw><%loop>
+; MERGE-NEXT: Member: {(24 + (2 * %cdj) + %a),+,8}<nw><%loop>
+; MERGE-NEXT: Member: {(24 + %cdj + %a),+,8}<nw><%loop>
+; MERGE-NEXT: Member: {(24 + (-1 * %cdj) + %a),+,8}<nw><%loop>
+; MERGE-NEXT: Member: {(24 + (-2 * %cdj) + %a),+,8}<nw><%loop>
+; MERGE-NEXT: Member: {(24 + (-3 * %cdj) + %a),+,8}<nw><%loop>
+; MERGE-EMPTY:
+; MERGE-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; MERGE-NEXT: SCEV assumptions:
+; MERGE-NEXT: Compare predicate: %cdj sgt) 0
+; MERGE-EMPTY:
+; MERGE-NEXT: Expressions re-written:
+;
+; NOMERGE-LABEL: 'stencil_merge_single_stride'
+; NOMERGE-NEXT: loop:
+; NOMERGE-NEXT: Memory dependences are safe with run-time checks
+; NOMERGE-NEXT: Dependences:
+; NOMERGE-NEXT: Run-time memory checks:
+; NOMERGE-NEXT: Check 0:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP1:
+; NOMERGE-NEXT: %p5 = getelementptr inbounds i8, ptr %base, i64 %pos3cdj
+; NOMERGE-NEXT: Check 1:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP2:
+; NOMERGE-NEXT: %p4 = getelementptr inbounds i8, ptr %base, i64 %pos2cdj
+; NOMERGE-NEXT: Check 2:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP3:
+; NOMERGE-NEXT: %p3 = getelementptr inbounds i8, ptr %base, i64 %cdj
+; NOMERGE-NEXT: Check 3:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP4:
+; NOMERGE-NEXT: %p2 = getelementptr inbounds i8, ptr %base, i64 %negcdj
+; NOMERGE-NEXT: Check 4:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP5:
+; NOMERGE-NEXT: %p1 = getelementptr inbounds i8, ptr %base, i64 %neg2cdj
+; NOMERGE-NEXT: Check 5:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP6:
+; NOMERGE-NEXT: %p0 = getelementptr inbounds i8, ptr %base, i64 %neg3cdj
+; NOMERGE-NEXT: Grouped accesses:
+; NOMERGE-NEXT: Group GRP0:
+; NOMERGE-NEXT: (Low: (24 + %out) High: (-24 + (8 * %n) + %out))
+; NOMERGE-NEXT: Member: {(24 + %out),+,8}<nuw><%loop>
+; NOMERGE-NEXT: Group GRP1:
+; NOMERGE-NEXT: (Low: (24 + (3 * %cdj) + %a) High: (-24 + (3 * %cdj) + (8 * %n) + %a))
+; NOMERGE-NEXT: Member: {(24 + (3 * %cdj) + %a),+,8}<nw><%loop>
+; NOMERGE-NEXT: Group GRP2:
+; NOMERGE-NEXT: (Low: (24 + (2 * %cdj) + %a) High: (-24 + (2 * %cdj) + (8 * %n) + %a))
+; NOMERGE-NEXT: Member: {(24 + (2 * %cdj) + %a),+,8}<nw><%loop>
+; NOMERGE-NEXT: Group GRP3:
+; NOMERGE-NEXT: (Low: (24 + %cdj + %a) High: (-24 + (8 * %n) + %cdj + %a))
+; NOMERGE-NEXT: Member: {(24 + %cdj + %a),+,8}<nw><%loop>
+; NOMERGE-NEXT: Group GRP4:
+; NOMERGE-NEXT: (Low: (24 + (-1 * %cdj) + %a) High: (-24 + (8 * %n) + (-1 * %cdj) + %a))
+; NOMERGE-NEXT: Member: {(24 + (-1 * %cdj) + %a),+,8}<nw><%loop>
+; NOMERGE-NEXT: Group GRP5:
+; NOMERGE-NEXT: (Low: (24 + (-2 * %cdj) + %a) High: (-24 + (8 * %n) + (-2 * %cdj) + %a))
+; NOMERGE-NEXT: Member: {(24 + (-2 * %cdj) + %a),+,8}<nw><%loop>
+; NOMERGE-NEXT: Group GRP6:
+; NOMERGE-NEXT: (Low: (24 + (-3 * %cdj) + %a) High: (-24 + (8 * %n) + (-3 * %cdj) + %a))
+; NOMERGE-NEXT: Member: {(24 + (-3 * %cdj) + %a),+,8}<nw><%loop>
+; NOMERGE-EMPTY:
+; NOMERGE-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; NOMERGE-NEXT: SCEV assumptions:
+; NOMERGE-EMPTY:
+; NOMERGE-NEXT: Expressions re-written:
+;
+entry:
+ %cmp = icmp sgt i64 %n, 6
+ br i1 %cmp, label %loop, label %exit
+
+loop:
+ %iv = phi i64 [ 3, %entry ], [ %iv.next, %loop ]
+ %base = getelementptr inbounds double, ptr %a, i64 %iv
+
+ %neg3cdj = mul nsw i64 %cdj, -3
+ %p0 = getelementptr inbounds i8, ptr %base, i64 %neg3cdj
+ %v0 = load double, ptr %p0, align 8
+
+ %neg2cdj = mul nsw i64 %cdj, -2
+ %p1 = getelementptr inbounds i8, ptr %base, i64 %neg2cdj
+ %v1 = load double, ptr %p1, align 8
+
+ %negcdj = sub nsw i64 0, %cdj
+ %p2 = getelementptr inbounds i8, ptr %base, i64 %negcdj
+ %v2 = load double, ptr %p2, align 8
+
+ %p3 = getelementptr inbounds i8, ptr %base, i64 %cdj
+ %v3 = load double, ptr %p3, align 8
+
+ %pos2cdj = mul nsw i64 %cdj, 2
+ %p4 = getelementptr inbounds i8, ptr %base, i64 %pos2cdj
+ %v4 = load double, ptr %p4, align 8
+
+ %pos3cdj = mul nsw i64 %cdj, 3
+ %p5 = getelementptr inbounds i8, ptr %base, i64 %pos3cdj
+ %v5 = load double, ptr %p5, align 8
+
+ %s0 = fadd double %v0, %v1
+ %s1 = fadd double %s0, %v2
+ %s2 = fadd double %s1, %v3
+ %s3 = fadd double %s2, %v4
+ %s4 = fadd double %s3, %v5
+
+ %outp = getelementptr inbounds double, ptr %out, i64 %iv
+ store double %s4, ptr %outp, align 8
+
+ %iv.next = add nuw nsw i64 %iv, 1
+ %sub = sub nsw i64 %n, 3
+ %cond = icmp slt i64 %iv.next, %sub
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+
+;; Test 2: Constant offsets only — existing grouping handles this.
+;; 4 loads from %a at constant byte offsets {-16, -8, +8, +16}. Store to %out.
+;; The standard grouping algorithm merges these into 1 group (constant SCEV diffs).
+;; Both with and without flag: 1 check, 2 groups, no predicates.
+define void @constant_offsets_only(ptr %a, ptr %out, i64 %n) {
+; CHECK-LABEL: 'constant_offsets_only'
+; CHECK-NEXT: loop:
+; CHECK-NEXT: Memory dependences are safe with run-time checks
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Check 0:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; CHECK-NEXT: Against group GRP1:
+; CHECK-NEXT: %p3 = getelementptr inbounds i8, ptr %base, i64 16
+; CHECK-NEXT: %p2 = getelementptr inbounds i8, ptr %base, i64 8
+; CHECK-NEXT: %p1 = getelementptr inbounds i8, ptr %base, i64 -8
+; CHECK-NEXT: %p0 = getelementptr inbounds i8, ptr %base, i64 -16
+; CHECK-NEXT: Grouped accesses:
+; CHECK-NEXT: Group GRP0:
+; CHECK-NEXT: (Low: (32 + %out) High: (-32 + (8 * %n) + %out))
+; CHECK-NEXT: Member: {(32 + %out),+,8}<nuw><%loop>
+; CHECK-NEXT: Group GRP1:
+; CHECK-NEXT: (Low: (16 + %a) High: (-16 + (8 * %n) + %a))
+; CHECK-NEXT: Member: {(48 + %a),+,8}<nuw><%loop>
+; CHECK-NEXT: Member: {(40 + %a),+,8}<nuw><%loop>
+; CHECK-NEXT: Member: {(24 + %a),+,8}<nw><%loop>
+; CHECK-NEXT: Member: {(16 + %a),+,8}<nw><%loop>
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ %cmp = icmp sgt i64 %n, 8
+ br i1 %cmp, label %loop, label %exit
+
+loop:
+ %iv = phi i64 [ 4, %entry ], [ %iv.next, %loop ]
+ %base = getelementptr inbounds double, ptr %a, i64 %iv
+
+ %p0 = getelementptr inbounds i8, ptr %base, i64 -16
+ %v0 = load double, ptr %p0, align 8
+
+ %p1 = getelementptr inbounds i8, ptr %base, i64 -8
+ %v1 = load double, ptr %p1, align 8
+
+ %p2 = getelementptr inbounds i8, ptr %base, i64 8
+ %v2 = load double, ptr %p2, align 8
+
+ %p3 = getelementptr inbounds i8, ptr %base, i64 16
+ %v3 = load double, ptr %p3, align 8
+
+ %s0 = fadd double %v0, %v1
+ %s1 = fadd double %s0, %v2
+ %s2 = fadd double %s1, %v3
+
+ %outp = getelementptr inbounds double, ptr %out, i64 %iv
+ store double %s2, ptr %outp, align 8
+
+ %iv.next = add nuw nsw i64 %iv, 1
+ %sub = sub nsw i64 %n, 4
+ %cond = icmp slt i64 %iv.next, %sub
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+
+;; Test 3: Cost model rejection — only 2 loads with runtime stride.
+;; Merging saves 1 check but costs 1 predicate = net 0 saving -> skip.
+;; Same result with and without the flag: 2 checks, 3 groups, no predicates.
+define void @cost_model_rejection(ptr %a, ptr %out, i64 %n, i64 %cdj) {
+; CHECK-LABEL: 'cost_model_rejection'
+; CHECK-NEXT: loop:
+; CHECK-NEXT: Memory dependences are safe with run-time checks
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Check 0:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; CHECK-NEXT: Against group GRP1:
+; CHECK-NEXT: %base = getelementptr inbounds double, ptr %a, i64 %iv
+; CHECK-NEXT: Check 1:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; CHECK-NEXT: Against group GRP2:
+; CHECK-NEXT: %p0 = getelementptr inbounds i8, ptr %base, i64 %cdj
+; CHECK-NEXT: Grouped accesses:
+; CHECK-NEXT: Group GRP0:
+; CHECK-NEXT: (Low: (16 + %out) High: (-16 + (8 * %n) + %out))
+; CHECK-NEXT: Member: {(16 + %out),+,8}<nuw><%loop>
+; CHECK-NEXT: Group GRP1:
+; CHECK-NEXT: (Low: (16 + %a) High: (-16 + (8 * %n) + %a))
+; CHECK-NEXT: Member: {(16 + %a),+,8}<nuw><%loop>
+; CHECK-NEXT: Group GRP2:
+; CHECK-NEXT: (Low: (16 + %cdj + %a) High: (-16 + (8 * %n) + %cdj + %a))
+; CHECK-NEXT: Member: {(16 + %cdj + %a),+,8}<nw><%loop>
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ %cmp = icmp sgt i64 %n, 4
+ br i1 %cmp, label %loop, label %exit
+
+loop:
+ %iv = phi i64 [ 2, %entry ], [ %iv.next, %loop ]
+ %base = getelementptr inbounds double, ptr %a, i64 %iv
+
+ %p0 = getelementptr inbounds i8, ptr %base, i64 %cdj
+ %v0 = load double, ptr %p0, align 8
+
+ %v1 = load double, ptr %base, align 8
+
+ %s = fadd double %v0, %v1
+
+ %outp = getelementptr inbounds double, ptr %out, i64 %iv
+ store double %s, ptr %outp, align 8
+
+ %iv.next = add nuw nsw i64 %iv, 1
+ %sub = sub nsw i64 %n, 2
+ %cond = icmp slt i64 %iv.next, %sub
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+
+;; Test 4: Invariant + strided reads from same base -> different access ranges.
+;; A strided read {%a,+,8} and an invariant read from %a have different
+;; access ranges (8*n vs 8), so merging must NOT combine them.
+;; Same result with and without flag: 2 checks, 3 separate groups.
+define void @different_steps_no_merge(ptr %a, ptr %out, i64 %n) {
+; CHECK-LABEL: 'different_steps_no_merge'
+; CHECK-NEXT: loop:
+; CHECK-NEXT: Memory dependences are safe with run-time checks
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Check 0:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %gep.out = getelementptr inbounds double, ptr %out, i64 %iv
+; CHECK-NEXT: Against group GRP1:
+; CHECK-NEXT: %gep.a = getelementptr inbounds double, ptr %a, i64 %iv
+; CHECK-NEXT: Check 1:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %gep.out = getelementptr inbounds double, ptr %out, i64 %iv
+; CHECK-NEXT: Against group GRP2:
+; CHECK-NEXT: ptr %a
+; CHECK-NEXT: Grouped accesses:
+; CHECK-NEXT: Group GRP0:
+; CHECK-NEXT: (Low: %out High: ((8 * %n) + %out))
+; CHECK-NEXT: Member: {%out,+,8}<nuw><%loop>
+; CHECK-NEXT: Group GRP1:
+; CHECK-NEXT: (Low: %a High: ((8 * %n) + %a))
+; CHECK-NEXT: Member: {%a,+,8}<nuw><%loop>
+; CHECK-NEXT: Group GRP2:
+; CHECK-NEXT: (Low: %a High: (8 + %a))
+; CHECK-NEXT: Member: %a
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+; GRP0: store to %out (write, different DepSet):
+; GRP1: strided read from %a (step=8):
+; GRP2: invariant read from %a (step=0, different from GRP1):
+entry:
+ %cmp = icmp sgt i64 %n, 2
+ br i1 %cmp, label %loop, label %exit
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+
+ ; Invariant read from %a (step = 0, different from strided)
+ %v.inv = load double, ptr %a, align 8
+
+ ; Strided read: {%a,+,8}
+ %gep.a = getelementptr inbounds double, ptr %a, i64 %iv
+ %v.strided = load double, ptr %gep.a, align 8
+
+ ; Store to %out (different DepSet, triggers runtime checks)
+ %sum = fadd double %v.strided, %v.inv
+ %gep.out = getelementptr inbounds double, ptr %out, i64 %iv
+ store double %sum, ptr %gep.out, align 8
+
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cond = icmp slt i64 %iv.next, %n
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+
+;; Test 5: A member with no stride term.
+;; 3 loads from %a at offsets {0, -cdj, -2*cdj}. The base load has no cdj
+;; term at all, so its cdj coefficient counts as 0. The candidate comparison
+;; must see that 0: the base member (+0) is the high bound, above -cdj and
+;; -2*cdj, even though it never mentions cdj.
+define void @coefficient_clamping_regression(ptr %a, ptr %out, i64 %n, i64 %cdj) {
+; MERGE-LABEL: 'coefficient_clamping_regression'
+; MERGE-NEXT: loop:
+; MERGE-NEXT: Memory dependences are safe with run-time checks
+; MERGE-NEXT: Dependences:
+; MERGE-NEXT: Run-time memory checks:
+; MERGE-NEXT: Check 0:
+; MERGE-NEXT: Comparing group GRP0:
+; MERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; MERGE-NEXT: Against group GRP1:
+; MERGE-NEXT: %p2 = getelementptr inbounds i8, ptr %base, i64 %neg2cdj
+; MERGE-NEXT: %p1 = getelementptr inbounds i8, ptr %base, i64 %negcdj
+; MERGE-NEXT: %base = getelementptr inbounds double, ptr %a, i64 %iv
+; MERGE-NEXT: Grouped accesses:
+; MERGE-NEXT: Group GRP0:
+; MERGE-NEXT: (Low: (16 + %out) High: (-16 + (8 * %n) + %out))
+; MERGE-NEXT: Member: {(16 + %out),+,8}<nuw><%loop>
+; MERGE-NEXT: Group GRP1:
+; MERGE-NEXT: (Low: (16 + (-2 * %cdj) + %a) High: (-16 + (8 * %n) + %a))
+; MERGE-NEXT: Member: {(16 + (-2 * %cdj) + %a),+,8}<nw><%loop>
+; MERGE-NEXT: Member: {(16 + (-1 * %cdj) + %a),+,8}<nw><%loop>
+; MERGE-NEXT: Member: {(16 + %a),+,8}<nuw><%loop>
+; MERGE-EMPTY:
+; MERGE-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; MERGE-NEXT: SCEV assumptions:
+; MERGE-NEXT: Compare predicate: %cdj sgt) 0
+; MERGE-EMPTY:
+; MERGE-NEXT: Expressions re-written:
+;
+; NOMERGE-LABEL: 'coefficient_clamping_regression'
+; NOMERGE-NEXT: loop:
+; NOMERGE-NEXT: Memory dependences are safe with run-time checks
+; NOMERGE-NEXT: Dependences:
+; NOMERGE-NEXT: Run-time memory checks:
+; NOMERGE-NEXT: Check 0:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP1:
+; NOMERGE-NEXT: %p2 = getelementptr inbounds i8, ptr %base, i64 %neg2cdj
+; NOMERGE-NEXT: Check 1:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP2:
+; NOMERGE-NEXT: %p1 = getelementptr inbounds i8, ptr %base, i64 %negcdj
+; NOMERGE-NEXT: Check 2:
+; NOMERGE-NEXT: Comparing group GRP0:
+; NOMERGE-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; NOMERGE-NEXT: Against group GRP3:
+; NOMERGE-NEXT: %base = getelementptr inbounds double, ptr %a, i64 %iv
+; NOMERGE-NEXT: Grouped accesses:
+; NOMERGE-NEXT: Group GRP0:
+; NOMERGE-NEXT: (Low: (16 + %out) High: (-16 + (8 * %n) + %out))
+; NOMERGE-NEXT: Member: {(16 + %out),+,8}<nuw><%loop>
+; NOMERGE-NEXT: Group GRP1:
+; NOMERGE-NEXT: (Low: (16 + (-2 * %cdj) + %a) High: (-16 + (8 * %n) + (-2 * %cdj) + %a))
+; NOMERGE-NEXT: Member: {(16 + (-2 * %cdj) + %a),+,8}<nw><%loop>
+; NOMERGE-NEXT: Group GRP2:
+; NOMERGE-NEXT: (Low: (16 + (-1 * %cdj) + %a) High: (-16 + (8 * %n) + (-1 * %cdj) + %a))
+; NOMERGE-NEXT: Member: {(16 + (-1 * %cdj) + %a),+,8}<nw><%loop>
+; NOMERGE-NEXT: Group GRP3:
+; NOMERGE-NEXT: (Low: (16 + %a) High: (-16 + (8 * %n) + %a))
+; NOMERGE-NEXT: Member: {(16 + %a),+,8}<nuw><%loop>
+; NOMERGE-EMPTY:
+; NOMERGE-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; NOMERGE-NEXT: SCEV assumptions:
+; NOMERGE-EMPTY:
+; NOMERGE-NEXT: Expressions re-written:
+;
+entry:
+ %cmp = icmp sgt i64 %n, 4
+ br i1 %cmp, label %loop, label %exit
+
+loop:
+ %iv = phi i64 [ 2, %entry ], [ %iv.next, %loop ]
+ %base = getelementptr inbounds double, ptr %a, i64 %iv
+
+ ; load at base + 0 (no stride offset -- this is the base member)
+ %v0 = load double, ptr %base, align 8
+
+ ; load at base - cdj
+ %negcdj = sub nsw i64 0, %cdj
+ %p1 = getelementptr inbounds i8, ptr %base, i64 %negcdj
+ %v1 = load double, ptr %p1, align 8
+
+ ; load at base - 2*cdj
+ %neg2cdj = mul nsw i64 %cdj, -2
+ %p2 = getelementptr inbounds i8, ptr %base, i64 %neg2cdj
+ %v2 = load double, ptr %p2, align 8
+
+ %s0 = fadd double %v0, %v1
+ %s1 = fadd double %s0, %v2
+
+ %outp = getelementptr inbounds double, ptr %out, i64 %iv
+ store double %s1, ptr %outp, align 8
+
+ %iv.next = add nuw nsw i64 %iv, 1
+ %sub = sub nsw i64 %n, 2
+ %cond = icmp slt i64 %iv.next, %sub
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+
+;; Test 6: Predicated accesses are rejected from stencil merging.
+;; Models the dilateKernel pattern (llvm-test-suite MicroBenchmarks/ImageProcessing/Dilate):
+;; 3 loads at {-cdj, 0, +cdj} where the -cdj and +cdj loads are conditional.
+;; Predicated loads have overapproximated SCEV bounds; merging would widen
+;; them further, causing false runtime overlap detection.
+;; Both modes: 3 checks (groups stay separate), no predicates.
+define void @predicated_access_rejection(ptr %a, ptr %out, i64 %n, i64 %cdj, i1 %c1, i1 %c2) {
+; CHECK-LABEL: 'predicated_access_rejection'
+; CHECK-NEXT: loop:
+; CHECK-NEXT: Memory dependences are safe with run-time checks
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Check 0:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; CHECK-NEXT: Against group GRP1:
+; CHECK-NEXT: %p2 = getelementptr inbounds i8, ptr %base, i64 %negcdj
+; CHECK-NEXT: Check 1:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; CHECK-NEXT: Against group GRP2:
+; CHECK-NEXT: %p1 = getelementptr inbounds i8, ptr %base, i64 %cdj
+; CHECK-NEXT: Check 2:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %outp = getelementptr inbounds double, ptr %out, i64 %iv
+; CHECK-NEXT: Against group GRP3:
+; CHECK-NEXT: %base = getelementptr inbounds double, ptr %a, i64 %iv
+; CHECK-NEXT: Grouped accesses:
+; CHECK-NEXT: Group GRP0:
+; CHECK-NEXT: (Low: %out High: ((8 * %n) + %out))
+; CHECK-NEXT: Member: {%out,+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP1:
+; CHECK-NEXT: (Low: ((-1 * %cdj) + %a) High: ((8 * %n) + (-1 * %cdj) + %a))
+; CHECK-NEXT: Member: {((-1 * %cdj) + %a),+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP2:
+; CHECK-NEXT: (Low: (%cdj + %a) High: ((8 * %n) + %cdj + %a))
+; CHECK-NEXT: Member: {(%cdj + %a),+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP3:
+; CHECK-NEXT: (Low: %a High: ((8 * %n) + %a))
+; CHECK-NEXT: Member: {%a,+,8}<nuw><%loop>
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+; 3 checks: each temp group separately vs %out (not merged due to predication):
----------------
igogo-x86 wrote:
Removed.
https://github.com/llvm/llvm-project/pull/187252
More information about the llvm-commits
mailing list