[llvm] [msan] handle llvm.masked.{udiv, sdiv, urem, srem} intrinsics (PR #225363)
Emilio Cota via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 08:53:33 PDT 2026
https://github.com/cota updated https://github.com/llvm/llvm-project/pull/225363
>From fe69a81f353d850c40b167f250029073deff5dfc Mon Sep 17 00:00:00 2001
From: Emilio Cota <ecg at google.com>
Date: Tue, 22 Sep 2026 10:30:08 +0000
Subject: [PATCH] [msan] handle llvm.masked.{udiv,sdiv,urem,srem} intrinsics
Check the divisor and propagate the dividend's shadow on active lanes,
poisoning disabled result lanes. This poisoning is consistent with
the definition of these intrinsics; see #189705 (9ba774566).
This gets rid of false positives from inactive input lanes since
visitInstruction() was checking every operand across all lanes,
active or not.
---
.../Instrumentation/MemorySanitizer.cpp | 36 ++
.../MemorySanitizer/masked-divrem.ll | 547 ++++++++++++++++++
2 files changed, 583 insertions(+)
create mode 100644 llvm/test/Instrumentation/MemorySanitizer/masked-divrem.ll
diff --git a/llvm/lib/Transforms/Instrumentation/MemorySanitizer.cpp b/llvm/lib/Transforms/Instrumentation/MemorySanitizer.cpp
index 4084f3d580e57a..0120b27fb13c51 100644
--- a/llvm/lib/Transforms/Instrumentation/MemorySanitizer.cpp
+++ b/llvm/lib/Transforms/Instrumentation/MemorySanitizer.cpp
@@ -4630,6 +4630,36 @@ struct MemorySanitizerVisitor : public InstVisitor<MemorySanitizerVisitor> {
setOrigin(&I, Origin);
}
+ // e.g., <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> %dividend,
+ // <4 x i32> %divisor,
+ // <4 x i1> %mask)
+ //
+ // As handleIntegerDiv(), but per-lane: strict on the divisor and propagating
+ // the dividend, both only on the enabled lanes. Disabled lanes cannot cause
+ // undefined behaviour, and their result is poison.
+ void handleMaskedIntegerDivRem(IntrinsicInst &I) {
+ IRBuilder<> IRB(&I);
+ Value *Dividend = I.getArgOperand(0);
+ Value *Divisor = I.getArgOperand(1);
+ Value *Mask = I.getArgOperand(2);
+
+ insertCheckShadowOf(Mask, &I);
+
+ Value *MaskedDivisorShadow = IRB.CreateSelect(
+ Mask, getShadow(Divisor), getCleanShadow(Divisor), "_msmaskeddivisor");
+ insertCheckShadow(MaskedDivisorShadow, getOrigin(Divisor), &I);
+
+ if (!PropagateShadow) {
+ setShadow(&I, getCleanShadow(&I));
+ setOrigin(&I, getCleanOrigin());
+ return;
+ }
+
+ setShadow(&I, IRB.CreateSelect(Mask, getShadow(Dividend),
+ getPoisonedShadow(&I), "_msmaskeddiv"));
+ setOrigin(&I, getOrigin(Dividend));
+ }
+
// e.g., void @llvm.x86.avx.maskstore.ps.256(ptr, <8 x i32>, <8 x float>)
// dst mask src
//
@@ -5937,6 +5967,12 @@ struct MemorySanitizerVisitor : public InstVisitor<MemorySanitizerVisitor> {
case Intrinsic::masked_load:
handleMaskedLoad(I);
break;
+ case Intrinsic::masked_udiv:
+ case Intrinsic::masked_sdiv:
+ case Intrinsic::masked_urem:
+ case Intrinsic::masked_srem:
+ handleMaskedIntegerDivRem(I);
+ break;
case Intrinsic::vector_reduce_and:
handleVectorReduceAndIntrinsic(I);
break;
diff --git a/llvm/test/Instrumentation/MemorySanitizer/masked-divrem.ll b/llvm/test/Instrumentation/MemorySanitizer/masked-divrem.ll
new file mode 100644
index 00000000000000..28966688eda8b3
--- /dev/null
+++ b/llvm/test/Instrumentation/MemorySanitizer/masked-divrem.ll
@@ -0,0 +1,547 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -msan-check-access-address=0 -S -passes=msan 2>&1 | FileCheck %s --implicit-check-not="call void @__msan_warning"
+; RUN: opt < %s -msan-check-access-address=0 -msan-track-origins=1 -S -passes=msan 2>&1 | FileCheck %s --check-prefixes=ORIGINS --implicit-check-not="call void @__msan_warning"
+
+target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f32:32:32-f64:64:64-v64:64:64-v128:128:128-a0:0:64-s0:64:64-f80:128:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+declare <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32>, <4 x i32>, <4 x i1>)
+declare <4 x i32> @llvm.masked.sdiv.v4i32(<4 x i32>, <4 x i32>, <4 x i1>)
+declare <4 x i32> @llvm.masked.urem.v4i32(<4 x i32>, <4 x i32>, <4 x i1>)
+declare <4 x i32> @llvm.masked.srem.v4i32(<4 x i32>, <4 x i32>, <4 x i1>)
+declare <4 x i32> @llvm.masked.load.v4i32.p0(ptr, <4 x i1>, <4 x i32>)
+declare <vscale x 4 x i32> @llvm.masked.udiv.nxv4i32(<vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i1>)
+
+define <4 x i32> @masked_udiv(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_udiv(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT: call void @llvm.donothing()
+; CHECK-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP1]], <4 x i32> zeroinitializer
+; CHECK-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT: [[TMP3:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; CHECK-NEXT: [[_MSCMP:%.*]] = icmp ne i4 [[TMP3]], 0
+; CHECK-NEXT: [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT: [[_MSCMP1:%.*]] = icmp ne i128 [[TMP4]], 0
+; CHECK-NEXT: [[_MSOR:%.*]] = or i1 [[_MSCMP]], [[_MSCMP1]]
+; CHECK-NEXT: br i1 [[_MSOR]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1:![0-9]+]]
+; CHECK: 5:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5:[0-9]+]]
+; CHECK-NEXT: unreachable
+; CHECK: 6:
+; CHECK-NEXT: [[RES:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; CHECK-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT: ret <4 x i32> [[RES]]
+;
+; ORIGINS-LABEL: @masked_udiv(
+; ORIGINS-NEXT: entry:
+; ORIGINS-NEXT: [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; ORIGINS-NEXT: [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 32), align 4
+; ORIGINS-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT: [[TMP3:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT: [[TMP4:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT: [[TMP5:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT: call void @llvm.donothing()
+; ORIGINS-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP2]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP4]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT: [[TMP6:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; ORIGINS-NEXT: [[_MSCMP:%.*]] = icmp ne i4 [[TMP6]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1:![0-9]+]]
+; ORIGINS: 7:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6:[0-9]+]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 8:
+; ORIGINS-NEXT: [[TMP9:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT: [[_MSCMP1:%.*]] = icmp ne i128 [[TMP9]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP1]], label [[TMP10:%.*]], label [[TMP11:%.*]], !prof [[PROF1]]
+; ORIGINS: 10:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP3]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 11:
+; ORIGINS-NEXT: [[RES:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; ORIGINS-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT: store i32 [[TMP5]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT: ret <4 x i32> [[RES]]
+;
+entry:
+ %res = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m)
+ ret <4 x i32> %res
+}
+
+define <4 x i32> @masked_sdiv(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_sdiv(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT: call void @llvm.donothing()
+; CHECK-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP1]], <4 x i32> zeroinitializer
+; CHECK-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT: [[TMP3:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; CHECK-NEXT: [[_MSCMP:%.*]] = icmp ne i4 [[TMP3]], 0
+; CHECK-NEXT: [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT: [[_MSCMP1:%.*]] = icmp ne i128 [[TMP4]], 0
+; CHECK-NEXT: [[_MSOR:%.*]] = or i1 [[_MSCMP]], [[_MSCMP1]]
+; CHECK-NEXT: br i1 [[_MSOR]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1]]
+; CHECK: 5:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT: unreachable
+; CHECK: 6:
+; CHECK-NEXT: [[RES:%.*]] = call <4 x i32> @llvm.masked.sdiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; CHECK-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT: ret <4 x i32> [[RES]]
+;
+; ORIGINS-LABEL: @masked_sdiv(
+; ORIGINS-NEXT: entry:
+; ORIGINS-NEXT: [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; ORIGINS-NEXT: [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 32), align 4
+; ORIGINS-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT: [[TMP3:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT: [[TMP4:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT: [[TMP5:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT: call void @llvm.donothing()
+; ORIGINS-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP2]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP4]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT: [[TMP6:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; ORIGINS-NEXT: [[_MSCMP:%.*]] = icmp ne i4 [[TMP6]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; ORIGINS: 7:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 8:
+; ORIGINS-NEXT: [[TMP9:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT: [[_MSCMP1:%.*]] = icmp ne i128 [[TMP9]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP1]], label [[TMP10:%.*]], label [[TMP11:%.*]], !prof [[PROF1]]
+; ORIGINS: 10:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP3]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 11:
+; ORIGINS-NEXT: [[RES:%.*]] = call <4 x i32> @llvm.masked.sdiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; ORIGINS-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT: store i32 [[TMP5]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT: ret <4 x i32> [[RES]]
+;
+entry:
+ %res = call <4 x i32> @llvm.masked.sdiv.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m)
+ ret <4 x i32> %res
+}
+
+define <4 x i32> @masked_urem(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_urem(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT: call void @llvm.donothing()
+; CHECK-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP1]], <4 x i32> zeroinitializer
+; CHECK-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT: [[TMP3:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; CHECK-NEXT: [[_MSCMP:%.*]] = icmp ne i4 [[TMP3]], 0
+; CHECK-NEXT: [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT: [[_MSCMP1:%.*]] = icmp ne i128 [[TMP4]], 0
+; CHECK-NEXT: [[_MSOR:%.*]] = or i1 [[_MSCMP]], [[_MSCMP1]]
+; CHECK-NEXT: br i1 [[_MSOR]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1]]
+; CHECK: 5:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT: unreachable
+; CHECK: 6:
+; CHECK-NEXT: [[RES:%.*]] = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; CHECK-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT: ret <4 x i32> [[RES]]
+;
+; ORIGINS-LABEL: @masked_urem(
+; ORIGINS-NEXT: entry:
+; ORIGINS-NEXT: [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; ORIGINS-NEXT: [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 32), align 4
+; ORIGINS-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT: [[TMP3:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT: [[TMP4:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT: [[TMP5:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT: call void @llvm.donothing()
+; ORIGINS-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP2]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP4]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT: [[TMP6:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; ORIGINS-NEXT: [[_MSCMP:%.*]] = icmp ne i4 [[TMP6]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; ORIGINS: 7:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 8:
+; ORIGINS-NEXT: [[TMP9:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT: [[_MSCMP1:%.*]] = icmp ne i128 [[TMP9]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP1]], label [[TMP10:%.*]], label [[TMP11:%.*]], !prof [[PROF1]]
+; ORIGINS: 10:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP3]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 11:
+; ORIGINS-NEXT: [[RES:%.*]] = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; ORIGINS-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT: store i32 [[TMP5]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT: ret <4 x i32> [[RES]]
+;
+entry:
+ %res = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m)
+ ret <4 x i32> %res
+}
+
+define <4 x i32> @masked_srem(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_srem(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT: call void @llvm.donothing()
+; CHECK-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP1]], <4 x i32> zeroinitializer
+; CHECK-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT: [[TMP3:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; CHECK-NEXT: [[_MSCMP:%.*]] = icmp ne i4 [[TMP3]], 0
+; CHECK-NEXT: [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT: [[_MSCMP1:%.*]] = icmp ne i128 [[TMP4]], 0
+; CHECK-NEXT: [[_MSOR:%.*]] = or i1 [[_MSCMP]], [[_MSCMP1]]
+; CHECK-NEXT: br i1 [[_MSOR]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1]]
+; CHECK: 5:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT: unreachable
+; CHECK: 6:
+; CHECK-NEXT: [[RES:%.*]] = call <4 x i32> @llvm.masked.srem.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; CHECK-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT: ret <4 x i32> [[RES]]
+;
+; ORIGINS-LABEL: @masked_srem(
+; ORIGINS-NEXT: entry:
+; ORIGINS-NEXT: [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; ORIGINS-NEXT: [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 32), align 4
+; ORIGINS-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT: [[TMP3:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT: [[TMP4:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT: [[TMP5:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT: call void @llvm.donothing()
+; ORIGINS-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP2]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP4]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT: [[TMP6:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; ORIGINS-NEXT: [[_MSCMP:%.*]] = icmp ne i4 [[TMP6]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; ORIGINS: 7:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 8:
+; ORIGINS-NEXT: [[TMP9:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT: [[_MSCMP1:%.*]] = icmp ne i128 [[TMP9]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP1]], label [[TMP10:%.*]], label [[TMP11:%.*]], !prof [[PROF1]]
+; ORIGINS: 10:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP3]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 11:
+; ORIGINS-NEXT: [[RES:%.*]] = call <4 x i32> @llvm.masked.srem.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; ORIGINS-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT: store i32 [[TMP5]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT: ret <4 x i32> [[RES]]
+;
+entry:
+ %res = call <4 x i32> @llvm.masked.srem.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m)
+ ret <4 x i32> %res
+}
+
+; Typical tail-folded vectorizer output: the masked-off lanes of the operands
+; come from masked loads with a poison pass-through, and must not be reported.
+define <4 x i32> @urem_masked_loads(ptr %pa, ptr %pb) sanitize_memory {
+; CHECK-LABEL: @urem_masked_loads(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: call void @llvm.donothing()
+; CHECK-NEXT: [[TMP0:%.*]] = ptrtoint ptr [[PA:%.*]] to i64
+; CHECK-NEXT: [[TMP1:%.*]] = xor i64 [[TMP0]], 87960930222080
+; CHECK-NEXT: [[TMP2:%.*]] = inttoptr i64 [[TMP1]] to ptr
+; CHECK-NEXT: [[_MSMASKEDLD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 1 [[TMP2]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> splat (i32 -1))
+; CHECK-NEXT: [[A:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr [[PA]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+; CHECK-NEXT: [[TMP3:%.*]] = ptrtoint ptr [[PB:%.*]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = xor i64 [[TMP3]], 87960930222080
+; CHECK-NEXT: [[TMP5:%.*]] = inttoptr i64 [[TMP4]] to ptr
+; CHECK-NEXT: [[_MSMASKEDLD1:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 1 [[TMP5]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> splat (i32 -1))
+; CHECK-NEXT: [[B:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr [[PB]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+; CHECK-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[_MSMASKEDLD1]], <4 x i32> zeroinitializer
+; CHECK-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[_MSMASKEDLD]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT: [[TMP6:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT: [[_MSCMP:%.*]] = icmp ne i128 [[TMP6]], 0
+; CHECK-NEXT: br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; CHECK: 7:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT: unreachable
+; CHECK: 8:
+; CHECK-NEXT: [[R:%.*]] = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> [[A]], <4 x i32> [[B]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+; CHECK-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT: ret <4 x i32> [[R]]
+;
+; ORIGINS-LABEL: @urem_masked_loads(
+; ORIGINS-NEXT: entry:
+; ORIGINS-NEXT: call void @llvm.donothing()
+; ORIGINS-NEXT: [[TMP0:%.*]] = ptrtoint ptr [[PA:%.*]] to i64
+; ORIGINS-NEXT: [[TMP1:%.*]] = xor i64 [[TMP0]], 87960930222080
+; ORIGINS-NEXT: [[TMP2:%.*]] = inttoptr i64 [[TMP1]] to ptr
+; ORIGINS-NEXT: [[TMP3:%.*]] = add i64 [[TMP1]], 17592186044416
+; ORIGINS-NEXT: [[TMP4:%.*]] = and i64 [[TMP3]], -4
+; ORIGINS-NEXT: [[TMP5:%.*]] = inttoptr i64 [[TMP4]] to ptr
+; ORIGINS-NEXT: [[_MSMASKEDLD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 1 [[TMP2]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> splat (i32 -1))
+; ORIGINS-NEXT: [[_MSCMP:%.*]] = icmp ne i128 bitcast (<4 x i32> <i32 -1, i32 -1, i32 -1, i32 0> to i128), 0
+; ORIGINS-NEXT: [[TMP6:%.*]] = load i32, ptr [[TMP5]], align 4
+; ORIGINS-NEXT: [[TMP7:%.*]] = select i1 [[_MSCMP]], i32 0, i32 [[TMP6]]
+; ORIGINS-NEXT: [[A:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr [[PA]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+; ORIGINS-NEXT: [[TMP8:%.*]] = ptrtoint ptr [[PB:%.*]] to i64
+; ORIGINS-NEXT: [[TMP9:%.*]] = xor i64 [[TMP8]], 87960930222080
+; ORIGINS-NEXT: [[TMP10:%.*]] = inttoptr i64 [[TMP9]] to ptr
+; ORIGINS-NEXT: [[TMP11:%.*]] = add i64 [[TMP9]], 17592186044416
+; ORIGINS-NEXT: [[TMP12:%.*]] = and i64 [[TMP11]], -4
+; ORIGINS-NEXT: [[TMP13:%.*]] = inttoptr i64 [[TMP12]] to ptr
+; ORIGINS-NEXT: [[_MSMASKEDLD1:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 1 [[TMP10]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> splat (i32 -1))
+; ORIGINS-NEXT: [[_MSCMP2:%.*]] = icmp ne i128 bitcast (<4 x i32> <i32 -1, i32 -1, i32 -1, i32 0> to i128), 0
+; ORIGINS-NEXT: [[TMP14:%.*]] = load i32, ptr [[TMP13]], align 4
+; ORIGINS-NEXT: [[TMP15:%.*]] = select i1 [[_MSCMP2]], i32 0, i32 [[TMP14]]
+; ORIGINS-NEXT: [[B:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr [[PB]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+; ORIGINS-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[_MSMASKEDLD1]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[_MSMASKEDLD]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT: [[TMP16:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT: [[_MSCMP3:%.*]] = icmp ne i128 [[TMP16]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP3]], label [[TMP17:%.*]], label [[TMP18:%.*]], !prof [[PROF1]]
+; ORIGINS: 17:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP15]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 18:
+; ORIGINS-NEXT: [[R:%.*]] = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> [[A]], <4 x i32> [[B]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+; ORIGINS-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT: store i32 [[TMP7]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT: ret <4 x i32> [[R]]
+;
+entry:
+ %a = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr %pa, <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+ %b = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr %pb, <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+ %r = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> %a, <4 x i32> %b, <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+ ret <4 x i32> %r
+}
+
+; Even when the dividend is clean across all lanes (e.g. a splat constant),
+; disabled lanes of llvm.masked.udiv produce poison and must carry a poisoned
+; shadow so that an unmasked read of a disabled lane is caught.
+define i1 @udiv_clean_dividend_poisons_disabled_lanes(<4 x i32> %b) sanitize_memory {
+; CHECK-LABEL: @udiv_clean_dividend_poisons_disabled_lanes(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT: call void @llvm.donothing()
+; CHECK-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[TMP0]], <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP1:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT: [[_MSCMP:%.*]] = icmp ne i128 [[TMP1]], 0
+; CHECK-NEXT: br i1 [[_MSCMP]], label [[TMP2:%.*]], label [[TMP3:%.*]], !prof [[PROF1]]
+; CHECK: 2:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT: unreachable
+; CHECK: 3:
+; CHECK-NEXT: [[R:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> splat (i32 100), <4 x i32> [[B:%.*]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+; CHECK-NEXT: [[LANE3:%.*]] = extractelement <4 x i32> [[R]], i32 3
+; CHECK-NEXT: [[TMP4:%.*]] = xor i32 [[LANE3]], 0
+; CHECK-NEXT: [[TMP5:%.*]] = and i32 0, [[TMP4]]
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i32 [[TMP5]], 0
+; CHECK-NEXT: [[_MSPROP_ICMP:%.*]] = and i1 true, [[TMP6]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i32 [[LANE3]], 0
+; CHECK-NEXT: br i1 [[_MSPROP_ICMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; CHECK: 7:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT: unreachable
+; CHECK: 8:
+; CHECK-NEXT: br i1 [[CMP]], label [[T:%.*]], label [[F:%.*]]
+; CHECK: t:
+; CHECK-NEXT: store i1 false, ptr @__msan_retval_tls, align 8
+; CHECK-NEXT: ret i1 true
+; CHECK: f:
+; CHECK-NEXT: store i1 false, ptr @__msan_retval_tls, align 8
+; CHECK-NEXT: ret i1 false
+;
+; ORIGINS-LABEL: @udiv_clean_dividend_poisons_disabled_lanes(
+; ORIGINS-NEXT: entry:
+; ORIGINS-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT: [[TMP1:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT: call void @llvm.donothing()
+; ORIGINS-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[TMP0]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT: [[TMP2:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT: [[_MSCMP:%.*]] = icmp ne i128 [[TMP2]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP]], label [[TMP3:%.*]], label [[TMP4:%.*]], !prof [[PROF1]]
+; ORIGINS: 3:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 4:
+; ORIGINS-NEXT: [[R:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> splat (i32 100), <4 x i32> [[B:%.*]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+; ORIGINS-NEXT: [[LANE3:%.*]] = extractelement <4 x i32> [[R]], i32 3
+; ORIGINS-NEXT: [[TMP5:%.*]] = xor i32 [[LANE3]], 0
+; ORIGINS-NEXT: [[TMP6:%.*]] = and i32 0, [[TMP5]]
+; ORIGINS-NEXT: [[TMP7:%.*]] = icmp eq i32 [[TMP6]], 0
+; ORIGINS-NEXT: [[_MSPROP_ICMP:%.*]] = and i1 true, [[TMP7]]
+; ORIGINS-NEXT: [[CMP:%.*]] = icmp eq i32 [[LANE3]], 0
+; ORIGINS-NEXT: br i1 [[_MSPROP_ICMP]], label [[TMP8:%.*]], label [[TMP9:%.*]], !prof [[PROF1]]
+; ORIGINS: 8:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 0) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 9:
+; ORIGINS-NEXT: br i1 [[CMP]], label [[T:%.*]], label [[F:%.*]]
+; ORIGINS: t:
+; ORIGINS-NEXT: store i1 false, ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT: store i32 0, ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT: ret i1 true
+; ORIGINS: f:
+; ORIGINS-NEXT: store i1 false, ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT: store i32 0, ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT: ret i1 false
+;
+entry:
+ %r = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> splat (i32 100), <4 x i32> %b, <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+ %lane3 = extractelement <4 x i32> %r, i32 3
+ %cmp = icmp eq i32 %lane3, 0
+ br i1 %cmp, label %t, label %f
+t:
+ ret i1 true
+f:
+ ret i1 false
+}
+
+; With every lane enabled there is nothing to poison, and the divisor is
+; checked in full, exactly as for a plain udiv.
+define <4 x i32> @udiv_all_lanes_enabled(<4 x i32> %x, <4 x i32> %y) sanitize_memory {
+; CHECK-LABEL: @udiv_all_lanes_enabled(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT: call void @llvm.donothing()
+; CHECK-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> splat (i1 true), <4 x i32> [[TMP0]], <4 x i32> zeroinitializer
+; CHECK-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> splat (i1 true), <4 x i32> [[TMP1]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT: [[TMP2:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT: [[_MSCMP:%.*]] = icmp ne i128 [[TMP2]], 0
+; CHECK-NEXT: br i1 [[_MSCMP]], label [[TMP3:%.*]], label [[TMP4:%.*]], !prof [[PROF1]]
+; CHECK: 3:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT: unreachable
+; CHECK: 4:
+; CHECK-NEXT: [[R:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> splat (i1 true))
+; CHECK-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT: ret <4 x i32> [[R]]
+;
+; ORIGINS-LABEL: @udiv_all_lanes_enabled(
+; ORIGINS-NEXT: entry:
+; ORIGINS-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT: [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT: [[TMP3:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT: call void @llvm.donothing()
+; ORIGINS-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> splat (i1 true), <4 x i32> [[TMP0]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT: [[_MSMASKEDDIV:%.*]] = select <4 x i1> splat (i1 true), <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT: [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT: [[_MSCMP:%.*]] = icmp ne i128 [[TMP4]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1]]
+; ORIGINS: 5:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 6:
+; ORIGINS-NEXT: [[R:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> splat (i1 true))
+; ORIGINS-NEXT: store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT: store i32 [[TMP3]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT: ret <4 x i32> [[R]]
+;
+entry:
+ %r = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> <i1 true, i1 true, i1 true, i1 true>)
+ ret <4 x i32> %r
+}
+
+; Scalable vectors: the disabled lanes get a poison shadow splat, and the
+; divisor check goes through a vector reduction rather than a bitcast.
+define void @masked_udiv_scalable(ptr %pa, ptr %pb, ptr %pr, <vscale x 4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_udiv_scalable(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: call void @llvm.donothing()
+; CHECK-NEXT: [[A:%.*]] = load <vscale x 4 x i32>, ptr [[PA:%.*]], align 16
+; CHECK-NEXT: [[TMP0:%.*]] = ptrtoint ptr [[PA]] to i64
+; CHECK-NEXT: [[TMP1:%.*]] = xor i64 [[TMP0]], 87960930222080
+; CHECK-NEXT: [[TMP2:%.*]] = inttoptr i64 [[TMP1]] to ptr
+; CHECK-NEXT: [[_MSLD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 16
+; CHECK-NEXT: [[B:%.*]] = load <vscale x 4 x i32>, ptr [[PB:%.*]], align 16
+; CHECK-NEXT: [[TMP3:%.*]] = ptrtoint ptr [[PB]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = xor i64 [[TMP3]], 87960930222080
+; CHECK-NEXT: [[TMP5:%.*]] = inttoptr i64 [[TMP4]] to ptr
+; CHECK-NEXT: [[_MSLD1:%.*]] = load <vscale x 4 x i32>, ptr [[TMP5]], align 16
+; CHECK-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <vscale x 4 x i1> [[M:%.*]], <vscale x 4 x i32> [[_MSLD1]], <vscale x 4 x i32> zeroinitializer
+; CHECK-NEXT: [[_MSMASKEDDIV:%.*]] = select <vscale x 4 x i1> [[M]], <vscale x 4 x i32> [[_MSLD]], <vscale x 4 x i32> splat (i32 -1)
+; CHECK-NEXT: [[TMP6:%.*]] = call i32 @llvm.vector.reduce.or.nxv4i32(<vscale x 4 x i32> [[_MSMASKEDDIVISOR]])
+; CHECK-NEXT: [[_MSCMP:%.*]] = icmp ne i32 [[TMP6]], 0
+; CHECK-NEXT: br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; CHECK: 7:
+; CHECK-NEXT: call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT: unreachable
+; CHECK: 8:
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i32> @llvm.masked.udiv.nxv4i32(<vscale x 4 x i32> [[A]], <vscale x 4 x i32> [[B]], <vscale x 4 x i1> [[M]])
+; CHECK-NEXT: [[TMP9:%.*]] = ptrtoint ptr [[PR:%.*]] to i64
+; CHECK-NEXT: [[TMP10:%.*]] = xor i64 [[TMP9]], 87960930222080
+; CHECK-NEXT: [[TMP11:%.*]] = inttoptr i64 [[TMP10]] to ptr
+; CHECK-NEXT: store <vscale x 4 x i32> [[_MSMASKEDDIV]], ptr [[TMP11]], align 16
+; CHECK-NEXT: store <vscale x 4 x i32> [[R]], ptr [[PR]], align 16
+; CHECK-NEXT: ret void
+;
+; ORIGINS-LABEL: @masked_udiv_scalable(
+; ORIGINS-NEXT: entry:
+; ORIGINS-NEXT: call void @llvm.donothing()
+; ORIGINS-NEXT: [[A:%.*]] = load <vscale x 4 x i32>, ptr [[PA:%.*]], align 16
+; ORIGINS-NEXT: [[TMP0:%.*]] = ptrtoint ptr [[PA]] to i64
+; ORIGINS-NEXT: [[TMP1:%.*]] = xor i64 [[TMP0]], 87960930222080
+; ORIGINS-NEXT: [[TMP2:%.*]] = inttoptr i64 [[TMP1]] to ptr
+; ORIGINS-NEXT: [[TMP3:%.*]] = add i64 [[TMP1]], 17592186044416
+; ORIGINS-NEXT: [[TMP4:%.*]] = inttoptr i64 [[TMP3]] to ptr
+; ORIGINS-NEXT: [[_MSLD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 16
+; ORIGINS-NEXT: [[TMP5:%.*]] = load i32, ptr [[TMP4]], align 16
+; ORIGINS-NEXT: [[B:%.*]] = load <vscale x 4 x i32>, ptr [[PB:%.*]], align 16
+; ORIGINS-NEXT: [[TMP6:%.*]] = ptrtoint ptr [[PB]] to i64
+; ORIGINS-NEXT: [[TMP7:%.*]] = xor i64 [[TMP6]], 87960930222080
+; ORIGINS-NEXT: [[TMP8:%.*]] = inttoptr i64 [[TMP7]] to ptr
+; ORIGINS-NEXT: [[TMP9:%.*]] = add i64 [[TMP7]], 17592186044416
+; ORIGINS-NEXT: [[TMP10:%.*]] = inttoptr i64 [[TMP9]] to ptr
+; ORIGINS-NEXT: [[_MSLD1:%.*]] = load <vscale x 4 x i32>, ptr [[TMP8]], align 16
+; ORIGINS-NEXT: [[TMP11:%.*]] = load i32, ptr [[TMP10]], align 16
+; ORIGINS-NEXT: [[_MSMASKEDDIVISOR:%.*]] = select <vscale x 4 x i1> [[M:%.*]], <vscale x 4 x i32> [[_MSLD1]], <vscale x 4 x i32> zeroinitializer
+; ORIGINS-NEXT: [[_MSMASKEDDIV:%.*]] = select <vscale x 4 x i1> [[M]], <vscale x 4 x i32> [[_MSLD]], <vscale x 4 x i32> splat (i32 -1)
+; ORIGINS-NEXT: [[TMP12:%.*]] = call i32 @llvm.vector.reduce.or.nxv4i32(<vscale x 4 x i32> [[_MSMASKEDDIVISOR]])
+; ORIGINS-NEXT: [[_MSCMP:%.*]] = icmp ne i32 [[TMP12]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP]], label [[TMP13:%.*]], label [[TMP14:%.*]], !prof [[PROF1]]
+; ORIGINS: 13:
+; ORIGINS-NEXT: call void @__msan_warning_with_origin_noreturn(i32 [[TMP11]]) #[[ATTR6]]
+; ORIGINS-NEXT: unreachable
+; ORIGINS: 14:
+; ORIGINS-NEXT: [[R:%.*]] = call <vscale x 4 x i32> @llvm.masked.udiv.nxv4i32(<vscale x 4 x i32> [[A]], <vscale x 4 x i32> [[B]], <vscale x 4 x i1> [[M]])
+; ORIGINS-NEXT: [[TMP15:%.*]] = ptrtoint ptr [[PR:%.*]] to i64
+; ORIGINS-NEXT: [[TMP16:%.*]] = xor i64 [[TMP15]], 87960930222080
+; ORIGINS-NEXT: [[TMP17:%.*]] = inttoptr i64 [[TMP16]] to ptr
+; ORIGINS-NEXT: [[TMP18:%.*]] = add i64 [[TMP16]], 17592186044416
+; ORIGINS-NEXT: [[TMP19:%.*]] = inttoptr i64 [[TMP18]] to ptr
+; ORIGINS-NEXT: store <vscale x 4 x i32> [[_MSMASKEDDIV]], ptr [[TMP17]], align 16
+; ORIGINS-NEXT: [[TMP20:%.*]] = call i32 @llvm.vector.reduce.or.nxv4i32(<vscale x 4 x i32> [[_MSMASKEDDIV]])
+; ORIGINS-NEXT: [[_MSCMP2:%.*]] = icmp ne i32 [[TMP20]], 0
+; ORIGINS-NEXT: br i1 [[_MSCMP2]], label [[TMP21:%.*]], label [[TMP27:%.*]], !prof [[PROF1]]
+; ORIGINS: 21:
+; ORIGINS-NEXT: [[TMP22:%.*]] = call i64 @llvm.vscale.i64()
+; ORIGINS-NEXT: [[TMP23:%.*]] = mul nuw i64 [[TMP22]], 16
+; ORIGINS-NEXT: [[TMP24:%.*]] = add i64 [[TMP23]], 3
+; ORIGINS-NEXT: [[TMP25:%.*]] = udiv i64 [[TMP24]], 4
+; ORIGINS-NEXT: br label [[DOTSPLIT:%.*]]
+; ORIGINS: .split:
+; ORIGINS-NEXT: [[IV:%.*]] = phi i64 [ 0, [[TMP21]] ], [ [[IV_NEXT:%.*]], [[DOTSPLIT]] ]
+; ORIGINS-NEXT: [[TMP26:%.*]] = getelementptr i32, ptr [[TMP19]], i64 [[IV]]
+; ORIGINS-NEXT: store i32 [[TMP5]], ptr [[TMP26]], align 4
+; ORIGINS-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; ORIGINS-NEXT: [[IV_CHECK:%.*]] = icmp eq i64 [[IV_NEXT]], [[TMP25]]
+; ORIGINS-NEXT: br i1 [[IV_CHECK]], label [[DOTSPLIT_SPLIT:%.*]], label [[DOTSPLIT]]
+; ORIGINS: .split.split:
+; ORIGINS-NEXT: br label [[TMP27]]
+; ORIGINS: 27:
+; ORIGINS-NEXT: store <vscale x 4 x i32> [[R]], ptr [[PR]], align 16
+; ORIGINS-NEXT: ret void
+;
+entry:
+ %a = load <vscale x 4 x i32>, ptr %pa
+ %b = load <vscale x 4 x i32>, ptr %pb
+ %r = call <vscale x 4 x i32> @llvm.masked.udiv.nxv4i32(<vscale x 4 x i32> %a, <vscale x 4 x i32> %b, <vscale x 4 x i1> %m)
+ store <vscale x 4 x i32> %r, ptr %pr
+ ret void
+}
More information about the llvm-commits
mailing list