[llvm] [msan] handle llvm.masked.{udiv, sdiv, urem, srem} intrinsics (PR #225363)

Emilio Cota via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 08:53:33 PDT 2026


https://github.com/cota updated https://github.com/llvm/llvm-project/pull/225363

>From fe69a81f353d850c40b167f250029073deff5dfc Mon Sep 17 00:00:00 2001
From: Emilio Cota <ecg at google.com>
Date: Tue, 22 Sep 2026 10:30:08 +0000
Subject: [PATCH] [msan] handle llvm.masked.{udiv,sdiv,urem,srem} intrinsics

Check the divisor and propagate the dividend's shadow on active lanes,
poisoning disabled result lanes. This poisoning is consistent with
the definition of these intrinsics; see #189705 (9ba774566).

This gets rid of false positives from inactive input lanes since
visitInstruction() was checking every operand across all lanes,
active or not.
---
 .../Instrumentation/MemorySanitizer.cpp       |  36 ++
 .../MemorySanitizer/masked-divrem.ll          | 547 ++++++++++++++++++
 2 files changed, 583 insertions(+)
 create mode 100644 llvm/test/Instrumentation/MemorySanitizer/masked-divrem.ll

diff --git a/llvm/lib/Transforms/Instrumentation/MemorySanitizer.cpp b/llvm/lib/Transforms/Instrumentation/MemorySanitizer.cpp
index 4084f3d580e57a..0120b27fb13c51 100644
--- a/llvm/lib/Transforms/Instrumentation/MemorySanitizer.cpp
+++ b/llvm/lib/Transforms/Instrumentation/MemorySanitizer.cpp
@@ -4630,6 +4630,36 @@ struct MemorySanitizerVisitor : public InstVisitor<MemorySanitizerVisitor> {
     setOrigin(&I, Origin);
   }
 
+  // e.g., <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> %dividend,
+  //                                         <4 x i32> %divisor,
+  //                                         <4 x i1>  %mask)
+  //
+  // As handleIntegerDiv(), but per-lane: strict on the divisor and propagating
+  // the dividend, both only on the enabled lanes. Disabled lanes cannot cause
+  // undefined behaviour, and their result is poison.
+  void handleMaskedIntegerDivRem(IntrinsicInst &I) {
+    IRBuilder<> IRB(&I);
+    Value *Dividend = I.getArgOperand(0);
+    Value *Divisor = I.getArgOperand(1);
+    Value *Mask = I.getArgOperand(2);
+
+    insertCheckShadowOf(Mask, &I);
+
+    Value *MaskedDivisorShadow = IRB.CreateSelect(
+        Mask, getShadow(Divisor), getCleanShadow(Divisor), "_msmaskeddivisor");
+    insertCheckShadow(MaskedDivisorShadow, getOrigin(Divisor), &I);
+
+    if (!PropagateShadow) {
+      setShadow(&I, getCleanShadow(&I));
+      setOrigin(&I, getCleanOrigin());
+      return;
+    }
+
+    setShadow(&I, IRB.CreateSelect(Mask, getShadow(Dividend),
+                                   getPoisonedShadow(&I), "_msmaskeddiv"));
+    setOrigin(&I, getOrigin(Dividend));
+  }
+
   // e.g., void @llvm.x86.avx.maskstore.ps.256(ptr, <8 x i32>, <8 x float>)
   //                                           dst  mask       src
   //
@@ -5937,6 +5967,12 @@ struct MemorySanitizerVisitor : public InstVisitor<MemorySanitizerVisitor> {
     case Intrinsic::masked_load:
       handleMaskedLoad(I);
       break;
+    case Intrinsic::masked_udiv:
+    case Intrinsic::masked_sdiv:
+    case Intrinsic::masked_urem:
+    case Intrinsic::masked_srem:
+      handleMaskedIntegerDivRem(I);
+      break;
     case Intrinsic::vector_reduce_and:
       handleVectorReduceAndIntrinsic(I);
       break;
diff --git a/llvm/test/Instrumentation/MemorySanitizer/masked-divrem.ll b/llvm/test/Instrumentation/MemorySanitizer/masked-divrem.ll
new file mode 100644
index 00000000000000..28966688eda8b3
--- /dev/null
+++ b/llvm/test/Instrumentation/MemorySanitizer/masked-divrem.ll
@@ -0,0 +1,547 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -msan-check-access-address=0 -S -passes=msan 2>&1 | FileCheck %s --implicit-check-not="call void @__msan_warning"
+; RUN: opt < %s -msan-check-access-address=0 -msan-track-origins=1 -S -passes=msan 2>&1 | FileCheck %s --check-prefixes=ORIGINS --implicit-check-not="call void @__msan_warning"
+
+target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f32:32:32-f64:64:64-v64:64:64-v128:128:128-a0:0:64-s0:64:64-f80:128:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+declare <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32>, <4 x i32>, <4 x i1>)
+declare <4 x i32> @llvm.masked.sdiv.v4i32(<4 x i32>, <4 x i32>, <4 x i1>)
+declare <4 x i32> @llvm.masked.urem.v4i32(<4 x i32>, <4 x i32>, <4 x i1>)
+declare <4 x i32> @llvm.masked.srem.v4i32(<4 x i32>, <4 x i32>, <4 x i1>)
+declare <4 x i32> @llvm.masked.load.v4i32.p0(ptr, <4 x i1>, <4 x i32>)
+declare <vscale x 4 x i32> @llvm.masked.udiv.nxv4i32(<vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i1>)
+
+define <4 x i32> @masked_udiv(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_udiv(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT:    call void @llvm.donothing()
+; CHECK-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP1]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; CHECK-NEXT:    [[_MSCMP:%.*]] = icmp ne i4 [[TMP3]], 0
+; CHECK-NEXT:    [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT:    [[_MSCMP1:%.*]] = icmp ne i128 [[TMP4]], 0
+; CHECK-NEXT:    [[_MSOR:%.*]] = or i1 [[_MSCMP]], [[_MSCMP1]]
+; CHECK-NEXT:    br i1 [[_MSOR]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1:![0-9]+]]
+; CHECK:       5:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5:[0-9]+]]
+; CHECK-NEXT:    unreachable
+; CHECK:       6:
+; CHECK-NEXT:    [[RES:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; CHECK-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT:    ret <4 x i32> [[RES]]
+;
+; ORIGINS-LABEL: @masked_udiv(
+; ORIGINS-NEXT:  entry:
+; ORIGINS-NEXT:    [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; ORIGINS-NEXT:    [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 32), align 4
+; ORIGINS-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT:    [[TMP3:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT:    [[TMP4:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT:    [[TMP5:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT:    call void @llvm.donothing()
+; ORIGINS-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP2]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP4]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT:    [[TMP6:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; ORIGINS-NEXT:    [[_MSCMP:%.*]] = icmp ne i4 [[TMP6]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1:![0-9]+]]
+; ORIGINS:       7:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6:[0-9]+]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       8:
+; ORIGINS-NEXT:    [[TMP9:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT:    [[_MSCMP1:%.*]] = icmp ne i128 [[TMP9]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP1]], label [[TMP10:%.*]], label [[TMP11:%.*]], !prof [[PROF1]]
+; ORIGINS:       10:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP3]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       11:
+; ORIGINS-NEXT:    [[RES:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; ORIGINS-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT:    store i32 [[TMP5]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT:    ret <4 x i32> [[RES]]
+;
+entry:
+  %res = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m)
+  ret <4 x i32> %res
+}
+
+define <4 x i32> @masked_sdiv(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_sdiv(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT:    call void @llvm.donothing()
+; CHECK-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP1]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; CHECK-NEXT:    [[_MSCMP:%.*]] = icmp ne i4 [[TMP3]], 0
+; CHECK-NEXT:    [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT:    [[_MSCMP1:%.*]] = icmp ne i128 [[TMP4]], 0
+; CHECK-NEXT:    [[_MSOR:%.*]] = or i1 [[_MSCMP]], [[_MSCMP1]]
+; CHECK-NEXT:    br i1 [[_MSOR]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1]]
+; CHECK:       5:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT:    unreachable
+; CHECK:       6:
+; CHECK-NEXT:    [[RES:%.*]] = call <4 x i32> @llvm.masked.sdiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; CHECK-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT:    ret <4 x i32> [[RES]]
+;
+; ORIGINS-LABEL: @masked_sdiv(
+; ORIGINS-NEXT:  entry:
+; ORIGINS-NEXT:    [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; ORIGINS-NEXT:    [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 32), align 4
+; ORIGINS-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT:    [[TMP3:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT:    [[TMP4:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT:    [[TMP5:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT:    call void @llvm.donothing()
+; ORIGINS-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP2]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP4]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT:    [[TMP6:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; ORIGINS-NEXT:    [[_MSCMP:%.*]] = icmp ne i4 [[TMP6]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; ORIGINS:       7:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       8:
+; ORIGINS-NEXT:    [[TMP9:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT:    [[_MSCMP1:%.*]] = icmp ne i128 [[TMP9]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP1]], label [[TMP10:%.*]], label [[TMP11:%.*]], !prof [[PROF1]]
+; ORIGINS:       10:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP3]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       11:
+; ORIGINS-NEXT:    [[RES:%.*]] = call <4 x i32> @llvm.masked.sdiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; ORIGINS-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT:    store i32 [[TMP5]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT:    ret <4 x i32> [[RES]]
+;
+entry:
+  %res = call <4 x i32> @llvm.masked.sdiv.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m)
+  ret <4 x i32> %res
+}
+
+define <4 x i32> @masked_urem(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_urem(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT:    call void @llvm.donothing()
+; CHECK-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP1]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; CHECK-NEXT:    [[_MSCMP:%.*]] = icmp ne i4 [[TMP3]], 0
+; CHECK-NEXT:    [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT:    [[_MSCMP1:%.*]] = icmp ne i128 [[TMP4]], 0
+; CHECK-NEXT:    [[_MSOR:%.*]] = or i1 [[_MSCMP]], [[_MSCMP1]]
+; CHECK-NEXT:    br i1 [[_MSOR]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1]]
+; CHECK:       5:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT:    unreachable
+; CHECK:       6:
+; CHECK-NEXT:    [[RES:%.*]] = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; CHECK-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT:    ret <4 x i32> [[RES]]
+;
+; ORIGINS-LABEL: @masked_urem(
+; ORIGINS-NEXT:  entry:
+; ORIGINS-NEXT:    [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; ORIGINS-NEXT:    [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 32), align 4
+; ORIGINS-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT:    [[TMP3:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT:    [[TMP4:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT:    [[TMP5:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT:    call void @llvm.donothing()
+; ORIGINS-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP2]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP4]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT:    [[TMP6:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; ORIGINS-NEXT:    [[_MSCMP:%.*]] = icmp ne i4 [[TMP6]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; ORIGINS:       7:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       8:
+; ORIGINS-NEXT:    [[TMP9:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT:    [[_MSCMP1:%.*]] = icmp ne i128 [[TMP9]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP1]], label [[TMP10:%.*]], label [[TMP11:%.*]], !prof [[PROF1]]
+; ORIGINS:       10:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP3]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       11:
+; ORIGINS-NEXT:    [[RES:%.*]] = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; ORIGINS-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT:    store i32 [[TMP5]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT:    ret <4 x i32> [[RES]]
+;
+entry:
+  %res = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m)
+  ret <4 x i32> %res
+}
+
+define <4 x i32> @masked_srem(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_srem(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT:    call void @llvm.donothing()
+; CHECK-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP1]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; CHECK-NEXT:    [[_MSCMP:%.*]] = icmp ne i4 [[TMP3]], 0
+; CHECK-NEXT:    [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT:    [[_MSCMP1:%.*]] = icmp ne i128 [[TMP4]], 0
+; CHECK-NEXT:    [[_MSOR:%.*]] = or i1 [[_MSCMP]], [[_MSCMP1]]
+; CHECK-NEXT:    br i1 [[_MSOR]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1]]
+; CHECK:       5:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT:    unreachable
+; CHECK:       6:
+; CHECK-NEXT:    [[RES:%.*]] = call <4 x i32> @llvm.masked.srem.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; CHECK-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT:    ret <4 x i32> [[RES]]
+;
+; ORIGINS-LABEL: @masked_srem(
+; ORIGINS-NEXT:  entry:
+; ORIGINS-NEXT:    [[TMP0:%.*]] = load <4 x i1>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 32), align 8
+; ORIGINS-NEXT:    [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 32), align 4
+; ORIGINS-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT:    [[TMP3:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT:    [[TMP4:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT:    [[TMP5:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT:    call void @llvm.donothing()
+; ORIGINS-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> [[M:%.*]], <4 x i32> [[TMP2]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> [[M]], <4 x i32> [[TMP4]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT:    [[TMP6:%.*]] = bitcast <4 x i1> [[TMP0]] to i4
+; ORIGINS-NEXT:    [[_MSCMP:%.*]] = icmp ne i4 [[TMP6]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; ORIGINS:       7:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       8:
+; ORIGINS-NEXT:    [[TMP9:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT:    [[_MSCMP1:%.*]] = icmp ne i128 [[TMP9]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP1]], label [[TMP10:%.*]], label [[TMP11:%.*]], !prof [[PROF1]]
+; ORIGINS:       10:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP3]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       11:
+; ORIGINS-NEXT:    [[RES:%.*]] = call <4 x i32> @llvm.masked.srem.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> [[M]])
+; ORIGINS-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT:    store i32 [[TMP5]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT:    ret <4 x i32> [[RES]]
+;
+entry:
+  %res = call <4 x i32> @llvm.masked.srem.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> %m)
+  ret <4 x i32> %res
+}
+
+; Typical tail-folded vectorizer output: the masked-off lanes of the operands
+; come from masked loads with a poison pass-through, and must not be reported.
+define <4 x i32> @urem_masked_loads(ptr %pa, ptr %pb) sanitize_memory {
+; CHECK-LABEL: @urem_masked_loads(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    call void @llvm.donothing()
+; CHECK-NEXT:    [[TMP0:%.*]] = ptrtoint ptr [[PA:%.*]] to i64
+; CHECK-NEXT:    [[TMP1:%.*]] = xor i64 [[TMP0]], 87960930222080
+; CHECK-NEXT:    [[TMP2:%.*]] = inttoptr i64 [[TMP1]] to ptr
+; CHECK-NEXT:    [[_MSMASKEDLD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 1 [[TMP2]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> splat (i32 -1))
+; CHECK-NEXT:    [[A:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr [[PA]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+; CHECK-NEXT:    [[TMP3:%.*]] = ptrtoint ptr [[PB:%.*]] to i64
+; CHECK-NEXT:    [[TMP4:%.*]] = xor i64 [[TMP3]], 87960930222080
+; CHECK-NEXT:    [[TMP5:%.*]] = inttoptr i64 [[TMP4]] to ptr
+; CHECK-NEXT:    [[_MSMASKEDLD1:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 1 [[TMP5]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> splat (i32 -1))
+; CHECK-NEXT:    [[B:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr [[PB]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+; CHECK-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[_MSMASKEDLD1]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[_MSMASKEDLD]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT:    [[TMP6:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT:    [[_MSCMP:%.*]] = icmp ne i128 [[TMP6]], 0
+; CHECK-NEXT:    br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; CHECK:       7:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT:    unreachable
+; CHECK:       8:
+; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> [[A]], <4 x i32> [[B]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+; CHECK-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT:    ret <4 x i32> [[R]]
+;
+; ORIGINS-LABEL: @urem_masked_loads(
+; ORIGINS-NEXT:  entry:
+; ORIGINS-NEXT:    call void @llvm.donothing()
+; ORIGINS-NEXT:    [[TMP0:%.*]] = ptrtoint ptr [[PA:%.*]] to i64
+; ORIGINS-NEXT:    [[TMP1:%.*]] = xor i64 [[TMP0]], 87960930222080
+; ORIGINS-NEXT:    [[TMP2:%.*]] = inttoptr i64 [[TMP1]] to ptr
+; ORIGINS-NEXT:    [[TMP3:%.*]] = add i64 [[TMP1]], 17592186044416
+; ORIGINS-NEXT:    [[TMP4:%.*]] = and i64 [[TMP3]], -4
+; ORIGINS-NEXT:    [[TMP5:%.*]] = inttoptr i64 [[TMP4]] to ptr
+; ORIGINS-NEXT:    [[_MSMASKEDLD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 1 [[TMP2]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> splat (i32 -1))
+; ORIGINS-NEXT:    [[_MSCMP:%.*]] = icmp ne i128 bitcast (<4 x i32> <i32 -1, i32 -1, i32 -1, i32 0> to i128), 0
+; ORIGINS-NEXT:    [[TMP6:%.*]] = load i32, ptr [[TMP5]], align 4
+; ORIGINS-NEXT:    [[TMP7:%.*]] = select i1 [[_MSCMP]], i32 0, i32 [[TMP6]]
+; ORIGINS-NEXT:    [[A:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr [[PA]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+; ORIGINS-NEXT:    [[TMP8:%.*]] = ptrtoint ptr [[PB:%.*]] to i64
+; ORIGINS-NEXT:    [[TMP9:%.*]] = xor i64 [[TMP8]], 87960930222080
+; ORIGINS-NEXT:    [[TMP10:%.*]] = inttoptr i64 [[TMP9]] to ptr
+; ORIGINS-NEXT:    [[TMP11:%.*]] = add i64 [[TMP9]], 17592186044416
+; ORIGINS-NEXT:    [[TMP12:%.*]] = and i64 [[TMP11]], -4
+; ORIGINS-NEXT:    [[TMP13:%.*]] = inttoptr i64 [[TMP12]] to ptr
+; ORIGINS-NEXT:    [[_MSMASKEDLD1:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 1 [[TMP10]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> splat (i32 -1))
+; ORIGINS-NEXT:    [[_MSCMP2:%.*]] = icmp ne i128 bitcast (<4 x i32> <i32 -1, i32 -1, i32 -1, i32 0> to i128), 0
+; ORIGINS-NEXT:    [[TMP14:%.*]] = load i32, ptr [[TMP13]], align 4
+; ORIGINS-NEXT:    [[TMP15:%.*]] = select i1 [[_MSCMP2]], i32 0, i32 [[TMP14]]
+; ORIGINS-NEXT:    [[B:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr [[PB]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+; ORIGINS-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[_MSMASKEDLD1]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[_MSMASKEDLD]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT:    [[TMP16:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT:    [[_MSCMP3:%.*]] = icmp ne i128 [[TMP16]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP3]], label [[TMP17:%.*]], label [[TMP18:%.*]], !prof [[PROF1]]
+; ORIGINS:       17:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP15]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       18:
+; ORIGINS-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> [[A]], <4 x i32> [[B]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+; ORIGINS-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT:    store i32 [[TMP7]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT:    ret <4 x i32> [[R]]
+;
+entry:
+  %a = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr %pa, <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+  %b = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr %pb, <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> poison)
+  %r = call <4 x i32> @llvm.masked.urem.v4i32(<4 x i32> %a, <4 x i32> %b, <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+  ret <4 x i32> %r
+}
+
+; Even when the dividend is clean across all lanes (e.g. a splat constant),
+; disabled lanes of llvm.masked.udiv produce poison and must carry a poisoned
+; shadow so that an unmasked read of a disabled lane is caught.
+define i1 @udiv_clean_dividend_poisons_disabled_lanes(<4 x i32> %b) sanitize_memory {
+; CHECK-LABEL: @udiv_clean_dividend_poisons_disabled_lanes(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT:    call void @llvm.donothing()
+; CHECK-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[TMP0]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT:    [[_MSCMP:%.*]] = icmp ne i128 [[TMP1]], 0
+; CHECK-NEXT:    br i1 [[_MSCMP]], label [[TMP2:%.*]], label [[TMP3:%.*]], !prof [[PROF1]]
+; CHECK:       2:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT:    unreachable
+; CHECK:       3:
+; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> splat (i32 100), <4 x i32> [[B:%.*]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+; CHECK-NEXT:    [[LANE3:%.*]] = extractelement <4 x i32> [[R]], i32 3
+; CHECK-NEXT:    [[TMP4:%.*]] = xor i32 [[LANE3]], 0
+; CHECK-NEXT:    [[TMP5:%.*]] = and i32 0, [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i32 [[TMP5]], 0
+; CHECK-NEXT:    [[_MSPROP_ICMP:%.*]] = and i1 true, [[TMP6]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i32 [[LANE3]], 0
+; CHECK-NEXT:    br i1 [[_MSPROP_ICMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; CHECK:       7:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT:    unreachable
+; CHECK:       8:
+; CHECK-NEXT:    br i1 [[CMP]], label [[T:%.*]], label [[F:%.*]]
+; CHECK:       t:
+; CHECK-NEXT:    store i1 false, ptr @__msan_retval_tls, align 8
+; CHECK-NEXT:    ret i1 true
+; CHECK:       f:
+; CHECK-NEXT:    store i1 false, ptr @__msan_retval_tls, align 8
+; CHECK-NEXT:    ret i1 false
+;
+; ORIGINS-LABEL: @udiv_clean_dividend_poisons_disabled_lanes(
+; ORIGINS-NEXT:  entry:
+; ORIGINS-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT:    [[TMP1:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT:    call void @llvm.donothing()
+; ORIGINS-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> <i1 true, i1 true, i1 true, i1 false>, <4 x i32> [[TMP0]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT:    [[TMP2:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT:    [[_MSCMP:%.*]] = icmp ne i128 [[TMP2]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP]], label [[TMP3:%.*]], label [[TMP4:%.*]], !prof [[PROF1]]
+; ORIGINS:       3:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       4:
+; ORIGINS-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> splat (i32 100), <4 x i32> [[B:%.*]], <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+; ORIGINS-NEXT:    [[LANE3:%.*]] = extractelement <4 x i32> [[R]], i32 3
+; ORIGINS-NEXT:    [[TMP5:%.*]] = xor i32 [[LANE3]], 0
+; ORIGINS-NEXT:    [[TMP6:%.*]] = and i32 0, [[TMP5]]
+; ORIGINS-NEXT:    [[TMP7:%.*]] = icmp eq i32 [[TMP6]], 0
+; ORIGINS-NEXT:    [[_MSPROP_ICMP:%.*]] = and i1 true, [[TMP7]]
+; ORIGINS-NEXT:    [[CMP:%.*]] = icmp eq i32 [[LANE3]], 0
+; ORIGINS-NEXT:    br i1 [[_MSPROP_ICMP]], label [[TMP8:%.*]], label [[TMP9:%.*]], !prof [[PROF1]]
+; ORIGINS:       8:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 0) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       9:
+; ORIGINS-NEXT:    br i1 [[CMP]], label [[T:%.*]], label [[F:%.*]]
+; ORIGINS:       t:
+; ORIGINS-NEXT:    store i1 false, ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT:    store i32 0, ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT:    ret i1 true
+; ORIGINS:       f:
+; ORIGINS-NEXT:    store i1 false, ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT:    store i32 0, ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT:    ret i1 false
+;
+entry:
+  %r = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> splat (i32 100), <4 x i32> %b, <4 x i1> <i1 true, i1 true, i1 true, i1 false>)
+  %lane3 = extractelement <4 x i32> %r, i32 3
+  %cmp = icmp eq i32 %lane3, 0
+  br i1 %cmp, label %t, label %f
+t:
+  ret i1 true
+f:
+  ret i1 false
+}
+
+; With every lane enabled there is nothing to poison, and the divisor is
+; checked in full, exactly as for a plain udiv.
+define <4 x i32> @udiv_all_lanes_enabled(<4 x i32> %x, <4 x i32> %y) sanitize_memory {
+; CHECK-LABEL: @udiv_all_lanes_enabled(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; CHECK-NEXT:    call void @llvm.donothing()
+; CHECK-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> splat (i1 true), <4 x i32> [[TMP0]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> splat (i1 true), <4 x i32> [[TMP1]], <4 x i32> splat (i32 -1)
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; CHECK-NEXT:    [[_MSCMP:%.*]] = icmp ne i128 [[TMP2]], 0
+; CHECK-NEXT:    br i1 [[_MSCMP]], label [[TMP3:%.*]], label [[TMP4:%.*]], !prof [[PROF1]]
+; CHECK:       3:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT:    unreachable
+; CHECK:       4:
+; CHECK-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; CHECK-NEXT:    ret <4 x i32> [[R]]
+;
+; ORIGINS-LABEL: @udiv_all_lanes_enabled(
+; ORIGINS-NEXT:  entry:
+; ORIGINS-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr getelementptr (i8, ptr @__msan_param_tls, i64 16), align 8
+; ORIGINS-NEXT:    [[TMP1:%.*]] = load i32, ptr getelementptr (i8, ptr @__msan_param_origin_tls, i64 16), align 4
+; ORIGINS-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr @__msan_param_tls, align 8
+; ORIGINS-NEXT:    [[TMP3:%.*]] = load i32, ptr @__msan_param_origin_tls, align 4
+; ORIGINS-NEXT:    call void @llvm.donothing()
+; ORIGINS-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <4 x i1> splat (i1 true), <4 x i32> [[TMP0]], <4 x i32> zeroinitializer
+; ORIGINS-NEXT:    [[_MSMASKEDDIV:%.*]] = select <4 x i1> splat (i1 true), <4 x i32> [[TMP2]], <4 x i32> splat (i32 -1)
+; ORIGINS-NEXT:    [[TMP4:%.*]] = bitcast <4 x i32> [[_MSMASKEDDIVISOR]] to i128
+; ORIGINS-NEXT:    [[_MSCMP:%.*]] = icmp ne i128 [[TMP4]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP]], label [[TMP5:%.*]], label [[TMP6:%.*]], !prof [[PROF1]]
+; ORIGINS:       5:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP1]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       6:
+; ORIGINS-NEXT:    [[R:%.*]] = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]], <4 x i1> splat (i1 true))
+; ORIGINS-NEXT:    store <4 x i32> [[_MSMASKEDDIV]], ptr @__msan_retval_tls, align 8
+; ORIGINS-NEXT:    store i32 [[TMP3]], ptr @__msan_retval_origin_tls, align 4
+; ORIGINS-NEXT:    ret <4 x i32> [[R]]
+;
+entry:
+  %r = call <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> %x, <4 x i32> %y, <4 x i1> <i1 true, i1 true, i1 true, i1 true>)
+  ret <4 x i32> %r
+}
+
+; Scalable vectors: the disabled lanes get a poison shadow splat, and the
+; divisor check goes through a vector reduction rather than a bitcast.
+define void @masked_udiv_scalable(ptr %pa, ptr %pb, ptr %pr, <vscale x 4 x i1> %m) sanitize_memory {
+; CHECK-LABEL: @masked_udiv_scalable(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    call void @llvm.donothing()
+; CHECK-NEXT:    [[A:%.*]] = load <vscale x 4 x i32>, ptr [[PA:%.*]], align 16
+; CHECK-NEXT:    [[TMP0:%.*]] = ptrtoint ptr [[PA]] to i64
+; CHECK-NEXT:    [[TMP1:%.*]] = xor i64 [[TMP0]], 87960930222080
+; CHECK-NEXT:    [[TMP2:%.*]] = inttoptr i64 [[TMP1]] to ptr
+; CHECK-NEXT:    [[_MSLD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 16
+; CHECK-NEXT:    [[B:%.*]] = load <vscale x 4 x i32>, ptr [[PB:%.*]], align 16
+; CHECK-NEXT:    [[TMP3:%.*]] = ptrtoint ptr [[PB]] to i64
+; CHECK-NEXT:    [[TMP4:%.*]] = xor i64 [[TMP3]], 87960930222080
+; CHECK-NEXT:    [[TMP5:%.*]] = inttoptr i64 [[TMP4]] to ptr
+; CHECK-NEXT:    [[_MSLD1:%.*]] = load <vscale x 4 x i32>, ptr [[TMP5]], align 16
+; CHECK-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <vscale x 4 x i1> [[M:%.*]], <vscale x 4 x i32> [[_MSLD1]], <vscale x 4 x i32> zeroinitializer
+; CHECK-NEXT:    [[_MSMASKEDDIV:%.*]] = select <vscale x 4 x i1> [[M]], <vscale x 4 x i32> [[_MSLD]], <vscale x 4 x i32> splat (i32 -1)
+; CHECK-NEXT:    [[TMP6:%.*]] = call i32 @llvm.vector.reduce.or.nxv4i32(<vscale x 4 x i32> [[_MSMASKEDDIVISOR]])
+; CHECK-NEXT:    [[_MSCMP:%.*]] = icmp ne i32 [[TMP6]], 0
+; CHECK-NEXT:    br i1 [[_MSCMP]], label [[TMP7:%.*]], label [[TMP8:%.*]], !prof [[PROF1]]
+; CHECK:       7:
+; CHECK-NEXT:    call void @__msan_warning_noreturn() #[[ATTR5]]
+; CHECK-NEXT:    unreachable
+; CHECK:       8:
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i32> @llvm.masked.udiv.nxv4i32(<vscale x 4 x i32> [[A]], <vscale x 4 x i32> [[B]], <vscale x 4 x i1> [[M]])
+; CHECK-NEXT:    [[TMP9:%.*]] = ptrtoint ptr [[PR:%.*]] to i64
+; CHECK-NEXT:    [[TMP10:%.*]] = xor i64 [[TMP9]], 87960930222080
+; CHECK-NEXT:    [[TMP11:%.*]] = inttoptr i64 [[TMP10]] to ptr
+; CHECK-NEXT:    store <vscale x 4 x i32> [[_MSMASKEDDIV]], ptr [[TMP11]], align 16
+; CHECK-NEXT:    store <vscale x 4 x i32> [[R]], ptr [[PR]], align 16
+; CHECK-NEXT:    ret void
+;
+; ORIGINS-LABEL: @masked_udiv_scalable(
+; ORIGINS-NEXT:  entry:
+; ORIGINS-NEXT:    call void @llvm.donothing()
+; ORIGINS-NEXT:    [[A:%.*]] = load <vscale x 4 x i32>, ptr [[PA:%.*]], align 16
+; ORIGINS-NEXT:    [[TMP0:%.*]] = ptrtoint ptr [[PA]] to i64
+; ORIGINS-NEXT:    [[TMP1:%.*]] = xor i64 [[TMP0]], 87960930222080
+; ORIGINS-NEXT:    [[TMP2:%.*]] = inttoptr i64 [[TMP1]] to ptr
+; ORIGINS-NEXT:    [[TMP3:%.*]] = add i64 [[TMP1]], 17592186044416
+; ORIGINS-NEXT:    [[TMP4:%.*]] = inttoptr i64 [[TMP3]] to ptr
+; ORIGINS-NEXT:    [[_MSLD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 16
+; ORIGINS-NEXT:    [[TMP5:%.*]] = load i32, ptr [[TMP4]], align 16
+; ORIGINS-NEXT:    [[B:%.*]] = load <vscale x 4 x i32>, ptr [[PB:%.*]], align 16
+; ORIGINS-NEXT:    [[TMP6:%.*]] = ptrtoint ptr [[PB]] to i64
+; ORIGINS-NEXT:    [[TMP7:%.*]] = xor i64 [[TMP6]], 87960930222080
+; ORIGINS-NEXT:    [[TMP8:%.*]] = inttoptr i64 [[TMP7]] to ptr
+; ORIGINS-NEXT:    [[TMP9:%.*]] = add i64 [[TMP7]], 17592186044416
+; ORIGINS-NEXT:    [[TMP10:%.*]] = inttoptr i64 [[TMP9]] to ptr
+; ORIGINS-NEXT:    [[_MSLD1:%.*]] = load <vscale x 4 x i32>, ptr [[TMP8]], align 16
+; ORIGINS-NEXT:    [[TMP11:%.*]] = load i32, ptr [[TMP10]], align 16
+; ORIGINS-NEXT:    [[_MSMASKEDDIVISOR:%.*]] = select <vscale x 4 x i1> [[M:%.*]], <vscale x 4 x i32> [[_MSLD1]], <vscale x 4 x i32> zeroinitializer
+; ORIGINS-NEXT:    [[_MSMASKEDDIV:%.*]] = select <vscale x 4 x i1> [[M]], <vscale x 4 x i32> [[_MSLD]], <vscale x 4 x i32> splat (i32 -1)
+; ORIGINS-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.or.nxv4i32(<vscale x 4 x i32> [[_MSMASKEDDIVISOR]])
+; ORIGINS-NEXT:    [[_MSCMP:%.*]] = icmp ne i32 [[TMP12]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP]], label [[TMP13:%.*]], label [[TMP14:%.*]], !prof [[PROF1]]
+; ORIGINS:       13:
+; ORIGINS-NEXT:    call void @__msan_warning_with_origin_noreturn(i32 [[TMP11]]) #[[ATTR6]]
+; ORIGINS-NEXT:    unreachable
+; ORIGINS:       14:
+; ORIGINS-NEXT:    [[R:%.*]] = call <vscale x 4 x i32> @llvm.masked.udiv.nxv4i32(<vscale x 4 x i32> [[A]], <vscale x 4 x i32> [[B]], <vscale x 4 x i1> [[M]])
+; ORIGINS-NEXT:    [[TMP15:%.*]] = ptrtoint ptr [[PR:%.*]] to i64
+; ORIGINS-NEXT:    [[TMP16:%.*]] = xor i64 [[TMP15]], 87960930222080
+; ORIGINS-NEXT:    [[TMP17:%.*]] = inttoptr i64 [[TMP16]] to ptr
+; ORIGINS-NEXT:    [[TMP18:%.*]] = add i64 [[TMP16]], 17592186044416
+; ORIGINS-NEXT:    [[TMP19:%.*]] = inttoptr i64 [[TMP18]] to ptr
+; ORIGINS-NEXT:    store <vscale x 4 x i32> [[_MSMASKEDDIV]], ptr [[TMP17]], align 16
+; ORIGINS-NEXT:    [[TMP20:%.*]] = call i32 @llvm.vector.reduce.or.nxv4i32(<vscale x 4 x i32> [[_MSMASKEDDIV]])
+; ORIGINS-NEXT:    [[_MSCMP2:%.*]] = icmp ne i32 [[TMP20]], 0
+; ORIGINS-NEXT:    br i1 [[_MSCMP2]], label [[TMP21:%.*]], label [[TMP27:%.*]], !prof [[PROF1]]
+; ORIGINS:       21:
+; ORIGINS-NEXT:    [[TMP22:%.*]] = call i64 @llvm.vscale.i64()
+; ORIGINS-NEXT:    [[TMP23:%.*]] = mul nuw i64 [[TMP22]], 16
+; ORIGINS-NEXT:    [[TMP24:%.*]] = add i64 [[TMP23]], 3
+; ORIGINS-NEXT:    [[TMP25:%.*]] = udiv i64 [[TMP24]], 4
+; ORIGINS-NEXT:    br label [[DOTSPLIT:%.*]]
+; ORIGINS:       .split:
+; ORIGINS-NEXT:    [[IV:%.*]] = phi i64 [ 0, [[TMP21]] ], [ [[IV_NEXT:%.*]], [[DOTSPLIT]] ]
+; ORIGINS-NEXT:    [[TMP26:%.*]] = getelementptr i32, ptr [[TMP19]], i64 [[IV]]
+; ORIGINS-NEXT:    store i32 [[TMP5]], ptr [[TMP26]], align 4
+; ORIGINS-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; ORIGINS-NEXT:    [[IV_CHECK:%.*]] = icmp eq i64 [[IV_NEXT]], [[TMP25]]
+; ORIGINS-NEXT:    br i1 [[IV_CHECK]], label [[DOTSPLIT_SPLIT:%.*]], label [[DOTSPLIT]]
+; ORIGINS:       .split.split:
+; ORIGINS-NEXT:    br label [[TMP27]]
+; ORIGINS:       27:
+; ORIGINS-NEXT:    store <vscale x 4 x i32> [[R]], ptr [[PR]], align 16
+; ORIGINS-NEXT:    ret void
+;
+entry:
+  %a = load <vscale x 4 x i32>, ptr %pa
+  %b = load <vscale x 4 x i32>, ptr %pb
+  %r = call <vscale x 4 x i32> @llvm.masked.udiv.nxv4i32(<vscale x 4 x i32> %a, <vscale x 4 x i32> %b, <vscale x 4 x i1> %m)
+  store <vscale x 4 x i32> %r, ptr %pr
+  ret void
+}



More information about the llvm-commits mailing list