1//===- MemorySanitizer.cpp - detector of uninitialized reads --------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file is a part of MemorySanitizer, a detector of uninitialized
11/// reads.
12///
13/// The algorithm of the tool is similar to Memcheck
14/// (https://static.usenix.org/event/usenix05/tech/general/full_papers/seward/seward_html/usenix2005.html)
15/// We associate a few shadow bits with every byte of the application memory,
16/// poison the shadow of the malloc-ed or alloca-ed memory, load the shadow,
17/// bits on every memory read, propagate the shadow bits through some of the
18/// arithmetic instruction (including MOV), store the shadow bits on every
19/// memory write, report a bug on some other instructions (e.g. JMP) if the
20/// associated shadow is poisoned.
21///
22/// But there are differences too. The first and the major one:
23/// compiler instrumentation instead of binary instrumentation. This
24/// gives us much better register allocation, possible compiler
25/// optimizations and a fast start-up. But this brings the major issue
26/// as well: msan needs to see all program events, including system
27/// calls and reads/writes in system libraries, so we either need to
28/// compile *everything* with msan or use a binary translation
29/// component (e.g. DynamoRIO) to instrument pre-built libraries.
30/// Another difference from Memcheck is that we use 8 shadow bits per
31/// byte of application memory and use a direct shadow mapping. This
32/// greatly simplifies the instrumentation code and avoids races on
33/// shadow updates (Memcheck is single-threaded so races are not a
34/// concern there. Memcheck uses 2 shadow bits per byte with a slow
35/// path storage that uses 8 bits per byte).
36///
37/// The default value of shadow is 0, which means "clean" (not poisoned).
38///
39/// Every module initializer should call __msan_init to ensure that the
40/// shadow memory is ready. On error, __msan_warning is called. Since
41/// parameters and return values may be passed via registers, we have a
42/// specialized thread-local shadow for return values
43/// (__msan_retval_tls) and parameters (__msan_param_tls).
44///
45/// Origin tracking.
46///
47/// MemorySanitizer can track origins (allocation points) of all uninitialized
48/// values. This behavior is controlled with a flag (msan-track-origins) and is
49/// disabled by default.
50///
51/// Origins are 4-byte values created and interpreted by the runtime library.
52/// They are stored in a second shadow mapping, one 4-byte value for 4 bytes
53/// of application memory. Propagation of origins is basically a bunch of
54/// "select" instructions that pick the origin of a dirty argument, if an
55/// instruction has one.
56///
57/// Every 4 aligned, consecutive bytes of application memory have one origin
58/// value associated with them. If these bytes contain uninitialized data
59/// coming from 2 different allocations, the last store wins. Because of this,
60/// MemorySanitizer reports can show unrelated origins, but this is unlikely in
61/// practice.
62///
63/// Origins are meaningless for fully initialized values, so MemorySanitizer
64/// avoids storing origin to memory when a fully initialized value is stored.
65/// This way it avoids needless overwriting origin of the 4-byte region on
66/// a short (i.e. 1 byte) clean store, and it is also good for performance.
67///
68/// Atomic handling.
69///
70/// Ideally, every atomic store of application value should update the
71/// corresponding shadow location in an atomic way. Unfortunately, atomic store
72/// of two disjoint locations can not be done without severe slowdown.
73///
74/// Therefore, we implement an approximation that may err on the safe side.
75/// In this implementation, every atomically accessed location in the program
76/// may only change from (partially) uninitialized to fully initialized, but
77/// not the other way around. We load the shadow _after_ the application load,
78/// and we store the shadow _before_ the app store. Also, we always store clean
79/// shadow (if the application store is atomic). This way, if the store-load
80/// pair constitutes a happens-before arc, shadow store and load are correctly
81/// ordered such that the load will get either the value that was stored, or
82/// some later value (which is always clean).
83///
84/// This does not work very well with Compare-And-Swap (CAS) and
85/// Read-Modify-Write (RMW) operations. To follow the above logic, CAS and RMW
86/// must store the new shadow before the app operation, and load the shadow
87/// after the app operation. Computers don't work this way. Current
88/// implementation ignores the load aspect of CAS/RMW, always returning a clean
89/// value. It implements the store part as a simple atomic store by storing a
90/// clean shadow.
91///
92/// Instrumenting inline assembly.
93///
94/// For inline assembly code LLVM has little idea about which memory locations
95/// become initialized depending on the arguments. It can be possible to figure
96/// out which arguments are meant to point to inputs and outputs, but the
97/// actual semantics can be only visible at runtime. In the Linux kernel it's
98/// also possible that the arguments only indicate the offset for a base taken
99/// from a segment register, so it's dangerous to treat any asm() arguments as
100/// pointers. We take a conservative approach generating calls to
101/// __msan_instrument_asm_store(ptr, size)
102/// , which defer the memory unpoisoning to the runtime library.
103/// The latter can perform more complex address checks to figure out whether
104/// it's safe to touch the shadow memory.
105/// Like with atomic operations, we call __msan_instrument_asm_store() before
106/// the assembly call, so that changes to the shadow memory will be seen by
107/// other threads together with main memory initialization.
108///
109/// KernelMemorySanitizer (KMSAN) implementation.
110///
111/// The major differences between KMSAN and MSan instrumentation are:
112/// - KMSAN always tracks the origins and implies msan-keep-going=true;
113/// - KMSAN allocates shadow and origin memory for each page separately, so
114/// there are no explicit accesses to shadow and origin in the
115/// instrumentation.
116/// Shadow and origin values for a particular X-byte memory location
117/// (X=1,2,4,8) are accessed through pointers obtained via the
118/// __msan_metadata_ptr_for_load_X(ptr)
119/// __msan_metadata_ptr_for_store_X(ptr)
120/// functions. The corresponding functions check that the X-byte accesses
121/// are possible and returns the pointers to shadow and origin memory.
122/// Arbitrary sized accesses are handled with:
123/// __msan_metadata_ptr_for_load_n(ptr, size)
124/// __msan_metadata_ptr_for_store_n(ptr, size);
125/// Note that the sanitizer code has to deal with how shadow/origin pairs
126/// returned by the these functions are represented in different ABIs. In
127/// the X86_64 ABI they are returned in RDX:RAX, in PowerPC64 they are
128/// returned in r3 and r4, and in the SystemZ ABI they are written to memory
129/// pointed to by a hidden parameter.
130/// - TLS variables are stored in a single per-task struct. A call to a
131/// function __msan_get_context_state() returning a pointer to that struct
132/// is inserted into every instrumented function before the entry block;
133/// - __msan_warning() takes a 32-bit origin parameter;
134/// - local variables are poisoned with __msan_poison_alloca() upon function
135/// entry and unpoisoned with __msan_unpoison_alloca() before leaving the
136/// function;
137/// - the pass doesn't declare any global variables or add global constructors
138/// to the translation unit.
139///
140/// Also, KMSAN currently ignores uninitialized memory passed into inline asm
141/// calls, making sure we're on the safe side wrt. possible false positives.
142///
143/// KernelMemorySanitizer only supports X86_64, SystemZ and PowerPC64 at the
144/// moment.
145///
146//
147// FIXME: This sanitizer does not yet handle scalable vectors
148//
149//===----------------------------------------------------------------------===//
150
151#include "llvm/Transforms/Instrumentation/MemorySanitizer.h"
152#include "llvm/ADT/APInt.h"
153#include "llvm/ADT/ArrayRef.h"
154#include "llvm/ADT/DenseMap.h"
155#include "llvm/ADT/DepthFirstIterator.h"
156#include "llvm/ADT/SetVector.h"
157#include "llvm/ADT/SmallPtrSet.h"
158#include "llvm/ADT/SmallVector.h"
159#include "llvm/ADT/StringExtras.h"
160#include "llvm/ADT/StringRef.h"
161#include "llvm/Analysis/GlobalsModRef.h"
162#include "llvm/Analysis/TargetLibraryInfo.h"
163#include "llvm/Analysis/ValueTracking.h"
164#include "llvm/IR/Argument.h"
165#include "llvm/IR/AttributeMask.h"
166#include "llvm/IR/Attributes.h"
167#include "llvm/IR/BasicBlock.h"
168#include "llvm/IR/CallingConv.h"
169#include "llvm/IR/Constant.h"
170#include "llvm/IR/Constants.h"
171#include "llvm/IR/DataLayout.h"
172#include "llvm/IR/DerivedTypes.h"
173#include "llvm/IR/Function.h"
174#include "llvm/IR/GlobalValue.h"
175#include "llvm/IR/GlobalVariable.h"
176#include "llvm/IR/IRBuilder.h"
177#include "llvm/IR/InlineAsm.h"
178#include "llvm/IR/InstVisitor.h"
179#include "llvm/IR/InstrTypes.h"
180#include "llvm/IR/Instruction.h"
181#include "llvm/IR/Instructions.h"
182#include "llvm/IR/IntrinsicInst.h"
183#include "llvm/IR/Intrinsics.h"
184#include "llvm/IR/IntrinsicsAArch64.h"
185#include "llvm/IR/IntrinsicsX86.h"
186#include "llvm/IR/MDBuilder.h"
187#include "llvm/IR/Module.h"
188#include "llvm/IR/Type.h"
189#include "llvm/IR/Value.h"
190#include "llvm/IR/ValueMap.h"
191#include "llvm/Support/Alignment.h"
192#include "llvm/Support/AtomicOrdering.h"
193#include "llvm/Support/Casting.h"
194#include "llvm/Support/CommandLine.h"
195#include "llvm/Support/Debug.h"
196#include "llvm/Support/DebugCounter.h"
197#include "llvm/Support/ErrorHandling.h"
198#include "llvm/Support/MathExtras.h"
199#include "llvm/Support/raw_ostream.h"
200#include "llvm/TargetParser/Triple.h"
201#include "llvm/Transforms/Utils/BasicBlockUtils.h"
202#include "llvm/Transforms/Utils/Instrumentation.h"
203#include "llvm/Transforms/Utils/Local.h"
204#include "llvm/Transforms/Utils/ModuleUtils.h"
205#include <algorithm>
206#include <cassert>
207#include <cstddef>
208#include <cstdint>
209#include <memory>
210#include <numeric>
211#include <string>
212#include <tuple>
213
214using namespace llvm;
215
216#define DEBUG_TYPE "msan"
217
218DEBUG_COUNTER(DebugInsertCheck, "msan-insert-check",
219 "Controls which checks to insert");
220
221DEBUG_COUNTER(DebugInstrumentInstruction, "msan-instrument-instruction",
222 "Controls which instruction to instrument");
223
224static const unsigned kOriginSize = 4;
225static const Align kMinOriginAlignment = Align(4);
226static const Align kShadowTLSAlignment = Align(8);
227
228// These constants must be kept in sync with the ones in msan.h.
229// TODO: increase size to match SVE/SVE2/SME/SME2 limits
230static const unsigned kParamTLSSize = 800;
231static const unsigned kRetvalTLSSize = 800;
232
233// Accesses sizes are powers of two: 1, 2, 4, 8.
234static const size_t kNumberOfAccessSizes = 4;
235
236/// Track origins of uninitialized values.
237///
238/// Adds a section to MemorySanitizer report that points to the allocation
239/// (stack or heap) the uninitialized bits came from originally.
240static cl::opt<int> ClTrackOrigins(
241 "msan-track-origins",
242 cl::desc("Track origins (allocation sites) of poisoned memory"), cl::Hidden,
243 cl::init(Val: 0));
244
245static cl::opt<bool> ClKeepGoing("msan-keep-going",
246 cl::desc("keep going after reporting a UMR"),
247 cl::Hidden, cl::init(Val: false));
248
249static cl::opt<bool>
250 ClPoisonStack("msan-poison-stack",
251 cl::desc("poison uninitialized stack variables"), cl::Hidden,
252 cl::init(Val: true));
253
254static cl::opt<bool> ClPoisonStackWithCall(
255 "msan-poison-stack-with-call",
256 cl::desc("poison uninitialized stack variables with a call"), cl::Hidden,
257 cl::init(Val: false));
258
259static cl::opt<int> ClPoisonStackPattern(
260 "msan-poison-stack-pattern",
261 cl::desc("poison uninitialized stack variables with the given pattern"),
262 cl::Hidden, cl::init(Val: 0xff));
263
264static cl::opt<bool>
265 ClPrintStackNames("msan-print-stack-names",
266 cl::desc("Print name of local stack variable"),
267 cl::Hidden, cl::init(Val: true));
268
269static cl::opt<bool>
270 ClPoisonUndef("msan-poison-undef",
271 cl::desc("Poison fully undef temporary values. "
272 "Partially undefined constant vectors "
273 "are unaffected by this flag (see "
274 "-msan-poison-undef-vectors)."),
275 cl::Hidden, cl::init(Val: true));
276
277static cl::opt<bool> ClPoisonUndefVectors(
278 "msan-poison-undef-vectors",
279 cl::desc("Precisely poison partially undefined constant vectors. "
280 "If false (legacy behavior), the entire vector is "
281 "considered fully initialized, which may lead to false "
282 "negatives. Fully undefined constant vectors are "
283 "unaffected by this flag (see -msan-poison-undef)."),
284 cl::Hidden, cl::init(Val: false));
285
286static cl::opt<bool> ClPreciseDisjointOr(
287 "msan-precise-disjoint-or",
288 cl::desc("Precisely poison disjoint OR. If false (legacy behavior), "
289 "disjointedness is ignored (i.e., 1|1 is initialized)."),
290 cl::Hidden, cl::init(Val: false));
291
292static cl::opt<bool>
293 ClHandleICmp("msan-handle-icmp",
294 cl::desc("propagate shadow through ICmpEQ and ICmpNE"),
295 cl::Hidden, cl::init(Val: true));
296
297static cl::opt<bool>
298 ClHandleICmpExact("msan-handle-icmp-exact",
299 cl::desc("exact handling of relational integer ICmp"),
300 cl::Hidden, cl::init(Val: true));
301
302static cl::opt<int> ClSwitchPrecision(
303 "msan-switch-precision",
304 cl::desc("Controls the number of cases considered by MSan for LLVM switch "
305 "instructions. 0 means no UUMs detected. Higher values lead to "
306 "fewer false negatives but may impact compiler and/or "
307 "application performance. N.B. LLVM switch instructions do not "
308 "correspond exactly to C++ switch statements."),
309 cl::Hidden, cl::init(Val: 99));
310
311static cl::opt<bool> ClHandleLifetimeIntrinsics(
312 "msan-handle-lifetime-intrinsics",
313 cl::desc(
314 "when possible, poison scoped variables at the beginning of the scope "
315 "(slower, but more precise)"),
316 cl::Hidden, cl::init(Val: true));
317
318// When compiling the Linux kernel, we sometimes see false positives related to
319// MSan being unable to understand that inline assembly calls may initialize
320// local variables.
321// This flag makes the compiler conservatively unpoison every memory location
322// passed into an assembly call. Note that this may cause false positives.
323// Because it's impossible to figure out the array sizes, we can only unpoison
324// the first sizeof(type) bytes for each type* pointer.
325static cl::opt<bool> ClHandleAsmConservative(
326 "msan-handle-asm-conservative",
327 cl::desc("conservative handling of inline assembly"), cl::Hidden,
328 cl::init(Val: true));
329
330// This flag controls whether we check the shadow of the address
331// operand of load or store. Such bugs are very rare, since load from
332// a garbage address typically results in SEGV, but still happen
333// (e.g. only lower bits of address are garbage, or the access happens
334// early at program startup where malloc-ed memory is more likely to
335// be zeroed. As of 2012-08-28 this flag adds 20% slowdown.
336static cl::opt<bool> ClCheckAccessAddress(
337 "msan-check-access-address",
338 cl::desc("report accesses through a pointer which has poisoned shadow"),
339 cl::Hidden, cl::init(Val: true));
340
341static cl::opt<bool> ClEagerChecks(
342 "msan-eager-checks",
343 cl::desc("check arguments and return values at function call boundaries"),
344 cl::Hidden, cl::init(Val: false));
345
346static cl::opt<bool> ClDumpStrictInstructions(
347 "msan-dump-strict-instructions",
348 cl::desc("print out instructions with default strict semantics i.e.,"
349 "check that all the inputs are fully initialized, and mark "
350 "the output as fully initialized. These semantics are applied "
351 "to instructions that could not be handled explicitly nor "
352 "heuristically."),
353 cl::Hidden, cl::init(Val: false));
354
355// Currently, all the heuristically handled instructions are specifically
356// IntrinsicInst. However, we use the broader "HeuristicInstructions" name
357// to parallel 'msan-dump-strict-instructions', and to keep the door open to
358// handling non-intrinsic instructions heuristically.
359static cl::opt<bool> ClDumpHeuristicInstructions(
360 "msan-dump-heuristic-instructions",
361 cl::desc("Prints 'unknown' instructions that were handled heuristically. "
362 "Use -msan-dump-strict-instructions to print instructions that "
363 "could not be handled explicitly nor heuristically."),
364 cl::Hidden, cl::init(Val: false));
365
366static cl::opt<int> ClInstrumentationWithCallThreshold(
367 "msan-instrumentation-with-call-threshold",
368 cl::desc(
369 "If the function being instrumented requires more than "
370 "this number of checks and origin stores, use callbacks instead of "
371 "inline checks (-1 means never use callbacks)."),
372 cl::Hidden, cl::init(Val: 3500));
373
374static cl::opt<bool>
375 ClEnableKmsan("msan-kernel",
376 cl::desc("Enable KernelMemorySanitizer instrumentation"),
377 cl::Hidden, cl::init(Val: false));
378
379static cl::opt<bool>
380 ClDisableChecks("msan-disable-checks",
381 cl::desc("Apply no_sanitize to the whole file"), cl::Hidden,
382 cl::init(Val: false));
383
384static cl::opt<bool>
385 ClCheckConstantShadow("msan-check-constant-shadow",
386 cl::desc("Insert checks for constant shadow values"),
387 cl::Hidden, cl::init(Val: true));
388
389// This is off by default because of a bug in gold:
390// https://sourceware.org/bugzilla/show_bug.cgi?id=19002
391static cl::opt<bool>
392 ClWithComdat("msan-with-comdat",
393 cl::desc("Place MSan constructors in comdat sections"),
394 cl::Hidden, cl::init(Val: false));
395
396// These options allow to specify custom memory map parameters
397// See MemoryMapParams for details.
398static cl::opt<uint64_t> ClAndMask("msan-and-mask",
399 cl::desc("Define custom MSan AndMask"),
400 cl::Hidden, cl::init(Val: 0));
401
402static cl::opt<uint64_t> ClXorMask("msan-xor-mask",
403 cl::desc("Define custom MSan XorMask"),
404 cl::Hidden, cl::init(Val: 0));
405
406static cl::opt<uint64_t> ClShadowBase("msan-shadow-base",
407 cl::desc("Define custom MSan ShadowBase"),
408 cl::Hidden, cl::init(Val: 0));
409
410static cl::opt<uint64_t> ClOriginBase("msan-origin-base",
411 cl::desc("Define custom MSan OriginBase"),
412 cl::Hidden, cl::init(Val: 0));
413
414static cl::opt<int>
415 ClDisambiguateWarning("msan-disambiguate-warning-threshold",
416 cl::desc("Define threshold for number of checks per "
417 "debug location to force origin update."),
418 cl::Hidden, cl::init(Val: 3));
419
420const char kMsanModuleCtorName[] = "msan.module_ctor";
421const char kMsanInitName[] = "__msan_init";
422
423namespace {
424
425// Memory map parameters used in application-to-shadow address calculation.
426// Offset = (Addr & ~AndMask) ^ XorMask
427// Shadow = ShadowBase + Offset
428// Origin = OriginBase + Offset
429struct MemoryMapParams {
430 uint64_t AndMask;
431 uint64_t XorMask;
432 uint64_t ShadowBase;
433 uint64_t OriginBase;
434};
435
436struct PlatformMemoryMapParams {
437 const MemoryMapParams *bits32;
438 const MemoryMapParams *bits64;
439};
440
441} // end anonymous namespace
442
443// i386 Linux
444static const MemoryMapParams Linux_I386_MemoryMapParams = {
445 .AndMask: 0x000080000000, // AndMask
446 .XorMask: 0, // XorMask (not used)
447 .ShadowBase: 0, // ShadowBase (not used)
448 .OriginBase: 0x000040000000, // OriginBase
449};
450
451// x86_64 Linux
452static const MemoryMapParams Linux_X86_64_MemoryMapParams = {
453 .AndMask: 0, // AndMask (not used)
454 .XorMask: 0x500000000000, // XorMask
455 .ShadowBase: 0, // ShadowBase (not used)
456 .OriginBase: 0x100000000000, // OriginBase
457};
458
459// mips32 Linux
460// FIXME: Remove -msan-origin-base -msan-and-mask added by PR #109284 to tests
461// after picking good constants
462
463// mips64 Linux
464static const MemoryMapParams Linux_MIPS64_MemoryMapParams = {
465 .AndMask: 0, // AndMask (not used)
466 .XorMask: 0x008000000000, // XorMask
467 .ShadowBase: 0, // ShadowBase (not used)
468 .OriginBase: 0x002000000000, // OriginBase
469};
470
471// ppc32 Linux
472// FIXME: Remove -msan-origin-base -msan-and-mask added by PR #109284 to tests
473// after picking good constants
474
475// ppc64 Linux
476static const MemoryMapParams Linux_PowerPC64_MemoryMapParams = {
477 .AndMask: 0xE00000000000, // AndMask
478 .XorMask: 0x100000000000, // XorMask
479 .ShadowBase: 0x080000000000, // ShadowBase
480 .OriginBase: 0x1C0000000000, // OriginBase
481};
482
483// s390x Linux
484static const MemoryMapParams Linux_S390X_MemoryMapParams = {
485 .AndMask: 0xC00000000000, // AndMask
486 .XorMask: 0, // XorMask (not used)
487 .ShadowBase: 0x080000000000, // ShadowBase
488 .OriginBase: 0x1C0000000000, // OriginBase
489};
490
491// arm32 Linux
492// FIXME: Remove -msan-origin-base -msan-and-mask added by PR #109284 to tests
493// after picking good constants
494
495// aarch64 Linux
496static const MemoryMapParams Linux_AArch64_MemoryMapParams = {
497 .AndMask: 0, // AndMask (not used)
498 .XorMask: 0x0B00000000000, // XorMask
499 .ShadowBase: 0, // ShadowBase (not used)
500 .OriginBase: 0x0200000000000, // OriginBase
501};
502
503// loongarch64 Linux
504static const MemoryMapParams Linux_LoongArch64_MemoryMapParams = {
505 .AndMask: 0, // AndMask (not used)
506 .XorMask: 0x500000000000, // XorMask
507 .ShadowBase: 0, // ShadowBase (not used)
508 .OriginBase: 0x100000000000, // OriginBase
509};
510
511// hexagon Linux
512static const MemoryMapParams Linux_Hexagon_MemoryMapParams = {
513 .AndMask: 0, // AndMask (not used)
514 .XorMask: 0x20000000, // XorMask
515 .ShadowBase: 0, // ShadowBase (not used)
516 .OriginBase: 0x50000000, // OriginBase
517};
518
519// riscv32 Linux
520// FIXME: Remove -msan-origin-base -msan-and-mask added by PR #109284 to tests
521// after picking good constants
522
523// aarch64 FreeBSD
524static const MemoryMapParams FreeBSD_AArch64_MemoryMapParams = {
525 .AndMask: 0x1800000000000, // AndMask
526 .XorMask: 0x0400000000000, // XorMask
527 .ShadowBase: 0x0200000000000, // ShadowBase
528 .OriginBase: 0x0700000000000, // OriginBase
529};
530
531// i386 FreeBSD
532static const MemoryMapParams FreeBSD_I386_MemoryMapParams = {
533 .AndMask: 0x000180000000, // AndMask
534 .XorMask: 0x000040000000, // XorMask
535 .ShadowBase: 0x000020000000, // ShadowBase
536 .OriginBase: 0x000700000000, // OriginBase
537};
538
539// x86_64 FreeBSD
540static const MemoryMapParams FreeBSD_X86_64_MemoryMapParams = {
541 .AndMask: 0xc00000000000, // AndMask
542 .XorMask: 0x200000000000, // XorMask
543 .ShadowBase: 0x100000000000, // ShadowBase
544 .OriginBase: 0x380000000000, // OriginBase
545};
546
547// x86_64 NetBSD
548static const MemoryMapParams NetBSD_X86_64_MemoryMapParams = {
549 .AndMask: 0, // AndMask
550 .XorMask: 0x500000000000, // XorMask
551 .ShadowBase: 0, // ShadowBase
552 .OriginBase: 0x100000000000, // OriginBase
553};
554
555static const PlatformMemoryMapParams Linux_X86_MemoryMapParams = {
556 .bits32: &Linux_I386_MemoryMapParams,
557 .bits64: &Linux_X86_64_MemoryMapParams,
558};
559
560static const PlatformMemoryMapParams Linux_MIPS_MemoryMapParams = {
561 .bits32: nullptr,
562 .bits64: &Linux_MIPS64_MemoryMapParams,
563};
564
565static const PlatformMemoryMapParams Linux_PowerPC_MemoryMapParams = {
566 .bits32: nullptr,
567 .bits64: &Linux_PowerPC64_MemoryMapParams,
568};
569
570static const PlatformMemoryMapParams Linux_S390_MemoryMapParams = {
571 .bits32: nullptr,
572 .bits64: &Linux_S390X_MemoryMapParams,
573};
574
575static const PlatformMemoryMapParams Linux_ARM_MemoryMapParams = {
576 .bits32: nullptr,
577 .bits64: &Linux_AArch64_MemoryMapParams,
578};
579
580static const PlatformMemoryMapParams Linux_LoongArch_MemoryMapParams = {
581 .bits32: nullptr,
582 .bits64: &Linux_LoongArch64_MemoryMapParams,
583};
584
585static const PlatformMemoryMapParams Linux_Hexagon_MemoryMapParams_P = {
586 .bits32: &Linux_Hexagon_MemoryMapParams,
587 .bits64: nullptr,
588};
589
590static const PlatformMemoryMapParams FreeBSD_ARM_MemoryMapParams = {
591 .bits32: nullptr,
592 .bits64: &FreeBSD_AArch64_MemoryMapParams,
593};
594
595static const PlatformMemoryMapParams FreeBSD_X86_MemoryMapParams = {
596 .bits32: &FreeBSD_I386_MemoryMapParams,
597 .bits64: &FreeBSD_X86_64_MemoryMapParams,
598};
599
600static const PlatformMemoryMapParams NetBSD_X86_MemoryMapParams = {
601 .bits32: nullptr,
602 .bits64: &NetBSD_X86_64_MemoryMapParams,
603};
604
605enum OddOrEvenLanes { kBothLanes, kEvenLanes, kOddLanes };
606
607namespace {
608
609/// Instrument functions of a module to detect uninitialized reads.
610///
611/// Instantiating MemorySanitizer inserts the msan runtime library API function
612/// declarations into the module if they don't exist already. Instantiating
613/// ensures the __msan_init function is in the list of global constructors for
614/// the module.
615class MemorySanitizer {
616public:
617 MemorySanitizer(Module &M, MemorySanitizerOptions Options)
618 : CompileKernel(Options.Kernel), TrackOrigins(Options.TrackOrigins),
619 Recover(Options.Recover), EagerChecks(Options.EagerChecks) {
620 initializeModule(M);
621 }
622
623 // MSan cannot be moved or copied because of MapParams.
624 MemorySanitizer(MemorySanitizer &&) = delete;
625 MemorySanitizer &operator=(MemorySanitizer &&) = delete;
626 MemorySanitizer(const MemorySanitizer &) = delete;
627 MemorySanitizer &operator=(const MemorySanitizer &) = delete;
628
629 bool sanitizeFunction(Function &F, TargetLibraryInfo &TLI);
630
631private:
632 friend struct MemorySanitizerVisitor;
633 friend struct VarArgHelperBase;
634 friend struct VarArgAMD64Helper;
635 friend struct VarArgAArch64Helper;
636 friend struct VarArgPowerPC64Helper;
637 friend struct VarArgPowerPC32Helper;
638 friend struct VarArgSystemZHelper;
639 friend struct VarArgI386Helper;
640 friend struct VarArgGenericHelper;
641
642 void initializeModule(Module &M);
643 void initializeCallbacks(Module &M, const TargetLibraryInfo &TLI);
644 void createKernelApi(Module &M, const TargetLibraryInfo &TLI);
645 void createUserspaceApi(Module &M, const TargetLibraryInfo &TLI);
646
647 template <typename... ArgsTy>
648 FunctionCallee getOrInsertMsanMetadataFunction(Module &M, StringRef Name,
649 ArgsTy... Args);
650
651 /// True if we're compiling the Linux kernel.
652 bool CompileKernel;
653 /// Track origins (allocation points) of uninitialized values.
654 int TrackOrigins;
655 bool Recover;
656 bool EagerChecks;
657
658 Triple TargetTriple;
659 LLVMContext *C;
660 Type *IntptrTy; ///< Integer type with the size of a ptr in default AS.
661 Type *OriginTy;
662 PointerType *PtrTy; ///< Integer type with the size of a ptr in default AS.
663
664 // XxxTLS variables represent the per-thread state in MSan and per-task state
665 // in KMSAN.
666 // For the userspace these point to thread-local globals. In the kernel land
667 // they point to the members of a per-task struct obtained via a call to
668 // __msan_get_context_state().
669
670 /// Thread-local shadow storage for function parameters.
671 Value *ParamTLS;
672
673 /// Thread-local origin storage for function parameters.
674 Value *ParamOriginTLS;
675
676 /// Thread-local shadow storage for function return value.
677 Value *RetvalTLS;
678
679 /// Thread-local origin storage for function return value.
680 Value *RetvalOriginTLS;
681
682 /// Thread-local shadow storage for in-register va_arg function.
683 Value *VAArgTLS;
684
685 /// Thread-local shadow storage for in-register va_arg function.
686 Value *VAArgOriginTLS;
687
688 /// Thread-local shadow storage for va_arg overflow area.
689 Value *VAArgOverflowSizeTLS;
690
691 /// Are the instrumentation callbacks set up?
692 bool CallbacksInitialized = false;
693
694 /// The run-time callback to print a warning.
695 FunctionCallee WarningFn;
696
697 // These arrays are indexed by log2(AccessSize).
698 FunctionCallee MaybeWarningFn[kNumberOfAccessSizes];
699 FunctionCallee MaybeWarningVarSizeFn;
700 FunctionCallee MaybeStoreOriginFn[kNumberOfAccessSizes];
701
702 /// Run-time helper that generates a new origin value for a stack
703 /// allocation.
704 FunctionCallee MsanSetAllocaOriginWithDescriptionFn;
705 // No description version
706 FunctionCallee MsanSetAllocaOriginNoDescriptionFn;
707
708 /// Run-time helper that poisons stack on function entry.
709 FunctionCallee MsanPoisonStackFn;
710
711 /// Run-time helper that records a store (or any event) of an
712 /// uninitialized value and returns an updated origin id encoding this info.
713 FunctionCallee MsanChainOriginFn;
714
715 /// Run-time helper that paints an origin over a region.
716 FunctionCallee MsanSetOriginFn;
717
718 /// MSan runtime replacements for memmove, memcpy and memset.
719 FunctionCallee MemmoveFn, MemcpyFn, MemsetFn;
720
721 /// KMSAN callback for task-local function argument shadow.
722 StructType *MsanContextStateTy;
723 FunctionCallee MsanGetContextStateFn;
724
725 /// Functions for poisoning/unpoisoning local variables
726 FunctionCallee MsanPoisonAllocaFn, MsanUnpoisonAllocaFn;
727
728 /// Pair of shadow/origin pointers.
729 Type *MsanMetadata;
730
731 /// Each of the MsanMetadataPtrXxx functions returns a MsanMetadata.
732 FunctionCallee MsanMetadataPtrForLoadN, MsanMetadataPtrForStoreN;
733 FunctionCallee MsanMetadataPtrForLoad_1_8[4];
734 FunctionCallee MsanMetadataPtrForStore_1_8[4];
735 FunctionCallee MsanInstrumentAsmStoreFn;
736
737 /// Storage for return values of the MsanMetadataPtrXxx functions.
738 Value *MsanMetadataAlloca;
739
740 /// Helper to choose between different MsanMetadataPtrXxx().
741 FunctionCallee getKmsanShadowOriginAccessFn(bool isStore, int size);
742
743 /// Memory map parameters used in application-to-shadow calculation.
744 const MemoryMapParams *MapParams;
745
746 /// Custom memory map parameters used when -msan-shadow-base or
747 // -msan-origin-base is provided.
748 MemoryMapParams CustomMapParams;
749
750 MDNode *ColdCallWeights;
751
752 /// Branch weights for origin store.
753 MDNode *OriginStoreWeights;
754};
755
756void insertModuleCtor(Module &M) {
757 getOrCreateSanitizerCtorAndInitFunctions(
758 M, CtorName: kMsanModuleCtorName, InitName: kMsanInitName,
759 /*InitArgTypes=*/{},
760 /*InitArgs=*/{},
761 // This callback is invoked when the functions are created the first
762 // time. Hook them into the global ctors list in that case:
763 FunctionsCreatedCallback: [&](Function *Ctor, FunctionCallee) {
764 if (!ClWithComdat) {
765 appendToGlobalCtors(M, F: Ctor, Priority: 0);
766 return;
767 }
768 Comdat *MsanCtorComdat = M.getOrInsertComdat(Name: kMsanModuleCtorName);
769 Ctor->setComdat(MsanCtorComdat);
770 appendToGlobalCtors(M, F: Ctor, Priority: 0, Data: Ctor);
771 });
772}
773
774template <class T> T getOptOrDefault(const cl::opt<T> &Opt, T Default) {
775 return (Opt.getNumOccurrences() > 0) ? Opt : Default;
776}
777
778} // end anonymous namespace
779
780MemorySanitizerOptions::MemorySanitizerOptions(int TO, bool R, bool K,
781 bool EagerChecks)
782 : Kernel(getOptOrDefault(Opt: ClEnableKmsan, Default: K)),
783 TrackOrigins(getOptOrDefault(Opt: ClTrackOrigins, Default: Kernel ? 2 : TO)),
784 Recover(getOptOrDefault(Opt: ClKeepGoing, Default: Kernel || R)),
785 EagerChecks(getOptOrDefault(Opt: ClEagerChecks, Default: EagerChecks)) {}
786
787PreservedAnalyses MemorySanitizerPass::run(Module &M,
788 ModuleAnalysisManager &AM) {
789 // Return early if nosanitize_memory module flag is present for the module.
790 if (checkIfAlreadyInstrumented(M, Flag: "nosanitize_memory"))
791 return PreservedAnalyses::all();
792 bool Modified = false;
793 if (!Options.Kernel) {
794 insertModuleCtor(M);
795 Modified = true;
796 }
797
798 auto &FAM = AM.getResult<FunctionAnalysisManagerModuleProxy>(IR&: M).getManager();
799 for (Function &F : M) {
800 if (F.empty())
801 continue;
802 MemorySanitizer Msan(*F.getParent(), Options);
803 Modified |=
804 Msan.sanitizeFunction(F, TLI&: FAM.getResult<TargetLibraryAnalysis>(IR&: F));
805 }
806
807 if (!Modified)
808 return PreservedAnalyses::all();
809
810 PreservedAnalyses PA = PreservedAnalyses::none();
811 // GlobalsAA is considered stateless and does not get invalidated unless
812 // explicitly invalidated; PreservedAnalyses::none() is not enough. Sanitizers
813 // make changes that require GlobalsAA to be invalidated.
814 PA.abandon<GlobalsAA>();
815 return PA;
816}
817
818void MemorySanitizerPass::printPipeline(
819 raw_ostream &OS, function_ref<StringRef(StringRef)> MapClassName2PassName) {
820 static_cast<PassInfoMixin<MemorySanitizerPass> *>(this)->printPipeline(
821 OS, MapClassName2PassName);
822 OS << '<';
823 if (Options.Recover)
824 OS << "recover;";
825 if (Options.Kernel)
826 OS << "kernel;";
827 if (Options.EagerChecks)
828 OS << "eager-checks;";
829 OS << "track-origins=" << Options.TrackOrigins;
830 OS << '>';
831}
832
833/// Create a non-const global initialized with the given string.
834///
835/// Creates a writable global for Str so that we can pass it to the
836/// run-time lib. Runtime uses first 4 bytes of the string to store the
837/// frame ID, so the string needs to be mutable.
838static GlobalVariable *createPrivateConstGlobalForString(Module &M,
839 StringRef Str) {
840 Constant *StrConst = ConstantDataArray::getString(Context&: M.getContext(), Initializer: Str);
841 return new GlobalVariable(M, StrConst->getType(), /*isConstant=*/true,
842 GlobalValue::PrivateLinkage, StrConst, "");
843}
844
845template <typename... ArgsTy>
846FunctionCallee
847MemorySanitizer::getOrInsertMsanMetadataFunction(Module &M, StringRef Name,
848 ArgsTy... Args) {
849 if (TargetTriple.getArch() == Triple::systemz) {
850 // SystemZ ABI: shadow/origin pair is returned via a hidden parameter.
851 return M.getOrInsertFunction(Name, Type::getVoidTy(C&: *C), PtrTy,
852 std::forward<ArgsTy>(Args)...);
853 }
854
855 return M.getOrInsertFunction(Name, MsanMetadata,
856 std::forward<ArgsTy>(Args)...);
857}
858
859/// Create KMSAN API callbacks.
860void MemorySanitizer::createKernelApi(Module &M, const TargetLibraryInfo &TLI) {
861 IRBuilder<> IRB(M);
862
863 // These will be initialized in insertKmsanPrologue().
864 RetvalTLS = nullptr;
865 RetvalOriginTLS = nullptr;
866 ParamTLS = nullptr;
867 ParamOriginTLS = nullptr;
868 VAArgTLS = nullptr;
869 VAArgOriginTLS = nullptr;
870 VAArgOverflowSizeTLS = nullptr;
871
872 WarningFn = M.getOrInsertFunction(Name: "__msan_warning",
873 AttributeList: TLI.getAttrList(C, ArgNos: {0}, /*Signed=*/false),
874 RetTy: IRB.getVoidTy(), Args: IRB.getInt32Ty());
875
876 // Requests the per-task context state (kmsan_context_state*) from the
877 // runtime library.
878 MsanContextStateTy = StructType::get(
879 elt1: ArrayType::get(ElementType: IRB.getInt64Ty(), NumElements: kParamTLSSize / 8),
880 elts: ArrayType::get(ElementType: IRB.getInt64Ty(), NumElements: kRetvalTLSSize / 8),
881 elts: ArrayType::get(ElementType: IRB.getInt64Ty(), NumElements: kParamTLSSize / 8),
882 elts: ArrayType::get(ElementType: IRB.getInt64Ty(), NumElements: kParamTLSSize / 8), /* va_arg_origin */
883 elts: IRB.getInt64Ty(), elts: ArrayType::get(ElementType: OriginTy, NumElements: kParamTLSSize / 4), elts: OriginTy,
884 elts: OriginTy);
885 MsanGetContextStateFn =
886 M.getOrInsertFunction(Name: "__msan_get_context_state", RetTy: PtrTy);
887
888 MsanMetadata = StructType::get(elt1: PtrTy, elts: PtrTy);
889
890 for (int ind = 0, size = 1; ind < 4; ind++, size <<= 1) {
891 std::string name_load =
892 "__msan_metadata_ptr_for_load_" + std::to_string(val: size);
893 std::string name_store =
894 "__msan_metadata_ptr_for_store_" + std::to_string(val: size);
895 MsanMetadataPtrForLoad_1_8[ind] =
896 getOrInsertMsanMetadataFunction(M, Name: name_load, Args: PtrTy);
897 MsanMetadataPtrForStore_1_8[ind] =
898 getOrInsertMsanMetadataFunction(M, Name: name_store, Args: PtrTy);
899 }
900
901 MsanMetadataPtrForLoadN = getOrInsertMsanMetadataFunction(
902 M, Name: "__msan_metadata_ptr_for_load_n", Args: PtrTy, Args: IntptrTy);
903 MsanMetadataPtrForStoreN = getOrInsertMsanMetadataFunction(
904 M, Name: "__msan_metadata_ptr_for_store_n", Args: PtrTy, Args: IntptrTy);
905
906 // Functions for poisoning and unpoisoning memory.
907 MsanPoisonAllocaFn = M.getOrInsertFunction(
908 Name: "__msan_poison_alloca", RetTy: IRB.getVoidTy(), Args: PtrTy, Args: IntptrTy, Args: PtrTy);
909 MsanUnpoisonAllocaFn = M.getOrInsertFunction(
910 Name: "__msan_unpoison_alloca", RetTy: IRB.getVoidTy(), Args: PtrTy, Args: IntptrTy);
911}
912
913static Constant *getOrInsertGlobal(Module &M, StringRef Name, Type *Ty) {
914 return M.getOrInsertGlobal(Name, Ty, CreateGlobalCallback: [&] {
915 return new GlobalVariable(M, Ty, false, GlobalVariable::ExternalLinkage,
916 nullptr, Name, nullptr,
917 GlobalVariable::InitialExecTLSModel);
918 });
919}
920
921/// Insert declarations for userspace-specific functions and globals.
922void MemorySanitizer::createUserspaceApi(Module &M,
923 const TargetLibraryInfo &TLI) {
924 IRBuilder<> IRB(M);
925
926 // Create the callback.
927 // FIXME: this function should have "Cold" calling conv,
928 // which is not yet implemented.
929 if (TrackOrigins) {
930 StringRef WarningFnName = Recover ? "__msan_warning_with_origin"
931 : "__msan_warning_with_origin_noreturn";
932 WarningFn = M.getOrInsertFunction(Name: WarningFnName,
933 AttributeList: TLI.getAttrList(C, ArgNos: {0}, /*Signed=*/false),
934 RetTy: IRB.getVoidTy(), Args: IRB.getInt32Ty());
935 } else {
936 StringRef WarningFnName =
937 Recover ? "__msan_warning" : "__msan_warning_noreturn";
938 WarningFn = M.getOrInsertFunction(Name: WarningFnName, RetTy: IRB.getVoidTy());
939 }
940
941 // Create the global TLS variables.
942 RetvalTLS =
943 getOrInsertGlobal(M, Name: "__msan_retval_tls",
944 Ty: ArrayType::get(ElementType: IRB.getInt64Ty(), NumElements: kRetvalTLSSize / 8));
945
946 RetvalOriginTLS = getOrInsertGlobal(M, Name: "__msan_retval_origin_tls", Ty: OriginTy);
947
948 ParamTLS =
949 getOrInsertGlobal(M, Name: "__msan_param_tls",
950 Ty: ArrayType::get(ElementType: IRB.getInt64Ty(), NumElements: kParamTLSSize / 8));
951
952 ParamOriginTLS =
953 getOrInsertGlobal(M, Name: "__msan_param_origin_tls",
954 Ty: ArrayType::get(ElementType: OriginTy, NumElements: kParamTLSSize / 4));
955
956 VAArgTLS =
957 getOrInsertGlobal(M, Name: "__msan_va_arg_tls",
958 Ty: ArrayType::get(ElementType: IRB.getInt64Ty(), NumElements: kParamTLSSize / 8));
959
960 VAArgOriginTLS =
961 getOrInsertGlobal(M, Name: "__msan_va_arg_origin_tls",
962 Ty: ArrayType::get(ElementType: OriginTy, NumElements: kParamTLSSize / 4));
963
964 VAArgOverflowSizeTLS = getOrInsertGlobal(M, Name: "__msan_va_arg_overflow_size_tls",
965 Ty: IRB.getIntPtrTy(DL: M.getDataLayout()));
966
967 for (size_t AccessSizeIndex = 0; AccessSizeIndex < kNumberOfAccessSizes;
968 AccessSizeIndex++) {
969 unsigned AccessSize = 1 << AccessSizeIndex;
970 std::string FunctionName = "__msan_maybe_warning_" + itostr(X: AccessSize);
971 MaybeWarningFn[AccessSizeIndex] = M.getOrInsertFunction(
972 Name: FunctionName, AttributeList: TLI.getAttrList(C, ArgNos: {0, 1}, /*Signed=*/false),
973 RetTy: IRB.getVoidTy(), Args: IRB.getIntNTy(N: AccessSize * 8), Args: IRB.getInt32Ty());
974 MaybeWarningVarSizeFn = M.getOrInsertFunction(
975 Name: "__msan_maybe_warning_N", AttributeList: TLI.getAttrList(C, ArgNos: {}, /*Signed=*/false),
976 RetTy: IRB.getVoidTy(), Args: PtrTy, Args: IRB.getInt64Ty(), Args: IRB.getInt32Ty());
977 FunctionName = "__msan_maybe_store_origin_" + itostr(X: AccessSize);
978 MaybeStoreOriginFn[AccessSizeIndex] = M.getOrInsertFunction(
979 Name: FunctionName, AttributeList: TLI.getAttrList(C, ArgNos: {0, 2}, /*Signed=*/false),
980 RetTy: IRB.getVoidTy(), Args: IRB.getIntNTy(N: AccessSize * 8), Args: PtrTy,
981 Args: IRB.getInt32Ty());
982 }
983
984 MsanSetAllocaOriginWithDescriptionFn =
985 M.getOrInsertFunction(Name: "__msan_set_alloca_origin_with_descr",
986 RetTy: IRB.getVoidTy(), Args: PtrTy, Args: IntptrTy, Args: PtrTy, Args: PtrTy);
987 MsanSetAllocaOriginNoDescriptionFn =
988 M.getOrInsertFunction(Name: "__msan_set_alloca_origin_no_descr",
989 RetTy: IRB.getVoidTy(), Args: PtrTy, Args: IntptrTy, Args: PtrTy);
990 MsanPoisonStackFn = M.getOrInsertFunction(Name: "__msan_poison_stack",
991 RetTy: IRB.getVoidTy(), Args: PtrTy, Args: IntptrTy);
992}
993
994/// Insert extern declaration of runtime-provided functions and globals.
995void MemorySanitizer::initializeCallbacks(Module &M,
996 const TargetLibraryInfo &TLI) {
997 // Only do this once.
998 if (CallbacksInitialized)
999 return;
1000
1001 IRBuilder<> IRB(M);
1002 // Initialize callbacks that are common for kernel and userspace
1003 // instrumentation.
1004 MsanChainOriginFn = M.getOrInsertFunction(
1005 Name: "__msan_chain_origin",
1006 AttributeList: TLI.getAttrList(C, ArgNos: {0}, /*Signed=*/false, /*Ret=*/true), RetTy: IRB.getInt32Ty(),
1007 Args: IRB.getInt32Ty());
1008 MsanSetOriginFn = M.getOrInsertFunction(
1009 Name: "__msan_set_origin", AttributeList: TLI.getAttrList(C, ArgNos: {2}, /*Signed=*/false),
1010 RetTy: IRB.getVoidTy(), Args: PtrTy, Args: IntptrTy, Args: IRB.getInt32Ty());
1011 MemmoveFn =
1012 M.getOrInsertFunction(Name: "__msan_memmove", RetTy: PtrTy, Args: PtrTy, Args: PtrTy, Args: IntptrTy);
1013 MemcpyFn =
1014 M.getOrInsertFunction(Name: "__msan_memcpy", RetTy: PtrTy, Args: PtrTy, Args: PtrTy, Args: IntptrTy);
1015 MemsetFn = M.getOrInsertFunction(Name: "__msan_memset",
1016 AttributeList: TLI.getAttrList(C, ArgNos: {1}, /*Signed=*/true),
1017 RetTy: PtrTy, Args: PtrTy, Args: IRB.getInt32Ty(), Args: IntptrTy);
1018
1019 MsanInstrumentAsmStoreFn = M.getOrInsertFunction(
1020 Name: "__msan_instrument_asm_store", RetTy: IRB.getVoidTy(), Args: PtrTy, Args: IntptrTy);
1021
1022 if (CompileKernel) {
1023 createKernelApi(M, TLI);
1024 } else {
1025 createUserspaceApi(M, TLI);
1026 }
1027 CallbacksInitialized = true;
1028}
1029
1030FunctionCallee MemorySanitizer::getKmsanShadowOriginAccessFn(bool isStore,
1031 int size) {
1032 FunctionCallee *Fns =
1033 isStore ? MsanMetadataPtrForStore_1_8 : MsanMetadataPtrForLoad_1_8;
1034 switch (size) {
1035 case 1:
1036 return Fns[0];
1037 case 2:
1038 return Fns[1];
1039 case 4:
1040 return Fns[2];
1041 case 8:
1042 return Fns[3];
1043 default:
1044 return nullptr;
1045 }
1046}
1047
1048/// Module-level initialization.
1049///
1050/// inserts a call to __msan_init to the module's constructor list.
1051void MemorySanitizer::initializeModule(Module &M) {
1052 auto &DL = M.getDataLayout();
1053
1054 TargetTriple = M.getTargetTriple();
1055
1056 bool ShadowPassed = ClShadowBase.getNumOccurrences() > 0;
1057 bool OriginPassed = ClOriginBase.getNumOccurrences() > 0;
1058 // Check the overrides first
1059 if (ShadowPassed || OriginPassed) {
1060 CustomMapParams.AndMask = ClAndMask;
1061 CustomMapParams.XorMask = ClXorMask;
1062 CustomMapParams.ShadowBase = ClShadowBase;
1063 CustomMapParams.OriginBase = ClOriginBase;
1064 MapParams = &CustomMapParams;
1065 } else {
1066 switch (TargetTriple.getOS()) {
1067 case Triple::FreeBSD:
1068 switch (TargetTriple.getArch()) {
1069 case Triple::aarch64:
1070 MapParams = FreeBSD_ARM_MemoryMapParams.bits64;
1071 break;
1072 case Triple::x86_64:
1073 MapParams = FreeBSD_X86_MemoryMapParams.bits64;
1074 break;
1075 case Triple::x86:
1076 MapParams = FreeBSD_X86_MemoryMapParams.bits32;
1077 break;
1078 default:
1079 report_fatal_error(reason: "unsupported architecture");
1080 }
1081 break;
1082 case Triple::NetBSD:
1083 switch (TargetTriple.getArch()) {
1084 case Triple::x86_64:
1085 MapParams = NetBSD_X86_MemoryMapParams.bits64;
1086 break;
1087 default:
1088 report_fatal_error(reason: "unsupported architecture");
1089 }
1090 break;
1091 case Triple::Linux:
1092 switch (TargetTriple.getArch()) {
1093 case Triple::x86_64:
1094 MapParams = Linux_X86_MemoryMapParams.bits64;
1095 break;
1096 case Triple::x86:
1097 MapParams = Linux_X86_MemoryMapParams.bits32;
1098 break;
1099 case Triple::mips64:
1100 case Triple::mips64el:
1101 MapParams = Linux_MIPS_MemoryMapParams.bits64;
1102 break;
1103 case Triple::ppc64:
1104 case Triple::ppc64le:
1105 MapParams = Linux_PowerPC_MemoryMapParams.bits64;
1106 break;
1107 case Triple::systemz:
1108 MapParams = Linux_S390_MemoryMapParams.bits64;
1109 break;
1110 case Triple::aarch64:
1111 case Triple::aarch64_be:
1112 MapParams = Linux_ARM_MemoryMapParams.bits64;
1113 break;
1114 case Triple::loongarch64:
1115 MapParams = Linux_LoongArch_MemoryMapParams.bits64;
1116 break;
1117 case Triple::hexagon:
1118 MapParams = Linux_Hexagon_MemoryMapParams_P.bits32;
1119 break;
1120 default:
1121 report_fatal_error(reason: "unsupported architecture");
1122 }
1123 break;
1124 default:
1125 report_fatal_error(reason: "unsupported operating system");
1126 }
1127 }
1128
1129 C = &(M.getContext());
1130 IRBuilder<> IRB(M);
1131 IntptrTy = IRB.getIntPtrTy(DL);
1132 OriginTy = IRB.getInt32Ty();
1133 PtrTy = IRB.getPtrTy();
1134
1135 ColdCallWeights = MDBuilder(*C).createUnlikelyBranchWeights();
1136 OriginStoreWeights = MDBuilder(*C).createUnlikelyBranchWeights();
1137
1138 if (!CompileKernel) {
1139 if (TrackOrigins)
1140 M.getOrInsertGlobal(Name: "__msan_track_origins", Ty: IRB.getInt32Ty(), CreateGlobalCallback: [&] {
1141 return new GlobalVariable(
1142 M, IRB.getInt32Ty(), true, GlobalValue::WeakODRLinkage,
1143 IRB.getInt32(C: TrackOrigins), "__msan_track_origins");
1144 });
1145
1146 if (Recover)
1147 M.getOrInsertGlobal(Name: "__msan_keep_going", Ty: IRB.getInt32Ty(), CreateGlobalCallback: [&] {
1148 return new GlobalVariable(M, IRB.getInt32Ty(), true,
1149 GlobalValue::WeakODRLinkage,
1150 IRB.getInt32(C: Recover), "__msan_keep_going");
1151 });
1152 }
1153}
1154
1155namespace {
1156
1157/// A helper class that handles instrumentation of VarArg
1158/// functions on a particular platform.
1159///
1160/// Implementations are expected to insert the instrumentation
1161/// necessary to propagate argument shadow through VarArg function
1162/// calls. Visit* methods are called during an InstVisitor pass over
1163/// the function, and should avoid creating new basic blocks. A new
1164/// instance of this class is created for each instrumented function.
1165struct VarArgHelper {
1166 virtual ~VarArgHelper() = default;
1167
1168 /// Visit a CallBase.
1169 virtual void visitCallBase(CallBase &CB, IRBuilder<> &IRB) = 0;
1170
1171 /// Visit a va_start call.
1172 virtual void visitVAStartInst(VAStartInst &I) = 0;
1173
1174 /// Visit a va_copy call.
1175 virtual void visitVACopyInst(VACopyInst &I) = 0;
1176
1177 /// Finalize function instrumentation.
1178 ///
1179 /// This method is called after visiting all interesting (see above)
1180 /// instructions in a function.
1181 virtual void finalizeInstrumentation() = 0;
1182};
1183
1184struct MemorySanitizerVisitor;
1185
1186} // end anonymous namespace
1187
1188static VarArgHelper *CreateVarArgHelper(Function &Func, MemorySanitizer &Msan,
1189 MemorySanitizerVisitor &Visitor);
1190
1191static unsigned TypeSizeToSizeIndex(TypeSize TS) {
1192 if (TS.isScalable())
1193 // Scalable types unconditionally take slowpaths.
1194 return kNumberOfAccessSizes;
1195 unsigned TypeSizeFixed = TS.getFixedValue();
1196 if (TypeSizeFixed <= 8)
1197 return 0;
1198 return Log2_32_Ceil(Value: (TypeSizeFixed + 7) / 8);
1199}
1200
1201namespace {
1202
1203/// Helper class to attach debug information of the given instruction onto new
1204/// instructions inserted after.
1205class NextNodeIRBuilder : public IRBuilder<> {
1206public:
1207 explicit NextNodeIRBuilder(Instruction *IP) : IRBuilder<>(IP->getNextNode()) {
1208 SetCurrentDebugLocation(IP->getDebugLoc());
1209 }
1210};
1211
1212/// This class does all the work for a given function. Store and Load
1213/// instructions store and load corresponding shadow and origin
1214/// values. Most instructions propagate shadow from arguments to their
1215/// return values. Certain instructions (most importantly, BranchInst)
1216/// test their argument shadow and print reports (with a runtime call) if it's
1217/// non-zero.
1218struct MemorySanitizerVisitor : public InstVisitor<MemorySanitizerVisitor> {
1219 Function &F;
1220 MemorySanitizer &MS;
1221 SmallVector<PHINode *, 16> ShadowPHINodes, OriginPHINodes;
1222 ValueMap<Value *, Value *> ShadowMap, OriginMap;
1223 std::unique_ptr<VarArgHelper> VAHelper;
1224 const TargetLibraryInfo *TLI;
1225 Instruction *FnPrologueEnd;
1226 SmallVector<Instruction *, 16> Instructions;
1227
1228 // The following flags disable parts of MSan instrumentation based on
1229 // exclusion list contents and command-line options.
1230 bool InsertChecks;
1231 bool PropagateShadow;
1232 bool PoisonStack;
1233 bool PoisonUndef;
1234 bool PoisonUndefVectors;
1235
1236 struct ShadowOriginAndInsertPoint {
1237 Value *Shadow;
1238 Value *Origin;
1239 Instruction *OrigIns;
1240
1241 ShadowOriginAndInsertPoint(Value *S, Value *O, Instruction *I)
1242 : Shadow(S), Origin(O), OrigIns(I) {}
1243 };
1244 SmallVector<ShadowOriginAndInsertPoint, 16> InstrumentationList;
1245 DenseMap<const DILocation *, int> LazyWarningDebugLocationCount;
1246 SmallSetVector<AllocaInst *, 16> AllocaSet;
1247 SmallVector<std::pair<IntrinsicInst *, AllocaInst *>, 16> LifetimeStartList;
1248 SmallVector<StoreInst *, 16> StoreList;
1249 int64_t SplittableBlocksCount = 0;
1250
1251 MemorySanitizerVisitor(Function &F, MemorySanitizer &MS,
1252 const TargetLibraryInfo &TLI)
1253 : F(F), MS(MS), VAHelper(CreateVarArgHelper(Func&: F, Msan&: MS, Visitor&: *this)), TLI(&TLI) {
1254 bool SanitizeFunction =
1255 F.hasFnAttribute(Kind: Attribute::SanitizeMemory) && !ClDisableChecks;
1256 InsertChecks = SanitizeFunction;
1257 PropagateShadow = SanitizeFunction;
1258 PoisonStack = SanitizeFunction && ClPoisonStack;
1259 PoisonUndef = SanitizeFunction && ClPoisonUndef;
1260 PoisonUndefVectors = SanitizeFunction && ClPoisonUndefVectors;
1261
1262 // In the presence of unreachable blocks, we may see Phi nodes with
1263 // incoming nodes from such blocks. Since InstVisitor skips unreachable
1264 // blocks, such nodes will not have any shadow value associated with them.
1265 // It's easier to remove unreachable blocks than deal with missing shadow.
1266 removeUnreachableBlocks(F);
1267
1268 MS.initializeCallbacks(M&: *F.getParent(), TLI);
1269 FnPrologueEnd =
1270 IRBuilder<>(F.getEntryBlock().getFirstNonPHIIt())
1271 .CreateIntrinsicWithoutFolding(ID: Intrinsic::donothing, Args: {});
1272
1273 if (MS.CompileKernel) {
1274 IRBuilder<> IRB(FnPrologueEnd);
1275 insertKmsanPrologue(IRB);
1276 }
1277
1278 LLVM_DEBUG(if (!InsertChecks) dbgs()
1279 << "MemorySanitizer is not inserting checks into '"
1280 << F.getName() << "'\n");
1281 }
1282
1283 bool instrumentWithCalls(Value *V) {
1284 // Constants likely will be eliminated by follow-up passes.
1285 if (isa<Constant>(Val: V))
1286 return false;
1287 ++SplittableBlocksCount;
1288 return ClInstrumentationWithCallThreshold >= 0 &&
1289 SplittableBlocksCount > ClInstrumentationWithCallThreshold;
1290 }
1291
1292 bool isInPrologue(Instruction &I) {
1293 return I.getParent() == FnPrologueEnd->getParent() &&
1294 (&I == FnPrologueEnd || I.comesBefore(Other: FnPrologueEnd));
1295 }
1296
1297 // Creates a new origin and records the stack trace. In general we can call
1298 // this function for any origin manipulation we like. However it will cost
1299 // runtime resources. So use this wisely only if it can provide additional
1300 // information helpful to a user.
1301 Value *updateOrigin(Value *V, IRBuilder<> &IRB) {
1302 if (MS.TrackOrigins <= 1)
1303 return V;
1304 return IRB.CreateCall(Callee: MS.MsanChainOriginFn, Args: V);
1305 }
1306
1307 Value *originToIntptr(IRBuilder<> &IRB, Value *Origin) {
1308 const DataLayout &DL = F.getDataLayout();
1309 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
1310 if (IntptrSize == kOriginSize)
1311 return Origin;
1312 assert(IntptrSize == kOriginSize * 2);
1313 Origin = IRB.CreateIntCast(V: Origin, DestTy: MS.IntptrTy, /* isSigned */ false);
1314 return IRB.CreateOr(LHS: Origin, RHS: IRB.CreateShl(LHS: Origin, RHS: kOriginSize * 8));
1315 }
1316
1317 /// Fill memory range with the given origin value.
1318 void paintOrigin(IRBuilder<> &IRB, Value *Origin, Value *OriginPtr,
1319 TypeSize TS, Align Alignment) {
1320 const DataLayout &DL = F.getDataLayout();
1321 const Align IntptrAlignment = DL.getABITypeAlign(Ty: MS.IntptrTy);
1322 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
1323 assert(IntptrAlignment >= kMinOriginAlignment);
1324 assert(IntptrSize >= kOriginSize);
1325
1326 // Note: The loop based formation works for fixed length vectors too,
1327 // however we prefer to unroll and specialize alignment below.
1328 if (TS.isScalable()) {
1329 Value *Size = IRB.CreateTypeSize(Ty: MS.IntptrTy, Size: TS);
1330 Value *RoundUp =
1331 IRB.CreateAdd(LHS: Size, RHS: ConstantInt::get(Ty: MS.IntptrTy, V: kOriginSize - 1));
1332 Value *End =
1333 IRB.CreateUDiv(LHS: RoundUp, RHS: ConstantInt::get(Ty: MS.IntptrTy, V: kOriginSize));
1334 auto [InsertPt, Index] =
1335 SplitBlockAndInsertSimpleForLoop(End, SplitBefore: IRB.GetInsertPoint());
1336 IRB.SetInsertPoint(InsertPt);
1337
1338 Value *GEP = IRB.CreateGEP(Ty: MS.OriginTy, Ptr: OriginPtr, IdxList: Index);
1339 IRB.CreateAlignedStore(Val: Origin, Ptr: GEP, Align: kMinOriginAlignment);
1340 return;
1341 }
1342
1343 unsigned Size = TS.getFixedValue();
1344
1345 unsigned Ofs = 0;
1346 Align CurrentAlignment = Alignment;
1347 if (Alignment >= IntptrAlignment && IntptrSize > kOriginSize) {
1348 Value *IntptrOrigin = originToIntptr(IRB, Origin);
1349 Value *IntptrOriginPtr = IRB.CreatePointerCast(V: OriginPtr, DestTy: MS.PtrTy);
1350 for (unsigned i = 0; i < Size / IntptrSize; ++i) {
1351 Value *Ptr = i ? IRB.CreateConstGEP1_32(Ty: MS.IntptrTy, Ptr: IntptrOriginPtr, Idx0: i)
1352 : IntptrOriginPtr;
1353 IRB.CreateAlignedStore(Val: IntptrOrigin, Ptr, Align: CurrentAlignment);
1354 Ofs += IntptrSize / kOriginSize;
1355 CurrentAlignment = IntptrAlignment;
1356 }
1357 }
1358
1359 for (unsigned i = Ofs; i < (Size + kOriginSize - 1) / kOriginSize; ++i) {
1360 Value *GEP =
1361 i ? IRB.CreateConstGEP1_32(Ty: MS.OriginTy, Ptr: OriginPtr, Idx0: i) : OriginPtr;
1362 IRB.CreateAlignedStore(Val: Origin, Ptr: GEP, Align: CurrentAlignment);
1363 CurrentAlignment = kMinOriginAlignment;
1364 }
1365 }
1366
1367 void storeOrigin(IRBuilder<> &IRB, Value *Addr, Value *Shadow, Value *Origin,
1368 Value *OriginPtr, Align Alignment) {
1369 const DataLayout &DL = F.getDataLayout();
1370 const Align OriginAlignment = std::max(a: kMinOriginAlignment, b: Alignment);
1371 TypeSize StoreSize = DL.getTypeStoreSize(Ty: Shadow->getType());
1372 // ZExt cannot convert between vector and scalar
1373 Value *ConvertedShadow = convertShadowToScalar(V: Shadow, IRB);
1374 if (auto *ConstantShadow = dyn_cast<Constant>(Val: ConvertedShadow)) {
1375 if (!ClCheckConstantShadow || ConstantShadow->isNullValue()) {
1376 // Origin is not needed: value is initialized or const shadow is
1377 // ignored.
1378 return;
1379 }
1380 if (llvm::isKnownNonZero(V: ConvertedShadow, Q: DL)) {
1381 // Copy origin as the value is definitely uninitialized.
1382 paintOrigin(IRB, Origin: updateOrigin(V: Origin, IRB), OriginPtr, TS: StoreSize,
1383 Alignment: OriginAlignment);
1384 return;
1385 }
1386 // Fallback to runtime check, which still can be optimized out later.
1387 }
1388
1389 TypeSize TypeSizeInBits = DL.getTypeSizeInBits(Ty: ConvertedShadow->getType());
1390 unsigned SizeIndex = TypeSizeToSizeIndex(TS: TypeSizeInBits);
1391 if (instrumentWithCalls(V: ConvertedShadow) &&
1392 SizeIndex < kNumberOfAccessSizes && !MS.CompileKernel) {
1393 FunctionCallee Fn = MS.MaybeStoreOriginFn[SizeIndex];
1394 Value *ConvertedShadow2 =
1395 IRB.CreateZExt(V: ConvertedShadow, DestTy: IRB.getIntNTy(N: 8 * (1 << SizeIndex)));
1396 CallBase *CB = IRB.CreateCall(Callee: Fn, Args: {ConvertedShadow2, Addr, Origin});
1397 CB->addParamAttr(ArgNo: 0, Kind: Attribute::ZExt);
1398 CB->addParamAttr(ArgNo: 2, Kind: Attribute::ZExt);
1399 } else {
1400 Value *Cmp = convertToBool(V: ConvertedShadow, IRB, name: "_mscmp");
1401 Instruction *CheckTerm = SplitBlockAndInsertIfThen(
1402 Cond: Cmp, SplitBefore: &*IRB.GetInsertPoint(), Unreachable: false, BranchWeights: MS.OriginStoreWeights);
1403 IRBuilder<> IRBNew(CheckTerm);
1404 paintOrigin(IRB&: IRBNew, Origin: updateOrigin(V: Origin, IRB&: IRBNew), OriginPtr, TS: StoreSize,
1405 Alignment: OriginAlignment);
1406 }
1407 }
1408
1409 void materializeStores() {
1410 for (StoreInst *SI : StoreList) {
1411 IRBuilder<> IRB(SI);
1412 Value *Val = SI->getValueOperand();
1413 Value *Addr = SI->getPointerOperand();
1414 Value *Shadow = SI->isAtomic() ? getCleanShadow(V: Val) : getShadow(V: Val);
1415 Value *ShadowPtr, *OriginPtr;
1416 Type *ShadowTy = Shadow->getType();
1417 const Align Alignment = SI->getAlign();
1418 const Align OriginAlignment = std::max(a: kMinOriginAlignment, b: Alignment);
1419 std::tie(args&: ShadowPtr, args&: OriginPtr) =
1420 getShadowOriginPtr(Addr, IRB, ShadowTy, Alignment, /*isStore*/ true);
1421
1422 [[maybe_unused]] StoreInst *NewSI =
1423 IRB.CreateAlignedStore(Val: Shadow, Ptr: ShadowPtr, Align: Alignment);
1424 LLVM_DEBUG(dbgs() << " STORE: " << *NewSI << "\n");
1425
1426 if (SI->isAtomic())
1427 SI->setOrdering(addReleaseOrdering(a: SI->getOrdering()));
1428
1429 if (MS.TrackOrigins && !SI->isAtomic())
1430 storeOrigin(IRB, Addr, Shadow, Origin: getOrigin(V: Val), OriginPtr,
1431 Alignment: OriginAlignment);
1432 }
1433 }
1434
1435 // Returns true if Debug Location corresponds to multiple warnings.
1436 bool shouldDisambiguateWarningLocation(const DebugLoc &DebugLoc) {
1437 if (MS.TrackOrigins < 2)
1438 return false;
1439
1440 if (LazyWarningDebugLocationCount.empty())
1441 for (const auto &I : InstrumentationList)
1442 ++LazyWarningDebugLocationCount[I.OrigIns->getDebugLoc()];
1443
1444 return LazyWarningDebugLocationCount[DebugLoc] >= ClDisambiguateWarning;
1445 }
1446
1447 /// Helper function to insert a warning at IRB's current insert point.
1448 void insertWarningFn(IRBuilder<> &IRB, Value *Origin) {
1449 if (!Origin)
1450 Origin = (Value *)IRB.getInt32(C: 0);
1451 assert(Origin->getType()->isIntegerTy());
1452
1453 if (shouldDisambiguateWarningLocation(DebugLoc: IRB.getCurrentDebugLocation())) {
1454 // Try to create additional origin with debug info of the last origin
1455 // instruction. It may provide additional information to the user.
1456 if (Instruction *OI = dyn_cast_or_null<Instruction>(Val: Origin)) {
1457 assert(MS.TrackOrigins);
1458 auto NewDebugLoc = OI->getDebugLoc();
1459 // Origin update with missing or the same debug location provides no
1460 // additional value.
1461 if (NewDebugLoc && NewDebugLoc != IRB.getCurrentDebugLocation()) {
1462 // Insert update just before the check, so we call runtime only just
1463 // before the report.
1464 IRBuilder<> IRBOrigin(&*IRB.GetInsertPoint());
1465 IRBOrigin.SetCurrentDebugLocation(NewDebugLoc);
1466 Origin = updateOrigin(V: Origin, IRB&: IRBOrigin);
1467 }
1468 }
1469 }
1470
1471 if (MS.CompileKernel || MS.TrackOrigins)
1472 IRB.CreateCall(Callee: MS.WarningFn, Args: Origin)->setCannotMerge();
1473 else
1474 IRB.CreateCall(Callee: MS.WarningFn)->setCannotMerge();
1475 // FIXME: Insert UnreachableInst if !MS.Recover?
1476 // This may invalidate some of the following checks and needs to be done
1477 // at the very end.
1478 }
1479
1480 void materializeOneCheck(IRBuilder<> &IRB, Value *ConvertedShadow,
1481 Value *Origin) {
1482 const DataLayout &DL = F.getDataLayout();
1483 TypeSize TypeSizeInBits = DL.getTypeSizeInBits(Ty: ConvertedShadow->getType());
1484 unsigned SizeIndex = TypeSizeToSizeIndex(TS: TypeSizeInBits);
1485 if (instrumentWithCalls(V: ConvertedShadow) && !MS.CompileKernel) {
1486 // ZExt cannot convert between vector and scalar
1487 ConvertedShadow = convertShadowToScalar(V: ConvertedShadow, IRB);
1488 Value *ConvertedShadow2 =
1489 IRB.CreateZExt(V: ConvertedShadow, DestTy: IRB.getIntNTy(N: 8 * (1 << SizeIndex)));
1490
1491 if (SizeIndex < kNumberOfAccessSizes) {
1492 FunctionCallee Fn = MS.MaybeWarningFn[SizeIndex];
1493 CallBase *CB = IRB.CreateCall(
1494 Callee: Fn,
1495 Args: {ConvertedShadow2,
1496 MS.TrackOrigins && Origin ? Origin : (Value *)IRB.getInt32(C: 0)});
1497 CB->addParamAttr(ArgNo: 0, Kind: Attribute::ZExt);
1498 CB->addParamAttr(ArgNo: 1, Kind: Attribute::ZExt);
1499 } else {
1500 FunctionCallee Fn = MS.MaybeWarningVarSizeFn;
1501 Value *ShadowAlloca = IRB.CreateAlloca(Ty: ConvertedShadow2->getType(), AddrSpace: 0u);
1502 IRB.CreateStore(Val: ConvertedShadow2, Ptr: ShadowAlloca);
1503 unsigned ShadowSize = DL.getTypeAllocSize(Ty: ConvertedShadow2->getType());
1504 CallBase *CB = IRB.CreateCall(
1505 Callee: Fn,
1506 Args: {ShadowAlloca, ConstantInt::get(Ty: IRB.getInt64Ty(), V: ShadowSize),
1507 MS.TrackOrigins && Origin ? Origin : (Value *)IRB.getInt32(C: 0)});
1508 CB->addParamAttr(ArgNo: 1, Kind: Attribute::ZExt);
1509 CB->addParamAttr(ArgNo: 2, Kind: Attribute::ZExt);
1510 }
1511 } else {
1512 Value *Cmp = convertToBool(V: ConvertedShadow, IRB, name: "_mscmp");
1513 Instruction *CheckTerm = SplitBlockAndInsertIfThen(
1514 Cond: Cmp, SplitBefore: &*IRB.GetInsertPoint(),
1515 /* Unreachable */ !MS.Recover, BranchWeights: MS.ColdCallWeights);
1516
1517 IRB.SetInsertPoint(CheckTerm);
1518 insertWarningFn(IRB, Origin);
1519 LLVM_DEBUG(dbgs() << " CHECK: " << *Cmp << "\n");
1520 }
1521 }
1522
1523 void materializeInstructionChecks(
1524 ArrayRef<ShadowOriginAndInsertPoint> InstructionChecks) {
1525 const DataLayout &DL = F.getDataLayout();
1526 // Disable combining in some cases. TrackOrigins checks each shadow to pick
1527 // correct origin.
1528 bool Combine = !MS.TrackOrigins;
1529 Instruction *Instruction = InstructionChecks.front().OrigIns;
1530 Value *Shadow = nullptr;
1531 for (const auto &ShadowData : InstructionChecks) {
1532 assert(ShadowData.OrigIns == Instruction);
1533 IRBuilder<> IRB(Instruction);
1534
1535 Value *ConvertedShadow = ShadowData.Shadow;
1536
1537 if (auto *ConstantShadow = dyn_cast<Constant>(Val: ConvertedShadow)) {
1538 if (!ClCheckConstantShadow || ConstantShadow->isNullValue()) {
1539 // Skip, value is initialized or const shadow is ignored.
1540 continue;
1541 }
1542 if (llvm::isKnownNonZero(V: ConvertedShadow, Q: DL)) {
1543 // Report as the value is definitely uninitialized.
1544 insertWarningFn(IRB, Origin: ShadowData.Origin);
1545 if (!MS.Recover)
1546 return; // Always fail and stop here, not need to check the rest.
1547 // Skip entire instruction,
1548 continue;
1549 }
1550 // Fallback to runtime check, which still can be optimized out later.
1551 }
1552
1553 if (!Combine) {
1554 materializeOneCheck(IRB, ConvertedShadow, Origin: ShadowData.Origin);
1555 continue;
1556 }
1557
1558 if (!Shadow) {
1559 Shadow = ConvertedShadow;
1560 continue;
1561 }
1562
1563 Shadow = convertToBool(V: Shadow, IRB, name: "_mscmp");
1564 ConvertedShadow = convertToBool(V: ConvertedShadow, IRB, name: "_mscmp");
1565 Shadow = IRB.CreateOr(LHS: Shadow, RHS: ConvertedShadow, Name: "_msor");
1566 }
1567
1568 if (Shadow) {
1569 assert(Combine);
1570 IRBuilder<> IRB(Instruction);
1571 materializeOneCheck(IRB, ConvertedShadow: Shadow, Origin: nullptr);
1572 }
1573 }
1574
1575 static bool isAArch64SVCount(Type *Ty) {
1576 if (TargetExtType *TTy = dyn_cast<TargetExtType>(Val: Ty))
1577 return TTy->getName() == "aarch64.svcount";
1578 return false;
1579 }
1580
1581 // This is intended to match the "AArch64 Predicate-as-Counter Type" (aka
1582 // 'target("aarch64.svcount")', but not e.g., <vscale x 4 x i32>.
1583 static bool isScalableNonVectorType(Type *Ty) {
1584 if (!isAArch64SVCount(Ty))
1585 LLVM_DEBUG(dbgs() << "isScalableNonVectorType: Unexpected type " << *Ty
1586 << "\n");
1587
1588 return Ty->isScalableTy() && !isa<VectorType>(Val: Ty);
1589 }
1590
1591 void materializeChecks() {
1592#ifndef NDEBUG
1593 // For assert below.
1594 SmallPtrSet<Instruction *, 16> Done;
1595#endif
1596
1597 for (auto I = InstrumentationList.begin();
1598 I != InstrumentationList.end();) {
1599 auto OrigIns = I->OrigIns;
1600 // Checks are grouped by the original instruction. We call all
1601 // `insertShadowCheck` for an instruction at once.
1602 assert(Done.insert(OrigIns).second);
1603 auto J = std::find_if(first: I + 1, last: InstrumentationList.end(),
1604 pred: [OrigIns](const ShadowOriginAndInsertPoint &R) {
1605 return OrigIns != R.OrigIns;
1606 });
1607 // Process all checks of instruction at once.
1608 materializeInstructionChecks(InstructionChecks: ArrayRef<ShadowOriginAndInsertPoint>(I, J));
1609 I = J;
1610 }
1611
1612 LLVM_DEBUG(dbgs() << "DONE:\n" << F);
1613 }
1614
1615 // Returns the last instruction in the new prologue
1616 void insertKmsanPrologue(IRBuilder<> &IRB) {
1617 Value *ContextState = IRB.CreateCall(Callee: MS.MsanGetContextStateFn, Args: {});
1618 Constant *Zero = IRB.getInt32(C: 0);
1619 MS.ParamTLS = IRB.CreateGEP(Ty: MS.MsanContextStateTy, Ptr: ContextState,
1620 IdxList: {Zero, IRB.getInt32(C: 0)}, Name: "param_shadow");
1621 MS.RetvalTLS = IRB.CreateGEP(Ty: MS.MsanContextStateTy, Ptr: ContextState,
1622 IdxList: {Zero, IRB.getInt32(C: 1)}, Name: "retval_shadow");
1623 MS.VAArgTLS = IRB.CreateGEP(Ty: MS.MsanContextStateTy, Ptr: ContextState,
1624 IdxList: {Zero, IRB.getInt32(C: 2)}, Name: "va_arg_shadow");
1625 MS.VAArgOriginTLS = IRB.CreateGEP(Ty: MS.MsanContextStateTy, Ptr: ContextState,
1626 IdxList: {Zero, IRB.getInt32(C: 3)}, Name: "va_arg_origin");
1627 MS.VAArgOverflowSizeTLS =
1628 IRB.CreateGEP(Ty: MS.MsanContextStateTy, Ptr: ContextState,
1629 IdxList: {Zero, IRB.getInt32(C: 4)}, Name: "va_arg_overflow_size");
1630 MS.ParamOriginTLS = IRB.CreateGEP(Ty: MS.MsanContextStateTy, Ptr: ContextState,
1631 IdxList: {Zero, IRB.getInt32(C: 5)}, Name: "param_origin");
1632 MS.RetvalOriginTLS =
1633 IRB.CreateGEP(Ty: MS.MsanContextStateTy, Ptr: ContextState,
1634 IdxList: {Zero, IRB.getInt32(C: 6)}, Name: "retval_origin");
1635 if (MS.TargetTriple.getArch() == Triple::systemz)
1636 MS.MsanMetadataAlloca = IRB.CreateAlloca(Ty: MS.MsanMetadata, AddrSpace: 0u);
1637 }
1638
1639 /// Add MemorySanitizer instrumentation to a function.
1640 bool runOnFunction() {
1641 // Iterate all BBs in depth-first order and create shadow instructions
1642 // for all instructions (where applicable).
1643 // For PHI nodes we create dummy shadow PHIs which will be finalized later.
1644 for (BasicBlock *BB : depth_first(G: FnPrologueEnd->getParent()))
1645 visit(BB&: *BB);
1646
1647 // `visit` above only collects instructions. Process them after iterating
1648 // CFG to avoid requirement on CFG transformations.
1649 for (Instruction *I : Instructions)
1650 InstVisitor<MemorySanitizerVisitor>::visit(I&: *I);
1651
1652 // Finalize PHI nodes.
1653 for (PHINode *PN : ShadowPHINodes) {
1654 PHINode *PNS = cast<PHINode>(Val: getShadow(V: PN));
1655 PHINode *PNO = MS.TrackOrigins ? cast<PHINode>(Val: getOrigin(V: PN)) : nullptr;
1656 size_t NumValues = PN->getNumIncomingValues();
1657 for (size_t v = 0; v < NumValues; v++) {
1658 PNS->addIncoming(V: getShadow(I: PN, i: v), BB: PN->getIncomingBlock(i: v));
1659 if (PNO)
1660 PNO->addIncoming(V: getOrigin(I: PN, i: v), BB: PN->getIncomingBlock(i: v));
1661 }
1662 }
1663
1664 VAHelper->finalizeInstrumentation();
1665
1666 // Poison llvm.lifetime.start intrinsics, if we haven't fallen back to
1667 // instrumenting only allocas.
1668 if (ClHandleLifetimeIntrinsics) {
1669 for (auto Item : LifetimeStartList) {
1670 instrumentAlloca(I&: *Item.second, InsPoint: Item.first);
1671 AllocaSet.remove(X: Item.second);
1672 }
1673 }
1674 // Poison the allocas for which we didn't instrument the corresponding
1675 // lifetime intrinsics.
1676 for (AllocaInst *AI : AllocaSet)
1677 instrumentAlloca(I&: *AI);
1678
1679 // Insert shadow value checks.
1680 materializeChecks();
1681
1682 // Delayed instrumentation of StoreInst.
1683 // This may not add new address checks.
1684 materializeStores();
1685
1686 return true;
1687 }
1688
1689 /// Compute the shadow type that corresponds to a given Value.
1690 Type *getShadowTy(Value *V) { return getShadowTy(OrigTy: V->getType()); }
1691
1692 /// Compute the shadow type that corresponds to a given Type.
1693 Type *getShadowTy(Type *OrigTy) {
1694 if (!OrigTy->isSized()) {
1695 return nullptr;
1696 }
1697 // For integer type, shadow is the same as the original type.
1698 // This may return weird-sized types like i1.
1699 if (IntegerType *IT = dyn_cast<IntegerType>(Val: OrigTy))
1700 return IT;
1701 const DataLayout &DL = F.getDataLayout();
1702 if (VectorType *VT = dyn_cast<VectorType>(Val: OrigTy)) {
1703 uint32_t EltSize = DL.getTypeSizeInBits(Ty: VT->getElementType());
1704 return VectorType::get(ElementType: IntegerType::get(C&: *MS.C, NumBits: EltSize),
1705 EC: VT->getElementCount());
1706 }
1707 if (ArrayType *AT = dyn_cast<ArrayType>(Val: OrigTy)) {
1708 return ArrayType::get(ElementType: getShadowTy(OrigTy: AT->getElementType()),
1709 NumElements: AT->getNumElements());
1710 }
1711 if (StructType *ST = dyn_cast<StructType>(Val: OrigTy)) {
1712 SmallVector<Type *, 4> Elements;
1713 for (unsigned i = 0, n = ST->getNumElements(); i < n; i++)
1714 Elements.push_back(Elt: getShadowTy(OrigTy: ST->getElementType(N: i)));
1715 StructType *Res = StructType::get(Context&: *MS.C, Elements, isPacked: ST->isPacked());
1716 LLVM_DEBUG(dbgs() << "getShadowTy: " << *ST << " ===> " << *Res << "\n");
1717 return Res;
1718 }
1719 if (isScalableNonVectorType(Ty: OrigTy)) {
1720 LLVM_DEBUG(dbgs() << "getShadowTy: Scalable non-vector type: " << *OrigTy
1721 << "\n");
1722 return OrigTy;
1723 }
1724
1725 uint32_t TypeSize = DL.getTypeSizeInBits(Ty: OrigTy);
1726 return IntegerType::get(C&: *MS.C, NumBits: TypeSize);
1727 }
1728
1729 /// Extract combined shadow of struct elements as a bool
1730 Value *collapseStructShadow(StructType *Struct, Value *Shadow,
1731 IRBuilder<> &IRB) {
1732 Value *FalseVal = IRB.getIntN(/* width */ N: 1, /* value */ C: 0);
1733 Value *Aggregator = FalseVal;
1734
1735 for (unsigned Idx = 0; Idx < Struct->getNumElements(); Idx++) {
1736 // Combine by ORing together each element's bool shadow
1737 Value *ShadowItem = IRB.CreateExtractValue(Agg: Shadow, Idxs: Idx);
1738 Value *ShadowBool = convertToBool(V: ShadowItem, IRB);
1739
1740 if (Aggregator != FalseVal)
1741 Aggregator = IRB.CreateOr(LHS: Aggregator, RHS: ShadowBool);
1742 else
1743 Aggregator = ShadowBool;
1744 }
1745
1746 return Aggregator;
1747 }
1748
1749 // Extract combined shadow of array elements
1750 Value *collapseArrayShadow(ArrayType *Array, Value *Shadow,
1751 IRBuilder<> &IRB) {
1752 if (!Array->getNumElements())
1753 return IRB.getIntN(/* width */ N: 1, /* value */ C: 0);
1754
1755 Value *FirstItem = IRB.CreateExtractValue(Agg: Shadow, Idxs: 0);
1756 Value *Aggregator = convertShadowToScalar(V: FirstItem, IRB);
1757
1758 for (unsigned Idx = 1; Idx < Array->getNumElements(); Idx++) {
1759 Value *ShadowItem = IRB.CreateExtractValue(Agg: Shadow, Idxs: Idx);
1760 Value *ShadowInner = convertShadowToScalar(V: ShadowItem, IRB);
1761 Aggregator = IRB.CreateOr(LHS: Aggregator, RHS: ShadowInner);
1762 }
1763 return Aggregator;
1764 }
1765
1766 /// Convert a shadow value to it's flattened variant. The resulting
1767 /// shadow may not necessarily have the same bit width as the input
1768 /// value, but it will always be comparable to zero.
1769 Value *convertShadowToScalar(Value *V, IRBuilder<> &IRB) {
1770 if (StructType *Struct = dyn_cast<StructType>(Val: V->getType()))
1771 return collapseStructShadow(Struct, Shadow: V, IRB);
1772 if (ArrayType *Array = dyn_cast<ArrayType>(Val: V->getType()))
1773 return collapseArrayShadow(Array, Shadow: V, IRB);
1774 if (isa<VectorType>(Val: V->getType())) {
1775 if (isa<ScalableVectorType>(Val: V->getType()))
1776 return convertShadowToScalar(V: IRB.CreateOrReduce(Src: V), IRB);
1777 unsigned BitWidth =
1778 V->getType()->getPrimitiveSizeInBits().getFixedValue();
1779 return IRB.CreateBitCast(V, DestTy: IntegerType::get(C&: *MS.C, NumBits: BitWidth));
1780 }
1781 return V;
1782 }
1783
1784 // Convert a scalar value to an i1 by comparing with 0
1785 Value *convertToBool(Value *V, IRBuilder<> &IRB, const Twine &name = "") {
1786 Type *VTy = V->getType();
1787 if (!VTy->isIntegerTy())
1788 return convertToBool(V: convertShadowToScalar(V, IRB), IRB, name);
1789 if (VTy->getIntegerBitWidth() == 1)
1790 // Just converting a bool to a bool, so do nothing.
1791 return V;
1792 return IRB.CreateICmpNE(LHS: V, RHS: ConstantInt::get(Ty: VTy, V: 0), Name: name);
1793 }
1794
1795 Type *ptrToIntPtrType(Type *PtrTy) const {
1796 if (VectorType *VectTy = dyn_cast<VectorType>(Val: PtrTy)) {
1797 return VectorType::get(ElementType: ptrToIntPtrType(PtrTy: VectTy->getElementType()),
1798 EC: VectTy->getElementCount());
1799 }
1800 assert(PtrTy->isIntOrPtrTy());
1801 return MS.IntptrTy;
1802 }
1803
1804 Type *getPtrToShadowPtrType(Type *IntPtrTy, Type *ShadowTy) const {
1805 if (VectorType *VectTy = dyn_cast<VectorType>(Val: IntPtrTy)) {
1806 return VectorType::get(
1807 ElementType: getPtrToShadowPtrType(IntPtrTy: VectTy->getElementType(), ShadowTy),
1808 EC: VectTy->getElementCount());
1809 }
1810 assert(IntPtrTy == MS.IntptrTy);
1811 return MS.PtrTy;
1812 }
1813
1814 Constant *constToIntPtr(Type *IntPtrTy, uint64_t C) const {
1815 if (VectorType *VectTy = dyn_cast<VectorType>(Val: IntPtrTy)) {
1816 return ConstantVector::getSplat(
1817 EC: VectTy->getElementCount(),
1818 Elt: constToIntPtr(IntPtrTy: VectTy->getElementType(), C));
1819 }
1820 assert(IntPtrTy == MS.IntptrTy);
1821 // TODO: Avoid implicit trunc?
1822 // See https://github.com/llvm/llvm-project/issues/112510.
1823 return ConstantInt::get(Ty: MS.IntptrTy, V: C, /*IsSigned=*/false,
1824 /*ImplicitTrunc=*/true);
1825 }
1826
1827 /// Returns the integer shadow offset that corresponds to a given
1828 /// application address, whereby:
1829 ///
1830 /// Offset = (Addr & ~AndMask) ^ XorMask
1831 /// Shadow = ShadowBase + Offset
1832 /// Origin = (OriginBase + Offset) & ~Alignment
1833 ///
1834 /// Note: for efficiency, many shadow mappings only require use the XorMask
1835 /// and OriginBase; the AndMask and ShadowBase are often zero.
1836 Value *getShadowPtrOffset(Value *Addr, IRBuilder<> &IRB) {
1837 Type *IntptrTy = ptrToIntPtrType(PtrTy: Addr->getType());
1838 Value *OffsetLong = IRB.CreatePointerCast(V: Addr, DestTy: IntptrTy);
1839
1840 if (uint64_t AndMask = MS.MapParams->AndMask)
1841 OffsetLong = IRB.CreateAnd(LHS: OffsetLong, RHS: constToIntPtr(IntPtrTy: IntptrTy, C: ~AndMask));
1842
1843 if (uint64_t XorMask = MS.MapParams->XorMask)
1844 OffsetLong = IRB.CreateXor(LHS: OffsetLong, RHS: constToIntPtr(IntPtrTy: IntptrTy, C: XorMask));
1845 return OffsetLong;
1846 }
1847
1848 /// Compute the shadow and origin addresses corresponding to a given
1849 /// application address.
1850 ///
1851 /// Shadow = ShadowBase + Offset
1852 /// Origin = (OriginBase + Offset) & ~3ULL
1853 /// Addr can be a ptr or <N x ptr>. In both cases ShadowTy the shadow type of
1854 /// a single pointee.
1855 /// Returns <shadow_ptr, origin_ptr> or <<N x shadow_ptr>, <N x origin_ptr>>.
1856 std::pair<Value *, Value *>
1857 getShadowOriginPtrUserspace(Value *Addr, IRBuilder<> &IRB, Type *ShadowTy,
1858 MaybeAlign Alignment) {
1859 VectorType *VectTy = dyn_cast<VectorType>(Val: Addr->getType());
1860 if (!VectTy) {
1861 assert(Addr->getType()->isPointerTy());
1862 } else {
1863 assert(VectTy->getElementType()->isPointerTy());
1864 }
1865 Type *IntptrTy = ptrToIntPtrType(PtrTy: Addr->getType());
1866 Value *ShadowOffset = getShadowPtrOffset(Addr, IRB);
1867 Value *ShadowLong = ShadowOffset;
1868 if (uint64_t ShadowBase = MS.MapParams->ShadowBase) {
1869 ShadowLong =
1870 IRB.CreateAdd(LHS: ShadowLong, RHS: constToIntPtr(IntPtrTy: IntptrTy, C: ShadowBase));
1871 }
1872 Value *ShadowPtr = IRB.CreateIntToPtr(
1873 V: ShadowLong, DestTy: getPtrToShadowPtrType(IntPtrTy: IntptrTy, ShadowTy));
1874
1875 Value *OriginPtr = nullptr;
1876 if (MS.TrackOrigins) {
1877 Value *OriginLong = ShadowOffset;
1878 uint64_t OriginBase = MS.MapParams->OriginBase;
1879 if (OriginBase != 0)
1880 OriginLong =
1881 IRB.CreateAdd(LHS: OriginLong, RHS: constToIntPtr(IntPtrTy: IntptrTy, C: OriginBase));
1882 if (!Alignment || *Alignment < kMinOriginAlignment) {
1883 uint64_t Mask = kMinOriginAlignment.value() - 1;
1884 OriginLong = IRB.CreateAnd(LHS: OriginLong, RHS: constToIntPtr(IntPtrTy: IntptrTy, C: ~Mask));
1885 }
1886 OriginPtr = IRB.CreateIntToPtr(
1887 V: OriginLong, DestTy: getPtrToShadowPtrType(IntPtrTy: IntptrTy, ShadowTy: MS.OriginTy));
1888 }
1889 return std::make_pair(x&: ShadowPtr, y&: OriginPtr);
1890 }
1891
1892 template <typename... ArgsTy>
1893 Value *createMetadataCall(IRBuilder<> &IRB, FunctionCallee Callee,
1894 ArgsTy... Args) {
1895 if (MS.TargetTriple.getArch() == Triple::systemz) {
1896 IRB.CreateCall(Callee,
1897 {MS.MsanMetadataAlloca, std::forward<ArgsTy>(Args)...});
1898 return IRB.CreateLoad(Ty: MS.MsanMetadata, Ptr: MS.MsanMetadataAlloca);
1899 }
1900
1901 return IRB.CreateCall(Callee, {std::forward<ArgsTy>(Args)...});
1902 }
1903
1904 std::pair<Value *, Value *> getShadowOriginPtrKernelNoVec(Value *Addr,
1905 IRBuilder<> &IRB,
1906 Type *ShadowTy,
1907 bool isStore) {
1908 Value *ShadowOriginPtrs;
1909 const DataLayout &DL = F.getDataLayout();
1910 TypeSize Size = DL.getTypeStoreSize(Ty: ShadowTy);
1911
1912 FunctionCallee Getter = MS.getKmsanShadowOriginAccessFn(isStore, size: Size);
1913 Value *AddrCast = IRB.CreatePointerCast(V: Addr, DestTy: MS.PtrTy);
1914 if (Getter) {
1915 ShadowOriginPtrs = createMetadataCall(IRB, Callee: Getter, Args: AddrCast);
1916 } else {
1917 Value *SizeVal = ConstantInt::get(Ty: MS.IntptrTy, V: Size);
1918 ShadowOriginPtrs = createMetadataCall(
1919 IRB,
1920 Callee: isStore ? MS.MsanMetadataPtrForStoreN : MS.MsanMetadataPtrForLoadN,
1921 Args: AddrCast, Args: SizeVal);
1922 }
1923 Value *ShadowPtr = IRB.CreateExtractValue(Agg: ShadowOriginPtrs, Idxs: 0);
1924 ShadowPtr = IRB.CreatePointerCast(V: ShadowPtr, DestTy: MS.PtrTy);
1925 Value *OriginPtr = IRB.CreateExtractValue(Agg: ShadowOriginPtrs, Idxs: 1);
1926
1927 return std::make_pair(x&: ShadowPtr, y&: OriginPtr);
1928 }
1929
1930 /// Addr can be a ptr or <N x ptr>. In both cases ShadowTy the shadow type of
1931 /// a single pointee.
1932 /// Returns <shadow_ptr, origin_ptr> or <<N x shadow_ptr>, <N x origin_ptr>>.
1933 std::pair<Value *, Value *> getShadowOriginPtrKernel(Value *Addr,
1934 IRBuilder<> &IRB,
1935 Type *ShadowTy,
1936 bool isStore) {
1937 VectorType *VectTy = dyn_cast<VectorType>(Val: Addr->getType());
1938 if (!VectTy) {
1939 assert(Addr->getType()->isPointerTy());
1940 return getShadowOriginPtrKernelNoVec(Addr, IRB, ShadowTy, isStore);
1941 }
1942
1943 // TODO: Support callbacs with vectors of addresses.
1944 unsigned NumElements = cast<FixedVectorType>(Val: VectTy)->getNumElements();
1945 Value *ShadowPtrs = ConstantInt::getNullValue(
1946 Ty: FixedVectorType::get(ElementType: IRB.getPtrTy(), NumElts: NumElements));
1947 Value *OriginPtrs = nullptr;
1948 if (MS.TrackOrigins)
1949 OriginPtrs = ConstantInt::getNullValue(
1950 Ty: FixedVectorType::get(ElementType: IRB.getPtrTy(), NumElts: NumElements));
1951 for (unsigned i = 0; i < NumElements; ++i) {
1952 Value *OneAddr =
1953 IRB.CreateExtractElement(Vec: Addr, Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: i));
1954 auto [ShadowPtr, OriginPtr] =
1955 getShadowOriginPtrKernelNoVec(Addr: OneAddr, IRB, ShadowTy, isStore);
1956
1957 ShadowPtrs = IRB.CreateInsertElement(
1958 Vec: ShadowPtrs, NewElt: ShadowPtr, Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: i));
1959 if (MS.TrackOrigins)
1960 OriginPtrs = IRB.CreateInsertElement(
1961 Vec: OriginPtrs, NewElt: OriginPtr, Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: i));
1962 }
1963 return {ShadowPtrs, OriginPtrs};
1964 }
1965
1966 std::pair<Value *, Value *> getShadowOriginPtr(Value *Addr, IRBuilder<> &IRB,
1967 Type *ShadowTy,
1968 MaybeAlign Alignment,
1969 bool isStore) {
1970 if (MS.CompileKernel)
1971 return getShadowOriginPtrKernel(Addr, IRB, ShadowTy, isStore);
1972 return getShadowOriginPtrUserspace(Addr, IRB, ShadowTy, Alignment);
1973 }
1974
1975 /// Compute the shadow address for a given function argument.
1976 ///
1977 /// Shadow = ParamTLS+ArgOffset.
1978 Value *getShadowPtrForArgument(IRBuilder<> &IRB, int ArgOffset) {
1979 return IRB.CreatePtrAdd(Ptr: MS.ParamTLS,
1980 Offset: ConstantInt::get(Ty: MS.IntptrTy, V: ArgOffset), Name: "_msarg");
1981 }
1982
1983 /// Compute the origin address for a given function argument.
1984 Value *getOriginPtrForArgument(IRBuilder<> &IRB, int ArgOffset) {
1985 if (!MS.TrackOrigins)
1986 return nullptr;
1987 return IRB.CreatePtrAdd(Ptr: MS.ParamOriginTLS,
1988 Offset: ConstantInt::get(Ty: MS.IntptrTy, V: ArgOffset),
1989 Name: "_msarg_o");
1990 }
1991
1992 /// Compute the shadow address for a retval.
1993 Value *getShadowPtrForRetval(IRBuilder<> &IRB) {
1994 return IRB.CreatePointerCast(V: MS.RetvalTLS, DestTy: IRB.getPtrTy(AddrSpace: 0), Name: "_msret");
1995 }
1996
1997 /// Compute the origin address for a retval.
1998 Value *getOriginPtrForRetval() {
1999 // We keep a single origin for the entire retval. Might be too optimistic.
2000 return MS.RetvalOriginTLS;
2001 }
2002
2003 /// Set SV to be the shadow value for V.
2004 void setShadow(Value *V, Value *SV) {
2005 assert(!ShadowMap.count(V) && "Values may only have one shadow");
2006 ShadowMap[V] = PropagateShadow ? SV : getCleanShadow(V);
2007 }
2008
2009 /// Set Origin to be the origin value for V.
2010 void setOrigin(Value *V, Value *Origin) {
2011 if (!MS.TrackOrigins)
2012 return;
2013 assert(!OriginMap.count(V) && "Values may only have one origin");
2014 LLVM_DEBUG(dbgs() << "ORIGIN: " << *V << " ==> " << *Origin << "\n");
2015 OriginMap[V] = Origin;
2016 }
2017
2018 Constant *getCleanShadow(Type *OrigTy) {
2019 Type *ShadowTy = getShadowTy(OrigTy);
2020 if (!ShadowTy)
2021 return nullptr;
2022 return Constant::getNullValue(Ty: ShadowTy);
2023 }
2024
2025 /// Create a clean shadow value for a given value.
2026 ///
2027 /// Clean shadow (all zeroes) means all bits of the value are defined
2028 /// (initialized).
2029 Constant *getCleanShadow(Value *V) { return getCleanShadow(OrigTy: V->getType()); }
2030
2031 /// Create a dirty shadow of a given shadow type.
2032 Constant *getPoisonedShadow(Type *ShadowTy) {
2033 assert(ShadowTy);
2034 if (isa<IntegerType>(Val: ShadowTy) || isa<VectorType>(Val: ShadowTy))
2035 return Constant::getAllOnesValue(Ty: ShadowTy);
2036 if (ArrayType *AT = dyn_cast<ArrayType>(Val: ShadowTy)) {
2037 SmallVector<Constant *, 4> Vals(AT->getNumElements(),
2038 getPoisonedShadow(ShadowTy: AT->getElementType()));
2039 return ConstantArray::get(T: AT, V: Vals);
2040 }
2041 if (StructType *ST = dyn_cast<StructType>(Val: ShadowTy)) {
2042 SmallVector<Constant *, 4> Vals;
2043 for (unsigned i = 0, n = ST->getNumElements(); i < n; i++)
2044 Vals.push_back(Elt: getPoisonedShadow(ShadowTy: ST->getElementType(N: i)));
2045 return ConstantStruct::get(T: ST, V: Vals);
2046 }
2047 llvm_unreachable("Unexpected shadow type");
2048 }
2049
2050 /// Create a dirty shadow for a given value.
2051 Constant *getPoisonedShadow(Value *V) {
2052 Type *ShadowTy = getShadowTy(V);
2053 if (!ShadowTy)
2054 return nullptr;
2055 return getPoisonedShadow(ShadowTy);
2056 }
2057
2058 /// Create a clean (zero) origin.
2059 Value *getCleanOrigin() { return Constant::getNullValue(Ty: MS.OriginTy); }
2060
2061 /// Get the shadow value for a given Value.
2062 ///
2063 /// This function either returns the value set earlier with setShadow,
2064 /// or extracts if from ParamTLS (for function arguments).
2065 Value *getShadow(Value *V) {
2066 if (Instruction *I = dyn_cast<Instruction>(Val: V)) {
2067 if (!PropagateShadow || I->getMetadata(KindID: LLVMContext::MD_nosanitize))
2068 return getCleanShadow(V);
2069 // For instructions the shadow is already stored in the map.
2070 Value *Shadow = ShadowMap[V];
2071 if (!Shadow) {
2072 LLVM_DEBUG(dbgs() << "No shadow: " << *V << "\n" << *(I->getParent()));
2073 assert(Shadow && "No shadow for a value");
2074 }
2075 return Shadow;
2076 }
2077 // Handle fully undefined values
2078 // (partially undefined constant vectors are handled later)
2079 if ([[maybe_unused]] UndefValue *U = dyn_cast<UndefValue>(Val: V)) {
2080 Value *AllOnes = (PropagateShadow && PoisonUndef) ? getPoisonedShadow(V)
2081 : getCleanShadow(V);
2082 LLVM_DEBUG(dbgs() << "Undef: " << *U << " ==> " << *AllOnes << "\n");
2083 return AllOnes;
2084 }
2085 if (Argument *A = dyn_cast<Argument>(Val: V)) {
2086 // For arguments we compute the shadow on demand and store it in the map.
2087 Value *&ShadowPtr = ShadowMap[V];
2088 if (ShadowPtr)
2089 return ShadowPtr;
2090 Function *F = A->getParent();
2091 IRBuilder<> EntryIRB(FnPrologueEnd);
2092 unsigned ArgOffset = 0;
2093 const DataLayout &DL = F->getDataLayout();
2094 for (auto &FArg : F->args()) {
2095 if (!FArg.getType()->isSized() || FArg.getType()->isScalableTy()) {
2096 LLVM_DEBUG(dbgs() << (FArg.getType()->isScalableTy()
2097 ? "vscale not fully supported\n"
2098 : "Arg is not sized\n"));
2099 if (A == &FArg) {
2100 ShadowPtr = getCleanShadow(V);
2101 setOrigin(V: A, Origin: getCleanOrigin());
2102 break;
2103 }
2104 continue;
2105 }
2106
2107 unsigned Size = FArg.hasByValAttr()
2108 ? DL.getTypeAllocSize(Ty: FArg.getParamByValType())
2109 : DL.getTypeAllocSize(Ty: FArg.getType());
2110
2111 if (A == &FArg) {
2112 bool Overflow = ArgOffset + Size > kParamTLSSize;
2113 if (FArg.hasByValAttr()) {
2114 // ByVal pointer itself has clean shadow. We copy the actual
2115 // argument shadow to the underlying memory.
2116 // Figure out maximal valid memcpy alignment.
2117 const Align ArgAlign = DL.getValueOrABITypeAlignment(
2118 Alignment: FArg.getParamAlign(), Ty: FArg.getParamByValType());
2119 Value *CpShadowPtr, *CpOriginPtr;
2120 std::tie(args&: CpShadowPtr, args&: CpOriginPtr) =
2121 getShadowOriginPtr(Addr: V, IRB&: EntryIRB, ShadowTy: EntryIRB.getInt8Ty(), Alignment: ArgAlign,
2122 /*isStore*/ true);
2123 if (!PropagateShadow || Overflow) {
2124 // ParamTLS overflow.
2125 EntryIRB.CreateMemSet(
2126 Ptr: CpShadowPtr, Val: Constant::getNullValue(Ty: EntryIRB.getInt8Ty()),
2127 Size, Align: ArgAlign);
2128 } else {
2129 Value *Base = getShadowPtrForArgument(IRB&: EntryIRB, ArgOffset);
2130 const Align CopyAlign = std::min(a: ArgAlign, b: kShadowTLSAlignment);
2131 [[maybe_unused]] Value *Cpy = EntryIRB.CreateMemCpy(
2132 Dst: CpShadowPtr, DstAlign: CopyAlign, Src: Base, SrcAlign: CopyAlign, Size);
2133 LLVM_DEBUG(dbgs() << " ByValCpy: " << *Cpy << "\n");
2134
2135 if (MS.TrackOrigins) {
2136 Value *OriginPtr = getOriginPtrForArgument(IRB&: EntryIRB, ArgOffset);
2137 // FIXME: OriginSize should be:
2138 // alignTo(V % kMinOriginAlignment + Size, kMinOriginAlignment)
2139 unsigned OriginSize = alignTo(Size, A: kMinOriginAlignment);
2140 EntryIRB.CreateMemCpy(
2141 Dst: CpOriginPtr,
2142 /* by getShadowOriginPtr */ DstAlign: kMinOriginAlignment, Src: OriginPtr,
2143 /* by origin_tls[ArgOffset] */ SrcAlign: kMinOriginAlignment,
2144 Size: OriginSize);
2145 }
2146 }
2147 }
2148
2149 if (!PropagateShadow || Overflow || FArg.hasByValAttr() ||
2150 (MS.EagerChecks && FArg.hasAttribute(Kind: Attribute::NoUndef))) {
2151 ShadowPtr = getCleanShadow(V);
2152 setOrigin(V: A, Origin: getCleanOrigin());
2153 } else {
2154 // Shadow over TLS
2155 Value *Base = getShadowPtrForArgument(IRB&: EntryIRB, ArgOffset);
2156 ShadowPtr = EntryIRB.CreateAlignedLoad(Ty: getShadowTy(V: &FArg), Ptr: Base,
2157 Align: kShadowTLSAlignment);
2158 if (MS.TrackOrigins) {
2159 Value *OriginPtr = getOriginPtrForArgument(IRB&: EntryIRB, ArgOffset);
2160 setOrigin(V: A, Origin: EntryIRB.CreateLoad(Ty: MS.OriginTy, Ptr: OriginPtr));
2161 }
2162 }
2163 LLVM_DEBUG(dbgs()
2164 << " ARG: " << FArg << " ==> " << *ShadowPtr << "\n");
2165 break;
2166 }
2167
2168 ArgOffset += alignTo(Size, A: kShadowTLSAlignment);
2169 }
2170 assert(ShadowPtr && "Could not find shadow for an argument");
2171 return ShadowPtr;
2172 }
2173
2174 // Check for partially-undefined constant vectors
2175 // TODO: scalable vectors (this is hard because we do not have IRBuilder)
2176 if (isa<FixedVectorType>(Val: V->getType()) && isa<Constant>(Val: V) &&
2177 cast<Constant>(Val: V)->containsUndefOrPoisonElement() && PropagateShadow &&
2178 PoisonUndefVectors) {
2179 unsigned NumElems = cast<FixedVectorType>(Val: V->getType())->getNumElements();
2180 SmallVector<Constant *, 32> ShadowVector(NumElems);
2181 for (unsigned i = 0; i != NumElems; ++i) {
2182 Constant *Elem = cast<Constant>(Val: V)->getAggregateElement(Elt: i);
2183 ShadowVector[i] = isa<UndefValue>(Val: Elem) ? getPoisonedShadow(V: Elem)
2184 : getCleanShadow(V: Elem);
2185 }
2186
2187 Value *ShadowConstant = ConstantVector::get(V: ShadowVector);
2188 LLVM_DEBUG(dbgs() << "Partial undef constant vector: " << *V << " ==> "
2189 << *ShadowConstant << "\n");
2190
2191 return ShadowConstant;
2192 }
2193
2194 // TODO: partially-undefined constant arrays, structures, and nested types
2195
2196 // For everything else the shadow is zero.
2197 return getCleanShadow(V);
2198 }
2199
2200 /// Get the shadow for i-th argument of the instruction I.
2201 Value *getShadow(Instruction *I, int i) {
2202 return getShadow(V: I->getOperand(i));
2203 }
2204
2205 /// Get the origin for a value.
2206 Value *getOrigin(Value *V) {
2207 if (!MS.TrackOrigins)
2208 return nullptr;
2209 if (!PropagateShadow || isa<Constant>(Val: V) || isa<InlineAsm>(Val: V))
2210 return getCleanOrigin();
2211 assert((isa<Instruction>(V) || isa<Argument>(V)) &&
2212 "Unexpected value type in getOrigin()");
2213 if (Instruction *I = dyn_cast<Instruction>(Val: V)) {
2214 if (I->getMetadata(KindID: LLVMContext::MD_nosanitize))
2215 return getCleanOrigin();
2216 }
2217 Value *Origin = OriginMap[V];
2218 assert(Origin && "Missing origin");
2219 return Origin;
2220 }
2221
2222 /// Get the origin for i-th argument of the instruction I.
2223 Value *getOrigin(Instruction *I, int i) {
2224 return getOrigin(V: I->getOperand(i));
2225 }
2226
2227 /// Remember the place where a shadow check should be inserted.
2228 ///
2229 /// This location will be later instrumented with a check that will print a
2230 /// UMR warning in runtime if the shadow value is not 0.
2231 void insertCheckShadow(Value *Shadow, Value *Origin, Instruction *OrigIns) {
2232 assert(Shadow);
2233 if (!InsertChecks)
2234 return;
2235
2236 if (!DebugCounter::shouldExecute(Counter&: DebugInsertCheck)) {
2237 LLVM_DEBUG(dbgs() << "Skipping check of " << *Shadow << " before "
2238 << *OrigIns << "\n");
2239 return;
2240 }
2241
2242 Type *ShadowTy = Shadow->getType();
2243 if (isScalableNonVectorType(Ty: ShadowTy)) {
2244 LLVM_DEBUG(dbgs() << "Skipping check of scalable non-vector " << *Shadow
2245 << " before " << *OrigIns << "\n");
2246 return;
2247 }
2248#ifndef NDEBUG
2249 assert((isa<IntegerType>(ShadowTy) || isa<VectorType>(ShadowTy) ||
2250 isa<StructType>(ShadowTy) || isa<ArrayType>(ShadowTy)) &&
2251 "Can only insert checks for integer, vector, and aggregate shadow "
2252 "types");
2253#endif
2254 InstrumentationList.push_back(
2255 Elt: ShadowOriginAndInsertPoint(Shadow, Origin, OrigIns));
2256 }
2257
2258 /// Get shadow for value, and remember the place where a shadow check should
2259 /// be inserted.
2260 ///
2261 /// This location will be later instrumented with a check that will print a
2262 /// UMR warning in runtime if the value is not fully defined.
2263 void insertCheckShadowOf(Value *Val, Instruction *OrigIns) {
2264 assert(Val);
2265 Value *Shadow, *Origin;
2266 if (ClCheckConstantShadow) {
2267 Shadow = getShadow(V: Val);
2268 if (!Shadow)
2269 return;
2270 Origin = getOrigin(V: Val);
2271 } else {
2272 Shadow = dyn_cast_or_null<Instruction>(Val: getShadow(V: Val));
2273 if (!Shadow)
2274 return;
2275 Origin = dyn_cast_or_null<Instruction>(Val: getOrigin(V: Val));
2276 }
2277 insertCheckShadow(Shadow, Origin, OrigIns);
2278 }
2279
2280 AtomicOrdering addReleaseOrdering(AtomicOrdering a) {
2281 switch (a) {
2282 case AtomicOrdering::NotAtomic:
2283 return AtomicOrdering::NotAtomic;
2284 case AtomicOrdering::Unordered:
2285 case AtomicOrdering::Monotonic:
2286 case AtomicOrdering::Release:
2287 return AtomicOrdering::Release;
2288 case AtomicOrdering::Acquire:
2289 case AtomicOrdering::AcquireRelease:
2290 return AtomicOrdering::AcquireRelease;
2291 case AtomicOrdering::SequentiallyConsistent:
2292 return AtomicOrdering::SequentiallyConsistent;
2293 }
2294 llvm_unreachable("Unknown ordering");
2295 }
2296
2297 Value *makeAddReleaseOrderingTable(IRBuilder<> &IRB) {
2298 constexpr int NumOrderings = (int)AtomicOrderingCABI::seq_cst + 1;
2299 uint32_t OrderingTable[NumOrderings] = {};
2300
2301 OrderingTable[(int)AtomicOrderingCABI::relaxed] =
2302 OrderingTable[(int)AtomicOrderingCABI::release] =
2303 (int)AtomicOrderingCABI::release;
2304 OrderingTable[(int)AtomicOrderingCABI::consume] =
2305 OrderingTable[(int)AtomicOrderingCABI::acquire] =
2306 OrderingTable[(int)AtomicOrderingCABI::acq_rel] =
2307 (int)AtomicOrderingCABI::acq_rel;
2308 OrderingTable[(int)AtomicOrderingCABI::seq_cst] =
2309 (int)AtomicOrderingCABI::seq_cst;
2310
2311 return ConstantDataVector::get(Context&: IRB.getContext(), Elts: OrderingTable);
2312 }
2313
2314 AtomicOrdering addAcquireOrdering(AtomicOrdering a) {
2315 switch (a) {
2316 case AtomicOrdering::NotAtomic:
2317 return AtomicOrdering::NotAtomic;
2318 case AtomicOrdering::Unordered:
2319 case AtomicOrdering::Monotonic:
2320 case AtomicOrdering::Acquire:
2321 return AtomicOrdering::Acquire;
2322 case AtomicOrdering::Release:
2323 case AtomicOrdering::AcquireRelease:
2324 return AtomicOrdering::AcquireRelease;
2325 case AtomicOrdering::SequentiallyConsistent:
2326 return AtomicOrdering::SequentiallyConsistent;
2327 }
2328 llvm_unreachable("Unknown ordering");
2329 }
2330
2331 Value *makeAddAcquireOrderingTable(IRBuilder<> &IRB) {
2332 constexpr int NumOrderings = (int)AtomicOrderingCABI::seq_cst + 1;
2333 uint32_t OrderingTable[NumOrderings] = {};
2334
2335 OrderingTable[(int)AtomicOrderingCABI::relaxed] =
2336 OrderingTable[(int)AtomicOrderingCABI::acquire] =
2337 OrderingTable[(int)AtomicOrderingCABI::consume] =
2338 (int)AtomicOrderingCABI::acquire;
2339 OrderingTable[(int)AtomicOrderingCABI::release] =
2340 OrderingTable[(int)AtomicOrderingCABI::acq_rel] =
2341 (int)AtomicOrderingCABI::acq_rel;
2342 OrderingTable[(int)AtomicOrderingCABI::seq_cst] =
2343 (int)AtomicOrderingCABI::seq_cst;
2344
2345 return ConstantDataVector::get(Context&: IRB.getContext(), Elts: OrderingTable);
2346 }
2347
2348 // ------------------- Visitors.
2349 using InstVisitor<MemorySanitizerVisitor>::visit;
2350 void visit(Instruction &I) {
2351 if (I.getMetadata(KindID: LLVMContext::MD_nosanitize))
2352 return;
2353 // Don't want to visit if we're in the prologue
2354 if (isInPrologue(I))
2355 return;
2356 if (!DebugCounter::shouldExecute(Counter&: DebugInstrumentInstruction)) {
2357 LLVM_DEBUG(dbgs() << "Skipping instruction: " << I << "\n");
2358 // We still need to set the shadow and origin to clean values.
2359 setShadow(V: &I, SV: getCleanShadow(V: &I));
2360 setOrigin(V: &I, Origin: getCleanOrigin());
2361 return;
2362 }
2363
2364 Instructions.push_back(Elt: &I);
2365 }
2366
2367 /// Instrument LoadInst
2368 ///
2369 /// Loads the corresponding shadow and (optionally) origin.
2370 /// Optionally, checks that the load address is fully defined.
2371 void visitLoadInst(LoadInst &I) {
2372 assert(I.getType()->isSized() && "Load type must have size");
2373 assert(!I.getMetadata(LLVMContext::MD_nosanitize));
2374 NextNodeIRBuilder IRB(&I);
2375 Type *ShadowTy = getShadowTy(V: &I);
2376 Value *Addr = I.getPointerOperand();
2377 Value *ShadowPtr = nullptr, *OriginPtr = nullptr;
2378 const Align Alignment = I.getAlign();
2379 if (PropagateShadow) {
2380 std::tie(args&: ShadowPtr, args&: OriginPtr) =
2381 getShadowOriginPtr(Addr, IRB, ShadowTy, Alignment, /*isStore*/ false);
2382 setShadow(V: &I,
2383 SV: IRB.CreateAlignedLoad(Ty: ShadowTy, Ptr: ShadowPtr, Align: Alignment, Name: "_msld"));
2384 } else {
2385 setShadow(V: &I, SV: getCleanShadow(V: &I));
2386 }
2387
2388 if (ClCheckAccessAddress)
2389 insertCheckShadowOf(Val: I.getPointerOperand(), OrigIns: &I);
2390
2391 if (I.isAtomic())
2392 I.setOrdering(addAcquireOrdering(a: I.getOrdering()));
2393
2394 if (MS.TrackOrigins) {
2395 if (PropagateShadow) {
2396 const Align OriginAlignment = std::max(a: kMinOriginAlignment, b: Alignment);
2397 setOrigin(
2398 V: &I, Origin: IRB.CreateAlignedLoad(Ty: MS.OriginTy, Ptr: OriginPtr, Align: OriginAlignment));
2399 } else {
2400 setOrigin(V: &I, Origin: getCleanOrigin());
2401 }
2402 }
2403 }
2404
2405 /// Instrument StoreInst
2406 ///
2407 /// Stores the corresponding shadow and (optionally) origin.
2408 /// Optionally, checks that the store address is fully defined.
2409 void visitStoreInst(StoreInst &I) {
2410 StoreList.push_back(Elt: &I);
2411 if (ClCheckAccessAddress)
2412 insertCheckShadowOf(Val: I.getPointerOperand(), OrigIns: &I);
2413 }
2414
2415 void handleCASOrRMW(Instruction &I) {
2416 assert(isa<AtomicRMWInst>(I) || isa<AtomicCmpXchgInst>(I));
2417
2418 IRBuilder<> IRB(&I);
2419 Value *Addr = I.getOperand(i: 0);
2420 Value *Val = I.getOperand(i: 1);
2421 Value *ShadowPtr = getShadowOriginPtr(Addr, IRB, ShadowTy: getShadowTy(V: Val), Alignment: Align(1),
2422 /*isStore*/ true)
2423 .first;
2424
2425 if (ClCheckAccessAddress)
2426 insertCheckShadowOf(Val: Addr, OrigIns: &I);
2427
2428 // Only test the conditional argument of cmpxchg instruction.
2429 // The other argument can potentially be uninitialized, but we can not
2430 // detect this situation reliably without possible false positives.
2431 if (isa<AtomicCmpXchgInst>(Val: I))
2432 insertCheckShadowOf(Val, OrigIns: &I);
2433
2434 IRB.CreateStore(Val: getCleanShadow(V: Val), Ptr: ShadowPtr);
2435
2436 setShadow(V: &I, SV: getCleanShadow(V: &I));
2437 setOrigin(V: &I, Origin: getCleanOrigin());
2438 }
2439
2440 void visitAtomicRMWInst(AtomicRMWInst &I) {
2441 handleCASOrRMW(I);
2442 I.setOrdering(addReleaseOrdering(a: I.getOrdering()));
2443 }
2444
2445 void visitAtomicCmpXchgInst(AtomicCmpXchgInst &I) {
2446 handleCASOrRMW(I);
2447 I.setSuccessOrdering(addReleaseOrdering(a: I.getSuccessOrdering()));
2448 }
2449
2450 /// Generic handler to compute shadow for == and != comparisons.
2451 ///
2452 /// This function is used by handleEqualityComparison and visitSwitchInst.
2453 ///
2454 /// Sometimes the comparison result is known even if some of the bits of the
2455 /// arguments are not.
2456 Value *propagateEqualityComparison(IRBuilder<> &IRB, Value *A, Value *B,
2457 Value *Sa, Value *Sb) {
2458 assert(getShadowTy(A) == Sa->getType());
2459 assert(getShadowTy(B) == Sb->getType());
2460
2461 // Get rid of pointers and vectors of pointers.
2462 // For ints (and vectors of ints), types of A and Sa match,
2463 // and this is a no-op.
2464 A = IRB.CreatePointerCast(V: A, DestTy: Sa->getType());
2465 B = IRB.CreatePointerCast(V: B, DestTy: Sb->getType());
2466
2467 // A == B <==> (C = A^B) == 0
2468 // A != B <==> (C = A^B) != 0
2469 // Sc = Sa | Sb
2470 Value *C = IRB.CreateXor(LHS: A, RHS: B);
2471 Value *Sc = IRB.CreateOr(LHS: Sa, RHS: Sb);
2472 // Now dealing with i = (C == 0) comparison (or C != 0, does not matter now)
2473 // Result is defined if one of the following is true
2474 // * there is a defined 1 bit in C
2475 // * C is fully defined
2476 // Si = !(C & ~Sc) && Sc
2477 Value *Zero = Constant::getNullValue(Ty: Sc->getType());
2478 Value *MinusOne = Constant::getAllOnesValue(Ty: Sc->getType());
2479 Value *LHS = IRB.CreateICmpNE(LHS: Sc, RHS: Zero);
2480 Value *RHS =
2481 IRB.CreateICmpEQ(LHS: IRB.CreateAnd(LHS: IRB.CreateXor(LHS: Sc, RHS: MinusOne), RHS: C), RHS: Zero);
2482 Value *Si = IRB.CreateAnd(LHS, RHS);
2483 Si->setName("_msprop_icmp");
2484
2485 return Si;
2486 }
2487
2488 // Instrument:
2489 // switch i32 %Val, label %else [ i32 0, label %A
2490 // i32 1, label %B
2491 // i32 2, label %C ]
2492 //
2493 // Typically, the switch input value (%Val) is fully initialized.
2494 //
2495 // Sometimes the compiler may convert (icmp + br) into a switch statement.
2496 // MSan allows icmp eq/ne with partly initialized inputs to still result in a
2497 // fully initialized output, if there exists a bit that is initialized in
2498 // both inputs with a differing value. For compatibility, we support this in
2499 // the switch instrumentation as well. Note that this edge case only applies
2500 // if the switch input value does not match *any* of the cases (matching any
2501 // of the cases requires an exact, fully initialized match).
2502 //
2503 // ShadowCases = 0
2504 // | propagateEqualityComparison(Val, 0)
2505 // | propagateEqualityComparison(Val, 1)
2506 // | propagateEqualityComparison(Val, 2))
2507 void visitSwitchInst(SwitchInst &SI) {
2508 IRBuilder<> IRB(&SI);
2509
2510 Value *Val = SI.getCondition();
2511 Value *ShadowVal = getShadow(V: Val);
2512 // TODO: add fast path - if the condition is fully initialized, we know
2513 // there is no UUM, without needing to consider the case values below.
2514
2515 // Some code (e.g., AMDGPUGenMCCodeEmitter.inc) has tens of thousands of
2516 // cases. This results in an extremely long chained expression for MSan's
2517 // switch instrumentation, which can cause the JumpThreadingPass to have a
2518 // stack overflow or excessive runtime. We limit the number of cases
2519 // considered, with the tradeoff of niche false negatives.
2520 // TODO: figure out a better solution.
2521 int casesToConsider = ClSwitchPrecision;
2522
2523 Value *ShadowCases = nullptr;
2524 for (auto Case : SI.cases()) {
2525 if (casesToConsider <= 0)
2526 break;
2527
2528 Value *Comparator = Case.getCaseValue();
2529 // TODO: some simplification is possible when comparing multiple cases
2530 // simultaneously.
2531 Value *ComparisonShadow = propagateEqualityComparison(
2532 IRB, A: Val, B: Comparator, Sa: ShadowVal, Sb: getShadow(V: Comparator));
2533
2534 if (ShadowCases)
2535 ShadowCases = IRB.CreateOr(LHS: ShadowCases, RHS: ComparisonShadow);
2536 else
2537 ShadowCases = ComparisonShadow;
2538
2539 casesToConsider--;
2540 }
2541
2542 if (ShadowCases)
2543 insertCheckShadow(Shadow: ShadowCases, Origin: getOrigin(V: Val), OrigIns: &SI);
2544 }
2545
2546 // Vector manipulation.
2547 void visitExtractElementInst(ExtractElementInst &I) {
2548 insertCheckShadowOf(Val: I.getOperand(i_nocapture: 1), OrigIns: &I);
2549 IRBuilder<> IRB(&I);
2550 setShadow(V: &I, SV: IRB.CreateExtractElement(Vec: getShadow(I: &I, i: 0), Idx: I.getOperand(i_nocapture: 1),
2551 Name: "_msprop"));
2552 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2553 }
2554
2555 void visitInsertElementInst(InsertElementInst &I) {
2556 insertCheckShadowOf(Val: I.getOperand(i_nocapture: 2), OrigIns: &I);
2557 IRBuilder<> IRB(&I);
2558 auto *Shadow0 = getShadow(I: &I, i: 0);
2559 auto *Shadow1 = getShadow(I: &I, i: 1);
2560 setShadow(V: &I, SV: IRB.CreateInsertElement(Vec: Shadow0, NewElt: Shadow1, Idx: I.getOperand(i_nocapture: 2),
2561 Name: "_msprop"));
2562 setOriginForNaryOp(I);
2563 }
2564
2565 void visitShuffleVectorInst(ShuffleVectorInst &I) {
2566 IRBuilder<> IRB(&I);
2567 auto *Shadow0 = getShadow(I: &I, i: 0);
2568 auto *Shadow1 = getShadow(I: &I, i: 1);
2569 setShadow(V: &I, SV: IRB.CreateShuffleVector(V1: Shadow0, V2: Shadow1, Mask: I.getShuffleMask(),
2570 Name: "_msprop"));
2571 setOriginForNaryOp(I);
2572 }
2573
2574 // Casts.
2575 void visitSExtInst(SExtInst &I) {
2576 IRBuilder<> IRB(&I);
2577 setShadow(V: &I, SV: IRB.CreateSExt(V: getShadow(I: &I, i: 0), DestTy: I.getType(), Name: "_msprop"));
2578 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2579 }
2580
2581 void visitZExtInst(ZExtInst &I) {
2582 IRBuilder<> IRB(&I);
2583 setShadow(V: &I, SV: IRB.CreateZExt(V: getShadow(I: &I, i: 0), DestTy: I.getType(), Name: "_msprop"));
2584 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2585 }
2586
2587 void visitTruncInst(TruncInst &I) {
2588 IRBuilder<> IRB(&I);
2589 setShadow(V: &I, SV: IRB.CreateTrunc(V: getShadow(I: &I, i: 0), DestTy: I.getType(), Name: "_msprop"));
2590 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2591 }
2592
2593 void visitBitCastInst(BitCastInst &I) {
2594 // Special case: if this is the bitcast (there is exactly 1 allowed) between
2595 // a musttail call and a ret, don't instrument. New instructions are not
2596 // allowed after a musttail call.
2597 if (auto *CI = dyn_cast<CallInst>(Val: I.getOperand(i_nocapture: 0)))
2598 if (CI->isMustTailCall())
2599 return;
2600 IRBuilder<> IRB(&I);
2601 setShadow(V: &I, SV: IRB.CreateBitCast(V: getShadow(I: &I, i: 0), DestTy: getShadowTy(V: &I)));
2602 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2603 }
2604
2605 void visitAddrSpaceCastInst(AddrSpaceCastInst &I) {
2606 IRBuilder<> IRB(&I);
2607 setShadow(V: &I, SV: IRB.CreateIntCast(V: getShadow(I: &I, i: 0), DestTy: getShadowTy(V: &I), isSigned: false,
2608 Name: "_msprop_addrspacecast"));
2609 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2610 }
2611
2612 void visitPtrToIntInst(PtrToIntInst &I) {
2613 IRBuilder<> IRB(&I);
2614 setShadow(V: &I, SV: IRB.CreateIntCast(V: getShadow(I: &I, i: 0), DestTy: getShadowTy(V: &I), isSigned: false,
2615 Name: "_msprop_ptrtoint"));
2616 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2617 }
2618
2619 void visitPtrToAddrInst(PtrToAddrInst &I) {
2620 IRBuilder<> IRB(&I);
2621 setShadow(V: &I, SV: IRB.CreateIntCast(V: getShadow(I: &I, i: 0), DestTy: getShadowTy(V: &I), isSigned: false,
2622 Name: "_msprop_ptrtoaddr"));
2623 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2624 }
2625
2626 void visitIntToPtrInst(IntToPtrInst &I) {
2627 IRBuilder<> IRB(&I);
2628 setShadow(V: &I, SV: IRB.CreateIntCast(V: getShadow(I: &I, i: 0), DestTy: getShadowTy(V: &I), isSigned: false,
2629 Name: "_msprop_inttoptr"));
2630 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
2631 }
2632
2633 /// Handle LLVM and NEON vector convert intrinsics.
2634 ///
2635 /// e.g., <4 x i32> @llvm.aarch64.neon.fcvtpu.v4i32.v4f32(<4 x float>)
2636 /// i32 @llvm.aarch64.neon.fcvtms.i32.f64 (double)
2637 /// <2 x i32> @fptoui (<2 x float>)
2638 /// i64 @llvm.fptosi.sat.i64.f64(double)
2639 ///
2640 /// Note that the size of input/output elements can differ e.g.,
2641 /// double @sitofp(i32)
2642 /// but the number of elements must be the same.
2643 ///
2644 /// For conversions to or from fixed-point, there is a trailing argument to
2645 /// indicate the fixed-point precision:
2646 /// - <4 x float> llvm.aarch64.neon.vcvtfxs2fp.v4f32.v4i32(<4 x i32>, i32)
2647 /// - <4 x i32> llvm.aarch64.neon.vcvtfp2fxu.v4i32.v4f32(<4 x float>, i32)
2648 ///
2649 /// For x86 SSE vector convert intrinsics, see
2650 /// handleSSEVectorConvertIntrinsic().
2651 void handleGenericVectorConvertIntrinsic(Instruction &I, bool FixedPoint) {
2652 [[maybe_unused]] unsigned NumArgs = I.getNumOperands();
2653 if (auto *CI = dyn_cast<CallInst>(Val: &I))
2654 NumArgs = CI->arg_size();
2655
2656 if (FixedPoint) {
2657 assert(NumArgs == 2);
2658 Value *Precision = I.getOperand(i: 1);
2659 insertCheckShadowOf(Val: Precision, OrigIns: &I);
2660 } else {
2661 assert(NumArgs == 1);
2662 }
2663
2664 IRBuilder<> IRB(&I);
2665 Value *S0 = getShadow(I: &I, i: 0);
2666
2667 /// For scalars:
2668 /// Since they are converting from floating-point to integer, or between
2669 /// different width floating-point values, the output is:
2670 /// - fully uninitialized if *any* bit of the input is uninitialized
2671 /// - fully ininitialized if all bits of the input are ininitialized
2672 /// We apply the same principle on a per-field basis for vectors.
2673 Value *OutShadow = IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: S0, RHS: getCleanShadow(V: S0)),
2674 DestTy: getShadowTy(V: &I));
2675 setShadow(V: &I, SV: OutShadow);
2676 setOriginForNaryOp(I);
2677 }
2678
2679 void visitFPToSIInst(CastInst &I) {
2680 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/false);
2681 }
2682 void visitFPToUIInst(CastInst &I) {
2683 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/false);
2684 }
2685 void visitSIToFPInst(CastInst &I) {
2686 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/false);
2687 }
2688 void visitUIToFPInst(CastInst &I) {
2689 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/false);
2690 }
2691
2692 void visitFPExtInst(CastInst &I) {
2693 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/false);
2694 }
2695 void visitFPTruncInst(CastInst &I) {
2696 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/false);
2697 }
2698
2699 /// Generic handler to compute shadow for bitwise AND.
2700 ///
2701 /// This is used by 'visitAnd' but also as a primitive for other handlers.
2702 ///
2703 /// This code is precise: it implements the rule that "And" of an initialized
2704 /// zero bit always results in an initialized value:
2705 // 1&1 => 1; 0&1 => 0; p&1 => p;
2706 // 1&0 => 0; 0&0 => 0; p&0 => 0;
2707 // 1&p => p; 0&p => 0; p&p => p;
2708 //
2709 // S = (S1 & S2) | (V1 & S2) | (S1 & V2)
2710 Value *handleBitwiseAnd(IRBuilder<> &IRB, Value *V1, Value *V2, Value *S1,
2711 Value *S2) {
2712 // "The two arguments to the ‘and’ instruction must be integer or vector
2713 // of integer values. Both arguments must have identical types."
2714 //
2715 // We enforce this condition for all callers to handleBitwiseAnd(); callers
2716 // with non-integer types should call CreateAppToShadowCast() themselves.
2717 assert(V1->getType()->isIntOrIntVectorTy());
2718 assert(V1->getType() == V2->getType());
2719
2720 // Conveniently, getShadowTy() of Int/IntVector returns the original type.
2721 assert(V1->getType() == S1->getType());
2722 assert(V2->getType() == S2->getType());
2723
2724 Value *S1S2 = IRB.CreateAnd(LHS: S1, RHS: S2);
2725 Value *V1S2 = IRB.CreateAnd(LHS: V1, RHS: S2);
2726 Value *S1V2 = IRB.CreateAnd(LHS: S1, RHS: V2);
2727
2728 return IRB.CreateOr(Ops: {S1S2, V1S2, S1V2});
2729 }
2730
2731 /// Handler for bitwise AND operator.
2732 void visitAnd(BinaryOperator &I) {
2733 IRBuilder<> IRB(&I);
2734 Value *V1 = I.getOperand(i_nocapture: 0);
2735 Value *V2 = I.getOperand(i_nocapture: 1);
2736 Value *S1 = getShadow(I: &I, i: 0);
2737 Value *S2 = getShadow(I: &I, i: 1);
2738
2739 Value *OutShadow = handleBitwiseAnd(IRB, V1, V2, S1, S2);
2740
2741 setShadow(V: &I, SV: OutShadow);
2742 setOriginForNaryOp(I);
2743 }
2744
2745 void visitOr(BinaryOperator &I) {
2746 IRBuilder<> IRB(&I);
2747 // "Or" of 1 and a poisoned value results in unpoisoned value:
2748 // 1|1 => 1; 0|1 => 1; p|1 => 1;
2749 // 1|0 => 1; 0|0 => 0; p|0 => p;
2750 // 1|p => 1; 0|p => p; p|p => p;
2751 //
2752 // S = (S1 & S2) | (~V1 & S2) | (S1 & ~V2)
2753 //
2754 // If the "disjoint OR" property is violated, the result is poison, and
2755 // hence the entire shadow is uninitialized:
2756 // S = S | SignExt(V1 & V2 != 0)
2757 Value *S1 = getShadow(I: &I, i: 0);
2758 Value *S2 = getShadow(I: &I, i: 1);
2759 Value *V1 = I.getOperand(i_nocapture: 0);
2760 Value *V2 = I.getOperand(i_nocapture: 1);
2761
2762 // "The two arguments to the ‘or’ instruction must be integer or vector
2763 // of integer values. Both arguments must have identical types."
2764 assert(V1->getType()->isIntOrIntVectorTy());
2765 assert(V1->getType() == V2->getType());
2766
2767 // Conveniently, getShadowTy() of Int/IntVector returns the original type.
2768 assert(V1->getType() == S1->getType());
2769 assert(V2->getType() == S2->getType());
2770
2771 Value *NotV1 = IRB.CreateNot(V: V1);
2772 Value *NotV2 = IRB.CreateNot(V: V2);
2773
2774 Value *S1S2 = IRB.CreateAnd(LHS: S1, RHS: S2);
2775 Value *S2NotV1 = IRB.CreateAnd(LHS: NotV1, RHS: S2);
2776 Value *S1NotV2 = IRB.CreateAnd(LHS: S1, RHS: NotV2);
2777
2778 Value *S = IRB.CreateOr(Ops: {S1S2, S2NotV1, S1NotV2});
2779
2780 if (ClPreciseDisjointOr && cast<PossiblyDisjointInst>(Val: &I)->isDisjoint()) {
2781 Value *V1V2 = IRB.CreateAnd(LHS: V1, RHS: V2);
2782 Value *DisjointOrShadow = IRB.CreateSExt(
2783 V: IRB.CreateICmpNE(LHS: V1V2, RHS: getCleanShadow(V: V1V2)), DestTy: V1V2->getType());
2784 S = IRB.CreateOr(LHS: S, RHS: DisjointOrShadow, Name: "_ms_disjoint");
2785 }
2786
2787 setShadow(V: &I, SV: S);
2788 setOriginForNaryOp(I);
2789 }
2790
2791 /// Default propagation of shadow and/or origin.
2792 ///
2793 /// This class implements the general case of shadow propagation, used in all
2794 /// cases where we don't know and/or don't care about what the operation
2795 /// actually does. It converts all input shadow values to a common type
2796 /// (extending or truncating as necessary), and bitwise OR's them.
2797 ///
2798 /// This is much cheaper than inserting checks (i.e. requiring inputs to be
2799 /// fully initialized), and less prone to false positives.
2800 ///
2801 /// This class also implements the general case of origin propagation. For a
2802 /// Nary operation, result origin is set to the origin of an argument that is
2803 /// not entirely initialized. If there is more than one such arguments, the
2804 /// rightmost of them is picked. It does not matter which one is picked if all
2805 /// arguments are initialized.
2806 template <bool CombineShadow> class Combiner {
2807 Value *Shadow = nullptr;
2808 Value *Origin = nullptr;
2809 IRBuilder<> &IRB;
2810 MemorySanitizerVisitor *MSV;
2811
2812 public:
2813 Combiner(MemorySanitizerVisitor *MSV, IRBuilder<> &IRB)
2814 : IRB(IRB), MSV(MSV) {}
2815
2816 /// Add a pair of shadow and origin values to the mix.
2817 Combiner &Add(Value *OpShadow, Value *OpOrigin) {
2818 if (CombineShadow) {
2819 assert(OpShadow);
2820 if (!Shadow)
2821 Shadow = OpShadow;
2822 else {
2823 OpShadow = MSV->CreateShadowCast(IRB, V: OpShadow, dstTy: Shadow->getType());
2824 Shadow = IRB.CreateOr(LHS: Shadow, RHS: OpShadow, Name: "_msprop");
2825 }
2826 }
2827
2828 if (MSV->MS.TrackOrigins) {
2829 assert(OpOrigin);
2830 if (!Origin) {
2831 Origin = OpOrigin;
2832 } else {
2833 Constant *ConstOrigin = dyn_cast<Constant>(Val: OpOrigin);
2834 // No point in adding something that might result in 0 origin value.
2835 if (!ConstOrigin || !ConstOrigin->isNullValue()) {
2836 Value *Cond = MSV->convertToBool(V: OpShadow, IRB);
2837 Origin = IRB.CreateSelect(C: Cond, True: OpOrigin, False: Origin);
2838 }
2839 }
2840 }
2841 return *this;
2842 }
2843
2844 /// Add an application value to the mix.
2845 Combiner &Add(Value *V) {
2846 Value *OpShadow = MSV->getShadow(V);
2847 Value *OpOrigin = MSV->MS.TrackOrigins ? MSV->getOrigin(V) : nullptr;
2848 return Add(OpShadow, OpOrigin);
2849 }
2850
2851 /// Set the current combined values as the given instruction's shadow
2852 /// and origin.
2853 void Done(Instruction *I) {
2854 if (CombineShadow) {
2855 assert(Shadow);
2856 Shadow = MSV->CreateShadowCast(IRB, V: Shadow, dstTy: MSV->getShadowTy(V: I));
2857 MSV->setShadow(V: I, SV: Shadow);
2858 }
2859 if (MSV->MS.TrackOrigins) {
2860 assert(Origin);
2861 MSV->setOrigin(V: I, Origin);
2862 }
2863 }
2864
2865 /// Store the current combined value at the specified origin
2866 /// location.
2867 void DoneAndStoreOrigin(TypeSize TS, Value *OriginPtr) {
2868 if (MSV->MS.TrackOrigins) {
2869 assert(Origin);
2870 MSV->paintOrigin(IRB, Origin, OriginPtr, TS, Alignment: kMinOriginAlignment);
2871 }
2872 }
2873 };
2874
2875 using ShadowAndOriginCombiner = Combiner<true>;
2876 using OriginCombiner = Combiner<false>;
2877
2878 /// Propagate origin for arbitrary operation.
2879 void setOriginForNaryOp(Instruction &I) {
2880 if (!MS.TrackOrigins)
2881 return;
2882 IRBuilder<> IRB(&I);
2883 OriginCombiner OC(this, IRB);
2884 for (Use &Op : I.operands())
2885 OC.Add(V: Op.get());
2886 OC.Done(I: &I);
2887 }
2888
2889 size_t VectorOrPrimitiveTypeSizeInBits(Type *Ty) {
2890 assert(!(Ty->isVectorTy() && Ty->getScalarType()->isPointerTy()) &&
2891 "Vector of pointers is not a valid shadow type");
2892 return Ty->isVectorTy() ? cast<FixedVectorType>(Val: Ty)->getNumElements() *
2893 Ty->getScalarSizeInBits()
2894 : Ty->getPrimitiveSizeInBits();
2895 }
2896
2897 /// Cast between two shadow types, extending or truncating as
2898 /// necessary.
2899 Value *CreateShadowCast(IRBuilder<> &IRB, Value *V, Type *dstTy,
2900 bool Signed = false) {
2901 Type *srcTy = V->getType();
2902 if (srcTy == dstTy)
2903 return V;
2904 size_t srcSizeInBits = VectorOrPrimitiveTypeSizeInBits(Ty: srcTy);
2905 size_t dstSizeInBits = VectorOrPrimitiveTypeSizeInBits(Ty: dstTy);
2906 if (srcSizeInBits > 1 && dstSizeInBits == 1)
2907 return IRB.CreateICmpNE(LHS: V, RHS: getCleanShadow(V));
2908
2909 if (dstTy->isIntegerTy() && srcTy->isIntegerTy())
2910 return IRB.CreateIntCast(V, DestTy: dstTy, isSigned: Signed);
2911 if (dstTy->isVectorTy() && srcTy->isVectorTy() &&
2912 cast<VectorType>(Val: dstTy)->getElementCount() ==
2913 cast<VectorType>(Val: srcTy)->getElementCount())
2914 return IRB.CreateIntCast(V, DestTy: dstTy, isSigned: Signed);
2915 Value *V1 = IRB.CreateBitCast(V, DestTy: Type::getIntNTy(C&: *MS.C, N: srcSizeInBits));
2916 Value *V2 =
2917 IRB.CreateIntCast(V: V1, DestTy: Type::getIntNTy(C&: *MS.C, N: dstSizeInBits), isSigned: Signed);
2918 return IRB.CreateBitCast(V: V2, DestTy: dstTy);
2919 // TODO: handle struct types.
2920 }
2921
2922 /// Cast an application value to the type of its own shadow.
2923 Value *CreateAppToShadowCast(IRBuilder<> &IRB, Value *V) {
2924 Type *ShadowTy = getShadowTy(V);
2925 if (V->getType() == ShadowTy)
2926 return V;
2927 if (V->getType()->isPtrOrPtrVectorTy())
2928 return IRB.CreatePtrToInt(V, DestTy: ShadowTy);
2929 else
2930 return IRB.CreateBitCast(V, DestTy: ShadowTy);
2931 }
2932
2933 /// Propagate shadow for arbitrary operation.
2934 void handleShadowOr(Instruction &I) {
2935 IRBuilder<> IRB(&I);
2936 ShadowAndOriginCombiner SC(this, IRB);
2937 for (Use &Op : I.operands())
2938 SC.Add(V: Op.get());
2939 SC.Done(I: &I);
2940 }
2941
2942 // Perform a bitwise OR on the horizontal pairs (or other specified grouping)
2943 // of elements.
2944 //
2945 // For example, suppose we have:
2946 // VectorA: <a0, a1, a2, a3, a4, a5>
2947 // VectorB: <b0, b1, b2, b3, b4, b5>
2948 // ReductionFactor: 3
2949 // Shards: 1
2950 // The output would be:
2951 // <a0|a1|a2, a3|a4|a5, b0|b1|b2, b3|b4|b5>
2952 //
2953 // If we have:
2954 // VectorA: <a0, a1, a2, a3, a4, a5, a6, a7>
2955 // VectorB: <b0, b1, b2, b3, b4, b5, b6, b7>
2956 // ReductionFactor: 2
2957 // Shards: 2
2958 // then a and be each have 2 "shards", resulting in the output being
2959 // interleaved:
2960 // <a0|a1, a2|a3, b0|b1, b2|b3, a4|a5, a6|a7, b4|b5, b6|b7>
2961 //
2962 // This is convenient for instrumenting horizontal add/sub.
2963 // For bitwise OR on "vertical" pairs, see maybeHandleSimpleNomemIntrinsic().
2964 Value *horizontalReduce(IntrinsicInst &I, unsigned ReductionFactor,
2965 unsigned Shards, Value *VectorA, Value *VectorB) {
2966 assert(isa<FixedVectorType>(VectorA->getType()));
2967 unsigned NumElems =
2968 cast<FixedVectorType>(Val: VectorA->getType())->getNumElements();
2969
2970 [[maybe_unused]] unsigned TotalNumElems = NumElems;
2971 if (VectorB) {
2972 assert(VectorA->getType() == VectorB->getType());
2973 TotalNumElems *= 2;
2974 }
2975
2976 assert(NumElems % (ReductionFactor * Shards) == 0);
2977
2978 Value *Or = nullptr;
2979
2980 IRBuilder<> IRB(&I);
2981 for (unsigned i = 0; i < ReductionFactor; i++) {
2982 SmallVector<int, 16> Mask;
2983
2984 for (unsigned j = 0; j < Shards; j++) {
2985 unsigned Offset = NumElems / Shards * j;
2986
2987 for (unsigned X = 0; X < NumElems / Shards; X += ReductionFactor)
2988 Mask.push_back(Elt: Offset + X + i);
2989
2990 if (VectorB) {
2991 for (unsigned X = 0; X < NumElems / Shards; X += ReductionFactor)
2992 Mask.push_back(Elt: NumElems + Offset + X + i);
2993 }
2994 }
2995
2996 Value *Masked;
2997 if (VectorB)
2998 Masked = IRB.CreateShuffleVector(V1: VectorA, V2: VectorB, Mask);
2999 else
3000 Masked = IRB.CreateShuffleVector(V: VectorA, Mask);
3001
3002 if (Or)
3003 Or = IRB.CreateOr(LHS: Or, RHS: Masked);
3004 else
3005 Or = Masked;
3006 }
3007
3008 return Or;
3009 }
3010
3011 /// Propagate shadow for 1- or 2-vector intrinsics that combine adjacent
3012 /// fields.
3013 ///
3014 /// e.g., <2 x i32> @llvm.aarch64.neon.saddlp.v2i32.v4i16(<4 x i16>)
3015 /// <16 x i8> @llvm.aarch64.neon.addp.v16i8(<16 x i8>, <16 x i8>)
3016 void handlePairwiseShadowOrIntrinsic(IntrinsicInst &I, unsigned Shards) {
3017 assert(I.arg_size() == 1 || I.arg_size() == 2);
3018
3019 assert(I.getType()->isVectorTy());
3020 assert(I.getArgOperand(0)->getType()->isVectorTy());
3021
3022 [[maybe_unused]] FixedVectorType *ParamType =
3023 cast<FixedVectorType>(Val: I.getArgOperand(i: 0)->getType());
3024 assert((I.arg_size() != 2) ||
3025 (ParamType == cast<FixedVectorType>(I.getArgOperand(1)->getType())));
3026 [[maybe_unused]] FixedVectorType *ReturnType =
3027 cast<FixedVectorType>(Val: I.getType());
3028 assert(ParamType->getNumElements() * I.arg_size() ==
3029 2 * ReturnType->getNumElements());
3030
3031 IRBuilder<> IRB(&I);
3032
3033 // Horizontal OR of shadow
3034 Value *FirstArgShadow = getShadow(I: &I, i: 0);
3035 Value *SecondArgShadow = nullptr;
3036 if (I.arg_size() == 2)
3037 SecondArgShadow = getShadow(I: &I, i: 1);
3038
3039 Value *OrShadow = horizontalReduce(I, /*ReductionFactor=*/2, Shards,
3040 VectorA: FirstArgShadow, VectorB: SecondArgShadow);
3041
3042 OrShadow = CreateShadowCast(IRB, V: OrShadow, dstTy: getShadowTy(V: &I));
3043
3044 setShadow(V: &I, SV: OrShadow);
3045 setOriginForNaryOp(I);
3046 }
3047
3048 /// Propagate shadow for 1- or 2-vector intrinsics that combine adjacent
3049 /// fields, with the parameters reinterpreted to have elements of a specified
3050 /// width. For example:
3051 /// @llvm.x86.ssse3.phadd.w(<1 x i64> [[VAR1]], <1 x i64> [[VAR2]])
3052 /// conceptually operates on
3053 /// (<4 x i16> [[VAR1]], <4 x i16> [[VAR2]])
3054 /// and can be handled with ReinterpretElemWidth == 16.
3055 void handlePairwiseShadowOrIntrinsic(IntrinsicInst &I, unsigned Shards,
3056 int ReinterpretElemWidth) {
3057 assert(I.arg_size() == 1 || I.arg_size() == 2);
3058
3059 assert(I.getType()->isVectorTy());
3060 assert(I.getArgOperand(0)->getType()->isVectorTy());
3061
3062 FixedVectorType *ParamType =
3063 cast<FixedVectorType>(Val: I.getArgOperand(i: 0)->getType());
3064 assert((I.arg_size() != 2) ||
3065 (ParamType == cast<FixedVectorType>(I.getArgOperand(1)->getType())));
3066
3067 [[maybe_unused]] FixedVectorType *ReturnType =
3068 cast<FixedVectorType>(Val: I.getType());
3069 assert(ParamType->getNumElements() * I.arg_size() ==
3070 2 * ReturnType->getNumElements());
3071
3072 IRBuilder<> IRB(&I);
3073
3074 FixedVectorType *ReinterpretShadowTy = nullptr;
3075 assert(isAligned(Align(ReinterpretElemWidth),
3076 ParamType->getPrimitiveSizeInBits()));
3077 ReinterpretShadowTy = FixedVectorType::get(
3078 ElementType: IRB.getIntNTy(N: ReinterpretElemWidth),
3079 NumElts: ParamType->getPrimitiveSizeInBits() / ReinterpretElemWidth);
3080
3081 // Horizontal OR of shadow
3082 Value *FirstArgShadow = getShadow(I: &I, i: 0);
3083 FirstArgShadow = IRB.CreateBitCast(V: FirstArgShadow, DestTy: ReinterpretShadowTy);
3084
3085 // If we had two parameters each with an odd number of elements, the total
3086 // number of elements is even, but we have never seen this in extant
3087 // instruction sets, so we enforce that each parameter must have an even
3088 // number of elements.
3089 assert(isAligned(
3090 Align(2),
3091 cast<FixedVectorType>(FirstArgShadow->getType())->getNumElements()));
3092
3093 Value *SecondArgShadow = nullptr;
3094 if (I.arg_size() == 2) {
3095 SecondArgShadow = getShadow(I: &I, i: 1);
3096 SecondArgShadow = IRB.CreateBitCast(V: SecondArgShadow, DestTy: ReinterpretShadowTy);
3097 }
3098
3099 Value *OrShadow = horizontalReduce(I, /*ReductionFactor=*/2, Shards,
3100 VectorA: FirstArgShadow, VectorB: SecondArgShadow);
3101
3102 OrShadow = CreateShadowCast(IRB, V: OrShadow, dstTy: getShadowTy(V: &I));
3103
3104 setShadow(V: &I, SV: OrShadow);
3105 setOriginForNaryOp(I);
3106 }
3107
3108 void visitFNeg(UnaryOperator &I) { handleShadowOr(I); }
3109
3110 // Handle multiplication by constant.
3111 //
3112 // Handle a special case of multiplication by constant that may have one or
3113 // more zeros in the lower bits. This makes corresponding number of lower bits
3114 // of the result zero as well. We model it by shifting the other operand
3115 // shadow left by the required number of bits. Effectively, we transform
3116 // (X * (A * 2**B)) to ((X << B) * A) and instrument (X << B) as (Sx << B).
3117 // We use multiplication by 2**N instead of shift to cover the case of
3118 // multiplication by 0, which may occur in some elements of a vector operand.
3119 void handleMulByConstant(BinaryOperator &I, Constant *ConstArg,
3120 Value *OtherArg) {
3121 Constant *ShadowMul;
3122 Type *Ty = ConstArg->getType();
3123 if (auto *VTy = dyn_cast<VectorType>(Val: Ty)) {
3124 unsigned NumElements = cast<FixedVectorType>(Val: VTy)->getNumElements();
3125 Type *EltTy = VTy->getElementType();
3126 SmallVector<Constant *, 16> Elements;
3127 for (unsigned Idx = 0; Idx < NumElements; ++Idx) {
3128 if (ConstantInt *Elt =
3129 dyn_cast<ConstantInt>(Val: ConstArg->getAggregateElement(Elt: Idx))) {
3130 const APInt &V = Elt->getValue();
3131 APInt V2 = APInt(V.getBitWidth(), 1) << V.countr_zero();
3132 Elements.push_back(Elt: ConstantInt::get(Ty: EltTy, V: V2));
3133 } else {
3134 Elements.push_back(Elt: ConstantInt::get(Ty: EltTy, V: 1));
3135 }
3136 }
3137 ShadowMul = ConstantVector::get(V: Elements);
3138 } else {
3139 if (ConstantInt *Elt = dyn_cast<ConstantInt>(Val: ConstArg)) {
3140 const APInt &V = Elt->getValue();
3141 APInt V2 = APInt(V.getBitWidth(), 1) << V.countr_zero();
3142 ShadowMul = ConstantInt::get(Ty, V: V2);
3143 } else {
3144 ShadowMul = ConstantInt::get(Ty, V: 1);
3145 }
3146 }
3147
3148 IRBuilder<> IRB(&I);
3149 setShadow(V: &I,
3150 SV: IRB.CreateMul(LHS: getShadow(V: OtherArg), RHS: ShadowMul, Name: "msprop_mul_cst"));
3151 setOrigin(V: &I, Origin: getOrigin(V: OtherArg));
3152 }
3153
3154 void visitMul(BinaryOperator &I) {
3155 Constant *constOp0 = dyn_cast<Constant>(Val: I.getOperand(i_nocapture: 0));
3156 Constant *constOp1 = dyn_cast<Constant>(Val: I.getOperand(i_nocapture: 1));
3157 if (constOp0 && !constOp1)
3158 handleMulByConstant(I, ConstArg: constOp0, OtherArg: I.getOperand(i_nocapture: 1));
3159 else if (constOp1 && !constOp0)
3160 handleMulByConstant(I, ConstArg: constOp1, OtherArg: I.getOperand(i_nocapture: 0));
3161 else
3162 handleShadowOr(I);
3163 }
3164
3165 void visitFAdd(BinaryOperator &I) { handleShadowOr(I); }
3166 void visitFSub(BinaryOperator &I) { handleShadowOr(I); }
3167 void visitFMul(BinaryOperator &I) { handleShadowOr(I); }
3168 void visitAdd(BinaryOperator &I) { handleShadowOr(I); }
3169 void visitSub(BinaryOperator &I) { handleShadowOr(I); }
3170 void visitXor(BinaryOperator &I) { handleShadowOr(I); }
3171
3172 void handleIntegerDiv(Instruction &I) {
3173 IRBuilder<> IRB(&I);
3174 // Strict on the second argument.
3175 insertCheckShadowOf(Val: I.getOperand(i: 1), OrigIns: &I);
3176 setShadow(V: &I, SV: getShadow(I: &I, i: 0));
3177 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
3178 }
3179
3180 void visitUDiv(BinaryOperator &I) { handleIntegerDiv(I); }
3181 void visitSDiv(BinaryOperator &I) { handleIntegerDiv(I); }
3182 void visitURem(BinaryOperator &I) { handleIntegerDiv(I); }
3183 void visitSRem(BinaryOperator &I) { handleIntegerDiv(I); }
3184
3185 // Floating point division is side-effect free. We can not require that the
3186 // divisor is fully initialized and must propagate shadow. See PR37523.
3187 void visitFDiv(BinaryOperator &I) { handleShadowOr(I); }
3188 void visitFRem(BinaryOperator &I) { handleShadowOr(I); }
3189
3190 /// Instrument == and != comparisons.
3191 ///
3192 /// Sometimes the comparison result is known even if some of the bits of the
3193 /// arguments are not.
3194 void handleEqualityComparison(ICmpInst &I) {
3195 IRBuilder<> IRB(&I);
3196 Value *A = I.getOperand(i_nocapture: 0);
3197 Value *B = I.getOperand(i_nocapture: 1);
3198 Value *Sa = getShadow(V: A);
3199 Value *Sb = getShadow(V: B);
3200
3201 Value *Si = propagateEqualityComparison(IRB, A, B, Sa, Sb);
3202
3203 setShadow(V: &I, SV: Si);
3204 setOriginForNaryOp(I);
3205 }
3206
3207 /// Instrument relational comparisons.
3208 ///
3209 /// This function does exact shadow propagation for all relational
3210 /// comparisons of integers, pointers and vectors of those.
3211 /// FIXME: output seems suboptimal when one of the operands is a constant
3212 void handleRelationalComparisonExact(ICmpInst &I) {
3213 IRBuilder<> IRB(&I);
3214 Value *A = I.getOperand(i_nocapture: 0);
3215 Value *B = I.getOperand(i_nocapture: 1);
3216 Value *Sa = getShadow(V: A);
3217 Value *Sb = getShadow(V: B);
3218
3219 // Get rid of pointers and vectors of pointers.
3220 // For ints (and vectors of ints), types of A and Sa match,
3221 // and this is a no-op.
3222 A = IRB.CreatePointerCast(V: A, DestTy: Sa->getType());
3223 B = IRB.CreatePointerCast(V: B, DestTy: Sb->getType());
3224
3225 // Let [a0, a1] be the interval of possible values of A, taking into account
3226 // its undefined bits. Let [b0, b1] be the interval of possible values of B.
3227 // Then (A cmp B) is defined iff (a0 cmp b1) == (a1 cmp b0).
3228 bool IsSigned = I.isSigned();
3229
3230 auto GetMinMaxUnsigned = [&](Value *V, Value *S) {
3231 if (IsSigned) {
3232 // Sign-flip to map from signed range to unsigned range. Relation A vs B
3233 // should be preserved, if checked with `getUnsignedPredicate()`.
3234 // Relationship between Amin, Amax, Bmin, Bmax also will not be
3235 // affected, as they are created by effectively adding/substructing from
3236 // A (or B) a value, derived from shadow, with no overflow, either
3237 // before or after sign flip.
3238 APInt MinVal =
3239 APInt::getSignedMinValue(numBits: V->getType()->getScalarSizeInBits());
3240 V = IRB.CreateXor(LHS: V, RHS: ConstantInt::get(Ty: V->getType(), V: MinVal));
3241 }
3242 // Minimize undefined bits.
3243 Value *Min = IRB.CreateAnd(LHS: V, RHS: IRB.CreateNot(V: S));
3244 Value *Max = IRB.CreateOr(LHS: V, RHS: S);
3245 return std::make_pair(x&: Min, y&: Max);
3246 };
3247
3248 auto [Amin, Amax] = GetMinMaxUnsigned(A, Sa);
3249 auto [Bmin, Bmax] = GetMinMaxUnsigned(B, Sb);
3250 Value *S1 = IRB.CreateICmp(P: I.getUnsignedPredicate(), LHS: Amin, RHS: Bmax);
3251 Value *S2 = IRB.CreateICmp(P: I.getUnsignedPredicate(), LHS: Amax, RHS: Bmin);
3252
3253 Value *Si = IRB.CreateXor(LHS: S1, RHS: S2);
3254 setShadow(V: &I, SV: Si);
3255 setOriginForNaryOp(I);
3256 }
3257
3258 /// Instrument signed relational comparisons.
3259 ///
3260 /// Handle sign bit tests: x<0, x>=0, x<=-1, x>-1 by propagating the highest
3261 /// bit of the shadow. Everything else is delegated to handleShadowOr().
3262 void handleSignedRelationalComparison(ICmpInst &I) {
3263 Constant *constOp;
3264 Value *op = nullptr;
3265 CmpInst::Predicate pre;
3266 if ((constOp = dyn_cast<Constant>(Val: I.getOperand(i_nocapture: 1)))) {
3267 op = I.getOperand(i_nocapture: 0);
3268 pre = I.getPredicate();
3269 } else if ((constOp = dyn_cast<Constant>(Val: I.getOperand(i_nocapture: 0)))) {
3270 op = I.getOperand(i_nocapture: 1);
3271 pre = I.getSwappedPredicate();
3272 } else {
3273 handleShadowOr(I);
3274 return;
3275 }
3276
3277 if ((constOp->isNullValue() &&
3278 (pre == CmpInst::ICMP_SLT || pre == CmpInst::ICMP_SGE)) ||
3279 (constOp->isAllOnesValue() &&
3280 (pre == CmpInst::ICMP_SGT || pre == CmpInst::ICMP_SLE))) {
3281 IRBuilder<> IRB(&I);
3282 Value *Shadow = IRB.CreateICmpSLT(LHS: getShadow(V: op), RHS: getCleanShadow(V: op),
3283 Name: "_msprop_icmp_s");
3284 setShadow(V: &I, SV: Shadow);
3285 setOrigin(V: &I, Origin: getOrigin(V: op));
3286 } else {
3287 handleShadowOr(I);
3288 }
3289 }
3290
3291 void visitICmpInst(ICmpInst &I) {
3292 if (!ClHandleICmp) {
3293 handleShadowOr(I);
3294 return;
3295 }
3296 if (I.isEquality()) {
3297 handleEqualityComparison(I);
3298 return;
3299 }
3300
3301 assert(I.isRelational());
3302 if (ClHandleICmpExact) {
3303 handleRelationalComparisonExact(I);
3304 return;
3305 }
3306 if (I.isSigned()) {
3307 handleSignedRelationalComparison(I);
3308 return;
3309 }
3310
3311 assert(I.isUnsigned());
3312 if ((isa<Constant>(Val: I.getOperand(i_nocapture: 0)) || isa<Constant>(Val: I.getOperand(i_nocapture: 1)))) {
3313 handleRelationalComparisonExact(I);
3314 return;
3315 }
3316
3317 handleShadowOr(I);
3318 }
3319
3320 void visitFCmpInst(FCmpInst &I) { handleShadowOr(I); }
3321
3322 void handleShift(BinaryOperator &I) {
3323 IRBuilder<> IRB(&I);
3324 // If any of the S2 bits are poisoned, the whole thing is poisoned.
3325 // Otherwise perform the same shift on S1.
3326 Value *S1 = getShadow(I: &I, i: 0);
3327 Value *S2 = getShadow(I: &I, i: 1);
3328 Value *S2Conv =
3329 IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: S2, RHS: getCleanShadow(V: S2)), DestTy: S2->getType());
3330 Value *V2 = I.getOperand(i_nocapture: 1);
3331 Value *Shift = IRB.CreateBinOp(Opc: I.getOpcode(), LHS: S1, RHS: V2);
3332 setShadow(V: &I, SV: IRB.CreateOr(LHS: Shift, RHS: S2Conv));
3333 setOriginForNaryOp(I);
3334 }
3335
3336 void visitShl(BinaryOperator &I) { handleShift(I); }
3337 void visitAShr(BinaryOperator &I) { handleShift(I); }
3338 void visitLShr(BinaryOperator &I) { handleShift(I); }
3339
3340 void handleFunnelShift(IntrinsicInst &I) {
3341 IRBuilder<> IRB(&I);
3342 // If any of the S2 bits are poisoned, the whole thing is poisoned.
3343 // Otherwise perform the same shift on S0 and S1.
3344 Value *S0 = getShadow(I: &I, i: 0);
3345 Value *S1 = getShadow(I: &I, i: 1);
3346 Value *S2 = getShadow(I: &I, i: 2);
3347 Value *S2Conv =
3348 IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: S2, RHS: getCleanShadow(V: S2)), DestTy: S2->getType());
3349 Value *V2 = I.getOperand(i_nocapture: 2);
3350 Value *Shift = IRB.CreateIntrinsic(ID: I.getIntrinsicID(), OverloadTypes: S2Conv->getType(),
3351 Args: {S0, S1, V2});
3352 setShadow(V: &I, SV: IRB.CreateOr(LHS: Shift, RHS: S2Conv));
3353 setOriginForNaryOp(I);
3354 }
3355
3356 // Instrument bit manipulation intrinsics.
3357 // All of these intrinsics are Z = I(SRC, MASK)
3358 // where the types of all operands and the result match.
3359 // The following instrumentation happens to work for all of them:
3360 // Sz = I(Ssrc, MASK) | (sext (Smask != 0))
3361 void handleGenericBitManipulation(IntrinsicInst &I) {
3362 IRBuilder<> IRB(&I);
3363 Type *ShadowTy = getShadowTy(V: &I);
3364
3365 // If any bit of the mask operand is poisoned, then the whole thing is.
3366 Value *SMask = getShadow(I: &I, i: 1);
3367 SMask = IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: SMask, RHS: getCleanShadow(OrigTy: ShadowTy)),
3368 DestTy: ShadowTy);
3369 // Apply the same intrinsic to the shadow of the first operand.
3370 Value *S;
3371 if (Function *Func = I.getCalledFunction())
3372 S = IRB.CreateCall(Callee: Func, Args: {getShadow(I: &I, i: 0), I.getOperand(i_nocapture: 1)});
3373 else
3374 S = IRB.CreateIntrinsic(ID: I.getIntrinsicID(), OverloadTypes: ShadowTy,
3375 Args: {getShadow(I: &I, i: 0), I.getOperand(i_nocapture: 1)});
3376
3377 setShadow(V: &I, SV: IRB.CreateOr(LHS: SMask, RHS: S));
3378 setOriginForNaryOp(I);
3379 }
3380
3381 /// Instrument llvm.memmove
3382 ///
3383 /// At this point we don't know if llvm.memmove will be inlined or not.
3384 /// If we don't instrument it and it gets inlined,
3385 /// our interceptor will not kick in and we will lose the memmove.
3386 /// If we instrument the call here, but it does not get inlined,
3387 /// we will memmove the shadow twice: which is bad in case
3388 /// of overlapping regions. So, we simply lower the intrinsic to a call.
3389 ///
3390 /// Similar situation exists for memcpy and memset.
3391 void visitMemMoveInst(MemMoveInst &I) {
3392 getShadow(V: I.getArgOperand(i: 1)); // Ensure shadow initialized
3393 IRBuilder<> IRB(&I);
3394 IRB.CreateCall(Callee: MS.MemmoveFn,
3395 Args: {I.getArgOperand(i: 0), I.getArgOperand(i: 1),
3396 IRB.CreateIntCast(V: I.getArgOperand(i: 2), DestTy: MS.IntptrTy, isSigned: false)});
3397 I.eraseFromParent();
3398 }
3399
3400 /// Instrument memcpy
3401 ///
3402 /// Similar to memmove: avoid copying shadow twice. This is somewhat
3403 /// unfortunate as it may slowdown small constant memcpys.
3404 /// FIXME: consider doing manual inline for small constant sizes and proper
3405 /// alignment.
3406 ///
3407 /// Note: This also handles memcpy.inline, which promises no calls to external
3408 /// functions as an optimization. However, with instrumentation enabled this
3409 /// is difficult to promise; additionally, we know that the MSan runtime
3410 /// exists and provides __msan_memcpy(). Therefore, we assume that with
3411 /// instrumentation it's safe to turn memcpy.inline into a call to
3412 /// __msan_memcpy(). Should this be wrong, such as when implementing memcpy()
3413 /// itself, instrumentation should be disabled with the no_sanitize attribute.
3414 void visitMemCpyInst(MemCpyInst &I) {
3415 getShadow(V: I.getArgOperand(i: 1)); // Ensure shadow initialized
3416 IRBuilder<> IRB(&I);
3417 IRB.CreateCall(Callee: MS.MemcpyFn,
3418 Args: {I.getArgOperand(i: 0), I.getArgOperand(i: 1),
3419 IRB.CreateIntCast(V: I.getArgOperand(i: 2), DestTy: MS.IntptrTy, isSigned: false)});
3420 I.eraseFromParent();
3421 }
3422
3423 // Same as memcpy.
3424 void visitMemSetInst(MemSetInst &I) {
3425 IRBuilder<> IRB(&I);
3426 IRB.CreateCall(
3427 Callee: MS.MemsetFn,
3428 Args: {I.getArgOperand(i: 0),
3429 IRB.CreateIntCast(V: I.getArgOperand(i: 1), DestTy: IRB.getInt32Ty(), isSigned: false),
3430 IRB.CreateIntCast(V: I.getArgOperand(i: 2), DestTy: MS.IntptrTy, isSigned: false)});
3431 I.eraseFromParent();
3432 }
3433
3434 void visitVAStartInst(VAStartInst &I) { VAHelper->visitVAStartInst(I); }
3435
3436 void visitVACopyInst(VACopyInst &I) { VAHelper->visitVACopyInst(I); }
3437
3438 /// Handle vector store-like intrinsics.
3439 ///
3440 /// Instrument intrinsics that look like a simple SIMD store: writes memory,
3441 /// has 1 pointer argument and 1 vector argument, returns void.
3442 bool handleVectorStoreIntrinsic(IntrinsicInst &I) {
3443 assert(I.arg_size() == 2);
3444
3445 IRBuilder<> IRB(&I);
3446 Value *Addr = I.getArgOperand(i: 0);
3447 Value *Shadow = getShadow(I: &I, i: 1);
3448 Value *ShadowPtr, *OriginPtr;
3449
3450 // We don't know the pointer alignment (could be unaligned SSE store!).
3451 // Have to assume to worst case.
3452 std::tie(args&: ShadowPtr, args&: OriginPtr) = getShadowOriginPtr(
3453 Addr, IRB, ShadowTy: Shadow->getType(), Alignment: Align(1), /*isStore*/ true);
3454 IRB.CreateAlignedStore(Val: Shadow, Ptr: ShadowPtr, Align: Align(1));
3455
3456 if (ClCheckAccessAddress)
3457 insertCheckShadowOf(Val: Addr, OrigIns: &I);
3458
3459 // FIXME: factor out common code from materializeStores
3460 if (MS.TrackOrigins)
3461 IRB.CreateStore(Val: getOrigin(I: &I, i: 1), Ptr: OriginPtr);
3462 return true;
3463 }
3464
3465 /// Handle vector load-like intrinsics.
3466 ///
3467 /// Instrument intrinsics that look like a simple SIMD load: reads memory,
3468 /// has 1 pointer argument, returns a vector.
3469 bool handleVectorLoadIntrinsic(IntrinsicInst &I) {
3470 assert(I.arg_size() == 1);
3471
3472 IRBuilder<> IRB(&I);
3473 Value *Addr = I.getArgOperand(i: 0);
3474
3475 Type *ShadowTy = getShadowTy(V: &I);
3476 Value *ShadowPtr = nullptr, *OriginPtr = nullptr;
3477 if (PropagateShadow) {
3478 // We don't know the pointer alignment (could be unaligned SSE load!).
3479 // Have to assume to worst case.
3480 const Align Alignment = Align(1);
3481 std::tie(args&: ShadowPtr, args&: OriginPtr) =
3482 getShadowOriginPtr(Addr, IRB, ShadowTy, Alignment, /*isStore*/ false);
3483 setShadow(V: &I,
3484 SV: IRB.CreateAlignedLoad(Ty: ShadowTy, Ptr: ShadowPtr, Align: Alignment, Name: "_msld"));
3485 } else {
3486 setShadow(V: &I, SV: getCleanShadow(V: &I));
3487 }
3488
3489 if (ClCheckAccessAddress)
3490 insertCheckShadowOf(Val: Addr, OrigIns: &I);
3491
3492 if (MS.TrackOrigins) {
3493 if (PropagateShadow)
3494 setOrigin(V: &I, Origin: IRB.CreateLoad(Ty: MS.OriginTy, Ptr: OriginPtr));
3495 else
3496 setOrigin(V: &I, Origin: getCleanOrigin());
3497 }
3498 return true;
3499 }
3500
3501 /// Handle (SIMD arithmetic)-like intrinsics.
3502 ///
3503 /// Instrument intrinsics with any number of arguments of the same type [*],
3504 /// equal to the return type, plus a specified number of trailing flags of
3505 /// any type.
3506 ///
3507 /// [*] The type should be simple (no aggregates or pointers; vectors are
3508 /// fine).
3509 ///
3510 /// Caller guarantees that this intrinsic does not access memory.
3511 ///
3512 /// TODO: "horizontal"/"pairwise" intrinsics are often incorrectly matched by
3513 /// by this handler. See horizontalReduce().
3514 ///
3515 /// TODO: permutation intrinsics are also often incorrectly matched.
3516 [[maybe_unused]] bool
3517 maybeHandleSimpleNomemIntrinsic(IntrinsicInst &I,
3518 unsigned int trailingFlags) {
3519 Type *RetTy = I.getType();
3520 if (!(RetTy->isIntOrIntVectorTy() || RetTy->isFPOrFPVectorTy()))
3521 return false;
3522
3523 unsigned NumArgOperands = I.arg_size();
3524 assert(NumArgOperands >= trailingFlags);
3525 for (unsigned i = 0; i < NumArgOperands - trailingFlags; ++i) {
3526 Type *Ty = I.getArgOperand(i)->getType();
3527 if (Ty != RetTy)
3528 return false;
3529 }
3530
3531 IRBuilder<> IRB(&I);
3532 ShadowAndOriginCombiner SC(this, IRB);
3533 for (unsigned i = 0; i < NumArgOperands; ++i)
3534 SC.Add(V: I.getArgOperand(i));
3535 SC.Done(I: &I);
3536
3537 return true;
3538 }
3539
3540 /// Returns whether it was able to heuristically instrument unknown
3541 /// intrinsics.
3542 ///
3543 /// The main purpose of this code is to do something reasonable with all
3544 /// random intrinsics we might encounter, most importantly - SIMD intrinsics.
3545 /// We recognize several classes of intrinsics by their argument types and
3546 /// ModRefBehaviour and apply special instrumentation when we are reasonably
3547 /// sure that we know what the intrinsic does.
3548 ///
3549 /// We special-case intrinsics where this approach fails. See llvm.bswap
3550 /// handling as an example of that.
3551 bool maybeHandleUnknownIntrinsicUnlogged(IntrinsicInst &I) {
3552 unsigned NumArgOperands = I.arg_size();
3553 if (NumArgOperands == 0)
3554 return false;
3555
3556 if (NumArgOperands == 2 && I.getArgOperand(i: 0)->getType()->isPointerTy() &&
3557 I.getArgOperand(i: 1)->getType()->isVectorTy() &&
3558 I.getType()->isVoidTy() && !I.onlyReadsMemory()) {
3559 // This looks like a vector store.
3560 return handleVectorStoreIntrinsic(I);
3561 }
3562
3563 if (NumArgOperands == 1 && I.getArgOperand(i: 0)->getType()->isPointerTy() &&
3564 I.getType()->isVectorTy() && I.onlyReadsMemory()) {
3565 // This looks like a vector load.
3566 return handleVectorLoadIntrinsic(I);
3567 }
3568
3569 if (I.doesNotAccessMemory())
3570 if (maybeHandleSimpleNomemIntrinsic(I, /*trailingFlags=*/0))
3571 return true;
3572
3573 // FIXME: detect and handle SSE maskstore/maskload?
3574 // Some cases are now handled in handleAVXMasked{Load,Store}.
3575 return false;
3576 }
3577
3578 bool maybeHandleUnknownIntrinsic(IntrinsicInst &I) {
3579 if (maybeHandleUnknownIntrinsicUnlogged(I)) {
3580 if (ClDumpHeuristicInstructions)
3581 dumpInst(I, Prefix: "Heuristic");
3582
3583 LLVM_DEBUG(dbgs() << "UNKNOWN INSTRUCTION HANDLED HEURISTICALLY: " << I
3584 << "\n");
3585 return true;
3586 } else
3587 return false;
3588 }
3589
3590 void handleInvariantGroup(IntrinsicInst &I) {
3591 setShadow(V: &I, SV: getShadow(I: &I, i: 0));
3592 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
3593 }
3594
3595 void handleLifetimeStart(IntrinsicInst &I) {
3596 if (!PoisonStack)
3597 return;
3598 AllocaInst *AI = dyn_cast<AllocaInst>(Val: I.getArgOperand(i: 0));
3599 if (AI)
3600 LifetimeStartList.push_back(Elt: std::make_pair(x: &I, y&: AI));
3601 }
3602
3603 void handleBswap(IntrinsicInst &I) {
3604 IRBuilder<> IRB(&I);
3605 Value *Op = I.getArgOperand(i: 0);
3606 Type *OpType = Op->getType();
3607 setShadow(V: &I, SV: IRB.CreateIntrinsic(ID: Intrinsic::bswap, OverloadTypes: ArrayRef(&OpType, 1),
3608 Args: getShadow(V: Op)));
3609 setOrigin(V: &I, Origin: getOrigin(V: Op));
3610 }
3611
3612 // Uninitialized bits are ok if they appear after the leading/trailing 0's
3613 // and a 1. If the input is all zero, it is fully initialized iff
3614 // !is_zero_poison.
3615 //
3616 // e.g., for ctlz, with little-endian, if 0/1 are initialized bits with
3617 // concrete value 0/1, and ? is an uninitialized bit:
3618 // - 0001 0??? is fully initialized
3619 // - 000? ???? is fully uninitialized (*)
3620 // - ???? ???? is fully uninitialized
3621 // - 0000 0000 is fully uninitialized if is_zero_poison,
3622 // fully initialized otherwise
3623 //
3624 // (*) TODO: arguably, since the number of zeros is in the range [3, 8], we
3625 // only need to poison 4 bits.
3626 //
3627 // OutputShadow =
3628 // ((ConcreteZerosCount >= ShadowZerosCount) && !AllZeroShadow)
3629 // || (is_zero_poison && AllZeroSrc)
3630 void handleCountLeadingTrailingZeros(IntrinsicInst &I) {
3631 IRBuilder<> IRB(&I);
3632 Value *Src = I.getArgOperand(i: 0);
3633 Value *SrcShadow = getShadow(V: Src);
3634
3635 Value *False = IRB.getInt1(V: false);
3636 Value *ConcreteZerosCount = IRB.CreateIntrinsic(
3637 RetTy: I.getType(), ID: I.getIntrinsicID(), Args: {Src, /*is_zero_poison=*/False});
3638 Value *ShadowZerosCount = IRB.CreateIntrinsic(
3639 RetTy: I.getType(), ID: I.getIntrinsicID(), Args: {SrcShadow, /*is_zero_poison=*/False});
3640
3641 Value *CompareConcreteZeros = IRB.CreateICmpUGE(
3642 LHS: ConcreteZerosCount, RHS: ShadowZerosCount, Name: "_mscz_cmp_zeros");
3643
3644 Value *NotAllZeroShadow =
3645 IRB.CreateIsNotNull(Arg: SrcShadow, Name: "_mscz_shadow_not_null");
3646 Value *OutputShadow =
3647 IRB.CreateAnd(LHS: CompareConcreteZeros, RHS: NotAllZeroShadow, Name: "_mscz_main");
3648
3649 // If zero poison is requested, mix in with the shadow
3650 Constant *IsZeroPoison = cast<Constant>(Val: I.getOperand(i_nocapture: 1));
3651 if (!IsZeroPoison->isNullValue()) {
3652 Value *BoolZeroPoison = IRB.CreateIsNull(Arg: Src, Name: "_mscz_bzp");
3653 OutputShadow = IRB.CreateOr(LHS: OutputShadow, RHS: BoolZeroPoison, Name: "_mscz_bs");
3654 }
3655
3656 OutputShadow = IRB.CreateSExt(V: OutputShadow, DestTy: getShadowTy(V: Src), Name: "_mscz_os");
3657
3658 setShadow(V: &I, SV: OutputShadow);
3659 setOriginForNaryOp(I);
3660 }
3661
3662 /// Some instructions have additional zero-elements in the return type
3663 /// e.g., <16 x i8> @llvm.x86.avx512.mask.pmov.qb.512(<8 x i64>, ...)
3664 ///
3665 /// This function will return a vector type with the same number of elements
3666 /// as the input, but same per-element width as the return value e.g.,
3667 /// <8 x i8>.
3668 FixedVectorType *maybeShrinkVectorShadowType(Value *Src, IntrinsicInst &I) {
3669 assert(isa<FixedVectorType>(getShadowTy(&I)));
3670 FixedVectorType *ShadowType = cast<FixedVectorType>(Val: getShadowTy(V: &I));
3671
3672 // TODO: generalize beyond 2x?
3673 if (ShadowType->getElementCount() ==
3674 cast<VectorType>(Val: Src->getType())->getElementCount() * 2)
3675 ShadowType = FixedVectorType::getHalfElementsVectorType(VTy: ShadowType);
3676
3677 assert(ShadowType->getElementCount() ==
3678 cast<VectorType>(Src->getType())->getElementCount());
3679
3680 return ShadowType;
3681 }
3682
3683 /// Doubles the length of a vector shadow (extending with zeros) if necessary
3684 /// to match the length of the shadow for the instruction.
3685 /// If scalar types of the vectors are different, it will use the type of the
3686 /// input vector.
3687 /// This is more type-safe than CreateShadowCast().
3688 Value *maybeExtendVectorShadowWithZeros(Value *Shadow, IntrinsicInst &I) {
3689 IRBuilder<> IRB(&I);
3690 assert(isa<FixedVectorType>(Shadow->getType()));
3691 assert(isa<FixedVectorType>(I.getType()));
3692
3693 Value *FullShadow = getCleanShadow(V: &I);
3694 unsigned ShadowNumElems =
3695 cast<FixedVectorType>(Val: Shadow->getType())->getNumElements();
3696 unsigned FullShadowNumElems =
3697 cast<FixedVectorType>(Val: FullShadow->getType())->getNumElements();
3698
3699 assert((ShadowNumElems == FullShadowNumElems) ||
3700 (ShadowNumElems * 2 == FullShadowNumElems));
3701
3702 if (ShadowNumElems == FullShadowNumElems) {
3703 FullShadow = Shadow;
3704 } else {
3705 // TODO: generalize beyond 2x?
3706 SmallVector<int, 32> ShadowMask(FullShadowNumElems);
3707 std::iota(first: ShadowMask.begin(), last: ShadowMask.end(), value: 0);
3708
3709 // Append zeros
3710 FullShadow =
3711 IRB.CreateShuffleVector(V1: Shadow, V2: getCleanShadow(V: Shadow), Mask: ShadowMask);
3712 }
3713
3714 return FullShadow;
3715 }
3716
3717 /// Handle x86 SSE vector conversion.
3718 ///
3719 /// e.g., single-precision to half-precision conversion:
3720 /// <8 x i16> @llvm.x86.vcvtps2ph.256(<8 x float> %a0, i32 0)
3721 /// <8 x i16> @llvm.x86.vcvtps2ph.128(<4 x float> %a0, i32 0)
3722 ///
3723 /// floating-point to integer:
3724 /// <4 x i32> @llvm.x86.sse2.cvtps2dq(<4 x float>)
3725 /// <4 x i32> @llvm.x86.sse2.cvtpd2dq(<2 x double>)
3726 ///
3727 /// Note: if the output has more elements, they are zero-initialized (and
3728 /// therefore the shadow will also be initialized).
3729 ///
3730 /// This differs from handleSSEVectorConvertIntrinsic() because it
3731 /// propagates uninitialized shadow (instead of checking the shadow).
3732 void handleSSEVectorConvertIntrinsicByProp(IntrinsicInst &I,
3733 bool HasRoundingMode) {
3734 if (HasRoundingMode) {
3735 assert(I.arg_size() == 2);
3736 [[maybe_unused]] Value *RoundingMode = I.getArgOperand(i: 1);
3737 assert(RoundingMode->getType()->isIntegerTy());
3738 } else {
3739 assert(I.arg_size() == 1);
3740 }
3741
3742 Value *Src = I.getArgOperand(i: 0);
3743 assert(Src->getType()->isVectorTy());
3744
3745 // The return type might have more elements than the input.
3746 // Temporarily shrink the return type's number of elements.
3747 VectorType *ShadowType = maybeShrinkVectorShadowType(Src, I);
3748
3749 IRBuilder<> IRB(&I);
3750 Value *S0 = getShadow(I: &I, i: 0);
3751
3752 /// For scalars:
3753 /// Since they are converting to and/or from floating-point, the output is:
3754 /// - fully uninitialized if *any* bit of the input is uninitialized
3755 /// - fully ininitialized if all bits of the input are ininitialized
3756 /// We apply the same principle on a per-field basis for vectors.
3757 Value *Shadow =
3758 IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: S0, RHS: getCleanShadow(V: S0)), DestTy: ShadowType);
3759
3760 // The return type might have more elements than the input.
3761 // Extend the return type back to its original width if necessary.
3762 Value *FullShadow = maybeExtendVectorShadowWithZeros(Shadow, I);
3763
3764 setShadow(V: &I, SV: FullShadow);
3765 setOriginForNaryOp(I);
3766 }
3767
3768 // Instrument x86 SSE vector convert intrinsic.
3769 //
3770 // This function instruments intrinsics like cvtsi2ss:
3771 // %Out = int_xxx_cvtyyy(%ConvertOp)
3772 // or
3773 // %Out = int_xxx_cvtyyy(%CopyOp, %ConvertOp)
3774 // Intrinsic converts \p NumUsedElements elements of \p ConvertOp to the same
3775 // number \p Out elements, and (if has 2 arguments) copies the rest of the
3776 // elements from \p CopyOp.
3777 // In most cases conversion involves floating-point value which may trigger a
3778 // hardware exception when not fully initialized. For this reason we require
3779 // \p ConvertOp[0:NumUsedElements] to be fully initialized and trap otherwise.
3780 // We copy the shadow of \p CopyOp[NumUsedElements:] to \p
3781 // Out[NumUsedElements:]. This means that intrinsics without \p CopyOp always
3782 // return a fully initialized value.
3783 //
3784 // For Arm NEON vector convert intrinsics, see
3785 // handleNEONVectorConvertIntrinsic().
3786 void handleSSEVectorConvertIntrinsic(IntrinsicInst &I, int NumUsedElements,
3787 bool HasRoundingMode = false) {
3788 IRBuilder<> IRB(&I);
3789 Value *CopyOp, *ConvertOp;
3790
3791 assert((!HasRoundingMode ||
3792 isa<ConstantInt>(I.getArgOperand(I.arg_size() - 1))) &&
3793 "Invalid rounding mode");
3794
3795 switch (I.arg_size() - HasRoundingMode) {
3796 case 2:
3797 CopyOp = I.getArgOperand(i: 0);
3798 ConvertOp = I.getArgOperand(i: 1);
3799 break;
3800 case 1:
3801 ConvertOp = I.getArgOperand(i: 0);
3802 CopyOp = nullptr;
3803 break;
3804 default:
3805 llvm_unreachable("Cvt intrinsic with unsupported number of arguments.");
3806 }
3807
3808 // The first *NumUsedElements* elements of ConvertOp are converted to the
3809 // same number of output elements. The rest of the output is copied from
3810 // CopyOp, or (if not available) filled with zeroes.
3811 // Combine shadow for elements of ConvertOp that are used in this operation,
3812 // and insert a check.
3813 // FIXME: consider propagating shadow of ConvertOp, at least in the case of
3814 // int->any conversion.
3815 Value *ConvertShadow = getShadow(V: ConvertOp);
3816 Value *AggShadow = nullptr;
3817 if (ConvertOp->getType()->isVectorTy()) {
3818 AggShadow = IRB.CreateExtractElement(
3819 Vec: ConvertShadow, Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: 0));
3820 for (int i = 1; i < NumUsedElements; ++i) {
3821 Value *MoreShadow = IRB.CreateExtractElement(
3822 Vec: ConvertShadow, Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: i));
3823 AggShadow = IRB.CreateOr(LHS: AggShadow, RHS: MoreShadow);
3824 }
3825 } else {
3826 AggShadow = ConvertShadow;
3827 }
3828 assert(AggShadow->getType()->isIntegerTy());
3829 insertCheckShadow(Shadow: AggShadow, Origin: getOrigin(V: ConvertOp), OrigIns: &I);
3830
3831 // Build result shadow by zero-filling parts of CopyOp shadow that come from
3832 // ConvertOp.
3833 if (CopyOp) {
3834 assert(CopyOp->getType() == I.getType());
3835 assert(CopyOp->getType()->isVectorTy());
3836 Value *ResultShadow = getShadow(V: CopyOp);
3837 Type *EltTy = cast<VectorType>(Val: ResultShadow->getType())->getElementType();
3838 for (int i = 0; i < NumUsedElements; ++i) {
3839 ResultShadow = IRB.CreateInsertElement(
3840 Vec: ResultShadow, NewElt: ConstantInt::getNullValue(Ty: EltTy),
3841 Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: i));
3842 }
3843 setShadow(V: &I, SV: ResultShadow);
3844 setOrigin(V: &I, Origin: getOrigin(V: CopyOp));
3845 } else {
3846 setShadow(V: &I, SV: getCleanShadow(V: &I));
3847 setOrigin(V: &I, Origin: getCleanOrigin());
3848 }
3849 }
3850
3851 // Given a scalar or vector, extract lower 64 bits (or less), and return all
3852 // zeroes if it is zero, and all ones otherwise.
3853 Value *Lower64ShadowExtend(IRBuilder<> &IRB, Value *S, Type *T) {
3854 if (S->getType()->isVectorTy())
3855 S = CreateShadowCast(IRB, V: S, dstTy: IRB.getInt64Ty(), /* Signed */ true);
3856 assert(S->getType()->getPrimitiveSizeInBits() <= 64);
3857 Value *S2 = IRB.CreateICmpNE(LHS: S, RHS: getCleanShadow(V: S));
3858 return CreateShadowCast(IRB, V: S2, dstTy: T, /* Signed */ true);
3859 }
3860
3861 // Given a vector, extract its first element, and return all
3862 // zeroes if it is zero, and all ones otherwise.
3863 Value *LowerElementShadowExtend(IRBuilder<> &IRB, Value *S, Type *T) {
3864 Value *S1 = IRB.CreateExtractElement(Vec: S, Idx: (uint64_t)0);
3865 Value *S2 = IRB.CreateICmpNE(LHS: S1, RHS: getCleanShadow(V: S1));
3866 return CreateShadowCast(IRB, V: S2, dstTy: T, /* Signed */ true);
3867 }
3868
3869 Value *VariableShadowExtend(IRBuilder<> &IRB, Value *S) {
3870 Type *T = S->getType();
3871 assert(T->isVectorTy());
3872 Value *S2 = IRB.CreateICmpNE(LHS: S, RHS: getCleanShadow(V: S));
3873 return IRB.CreateSExt(V: S2, DestTy: T);
3874 }
3875
3876 // Instrument vector shift intrinsic.
3877 //
3878 // This function instruments intrinsics like int_x86_avx2_psll_w.
3879 // Intrinsic shifts %In by %ShiftSize bits.
3880 // %ShiftSize may be a vector. In that case the lower 64 bits determine shift
3881 // size, and the rest is ignored. Behavior is defined even if shift size is
3882 // greater than register (or field) width.
3883 void handleVectorShiftIntrinsic(IntrinsicInst &I, bool Variable) {
3884 assert(I.arg_size() == 2);
3885 IRBuilder<> IRB(&I);
3886 // If any of the S2 bits are poisoned, the whole thing is poisoned.
3887 // Otherwise perform the same shift on S1.
3888 Value *S1 = getShadow(I: &I, i: 0);
3889 Value *S2 = getShadow(I: &I, i: 1);
3890 Value *S2Conv = Variable ? VariableShadowExtend(IRB, S: S2)
3891 : Lower64ShadowExtend(IRB, S: S2, T: getShadowTy(V: &I));
3892 Value *V1 = I.getOperand(i_nocapture: 0);
3893 Value *V2 = I.getOperand(i_nocapture: 1);
3894 Value *Shift = IRB.CreateCall(FTy: I.getFunctionType(), Callee: I.getCalledOperand(),
3895 Args: {IRB.CreateBitCast(V: S1, DestTy: V1->getType()), V2});
3896 Shift = IRB.CreateBitCast(V: Shift, DestTy: getShadowTy(V: &I));
3897 setShadow(V: &I, SV: IRB.CreateOr(LHS: Shift, RHS: S2Conv));
3898 setOriginForNaryOp(I);
3899 }
3900
3901 // Get an MMX-sized (64-bit) vector type, or optionally, other sized
3902 // vectors.
3903 Type *getMMXVectorTy(unsigned EltSizeInBits,
3904 unsigned X86_MMXSizeInBits = 64) {
3905 assert(EltSizeInBits != 0 && (X86_MMXSizeInBits % EltSizeInBits) == 0 &&
3906 "Illegal MMX vector element size");
3907 return FixedVectorType::get(ElementType: IntegerType::get(C&: *MS.C, NumBits: EltSizeInBits),
3908 NumElts: X86_MMXSizeInBits / EltSizeInBits);
3909 }
3910
3911 // Returns a signed counterpart for an (un)signed-saturate-and-pack
3912 // intrinsic.
3913 Intrinsic::ID getSignedPackIntrinsic(Intrinsic::ID id) {
3914 switch (id) {
3915 case Intrinsic::x86_sse2_packsswb_128:
3916 case Intrinsic::x86_sse2_packuswb_128:
3917 return Intrinsic::x86_sse2_packsswb_128;
3918
3919 case Intrinsic::x86_sse2_packssdw_128:
3920 case Intrinsic::x86_sse41_packusdw:
3921 return Intrinsic::x86_sse2_packssdw_128;
3922
3923 case Intrinsic::x86_avx2_packsswb:
3924 case Intrinsic::x86_avx2_packuswb:
3925 return Intrinsic::x86_avx2_packsswb;
3926
3927 case Intrinsic::x86_avx2_packssdw:
3928 case Intrinsic::x86_avx2_packusdw:
3929 return Intrinsic::x86_avx2_packssdw;
3930
3931 case Intrinsic::x86_mmx_packsswb:
3932 case Intrinsic::x86_mmx_packuswb:
3933 return Intrinsic::x86_mmx_packsswb;
3934
3935 case Intrinsic::x86_mmx_packssdw:
3936 return Intrinsic::x86_mmx_packssdw;
3937
3938 case Intrinsic::x86_avx512_packssdw_512:
3939 case Intrinsic::x86_avx512_packusdw_512:
3940 return Intrinsic::x86_avx512_packssdw_512;
3941
3942 case Intrinsic::x86_avx512_packsswb_512:
3943 case Intrinsic::x86_avx512_packuswb_512:
3944 return Intrinsic::x86_avx512_packsswb_512;
3945
3946 default:
3947 llvm_unreachable("unexpected intrinsic id");
3948 }
3949 }
3950
3951 // Instrument vector pack intrinsic.
3952 //
3953 // This function instruments intrinsics like x86_mmx_packsswb, that
3954 // packs elements of 2 input vectors into half as many bits with saturation.
3955 // Shadow is propagated with the signed variant of the same intrinsic applied
3956 // to sext(Sa != zeroinitializer), sext(Sb != zeroinitializer).
3957 // MMXEltSizeInBits is used only for x86mmx arguments.
3958 //
3959 // TODO: consider using GetMinMaxUnsigned() to handle saturation precisely
3960 void handleVectorPackIntrinsic(IntrinsicInst &I,
3961 unsigned MMXEltSizeInBits = 0) {
3962 assert(I.arg_size() == 2);
3963 IRBuilder<> IRB(&I);
3964 Value *S1 = getShadow(I: &I, i: 0);
3965 Value *S2 = getShadow(I: &I, i: 1);
3966 assert(S1->getType()->isVectorTy());
3967
3968 // SExt and ICmpNE below must apply to individual elements of input vectors.
3969 // In case of x86mmx arguments, cast them to appropriate vector types and
3970 // back.
3971 Type *T =
3972 MMXEltSizeInBits ? getMMXVectorTy(EltSizeInBits: MMXEltSizeInBits) : S1->getType();
3973 if (MMXEltSizeInBits) {
3974 S1 = IRB.CreateBitCast(V: S1, DestTy: T);
3975 S2 = IRB.CreateBitCast(V: S2, DestTy: T);
3976 }
3977 Value *S1_ext =
3978 IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: S1, RHS: Constant::getNullValue(Ty: T)), DestTy: T);
3979 Value *S2_ext =
3980 IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: S2, RHS: Constant::getNullValue(Ty: T)), DestTy: T);
3981 if (MMXEltSizeInBits) {
3982 S1_ext = IRB.CreateBitCast(V: S1_ext, DestTy: getMMXVectorTy(EltSizeInBits: 64));
3983 S2_ext = IRB.CreateBitCast(V: S2_ext, DestTy: getMMXVectorTy(EltSizeInBits: 64));
3984 }
3985
3986 Value *S = IRB.CreateIntrinsic(ID: getSignedPackIntrinsic(id: I.getIntrinsicID()),
3987 Args: {S1_ext, S2_ext}, /*FMFSource=*/nullptr,
3988 Name: "_msprop_vector_pack");
3989 if (MMXEltSizeInBits)
3990 S = IRB.CreateBitCast(V: S, DestTy: getShadowTy(V: &I));
3991 setShadow(V: &I, SV: S);
3992 setOriginForNaryOp(I);
3993 }
3994
3995 // Convert `Mask` into `<n x i1>`.
3996 Constant *createDppMask(unsigned Width, unsigned Mask) {
3997 SmallVector<Constant *, 4> R(Width);
3998 for (auto &M : R) {
3999 M = ConstantInt::getBool(Context&: F.getContext(), V: Mask & 1);
4000 Mask >>= 1;
4001 }
4002 return ConstantVector::get(V: R);
4003 }
4004
4005 // Calculate output shadow as array of booleans `<n x i1>`, assuming if any
4006 // arg is poisoned, entire dot product is poisoned.
4007 Value *findDppPoisonedOutput(IRBuilder<> &IRB, Value *S, unsigned SrcMask,
4008 unsigned DstMask) {
4009 const unsigned Width =
4010 cast<FixedVectorType>(Val: S->getType())->getNumElements();
4011
4012 S = IRB.CreateSelect(C: createDppMask(Width, Mask: SrcMask), True: S,
4013 False: Constant::getNullValue(Ty: S->getType()));
4014 Value *SElem = IRB.CreateOrReduce(Src: S);
4015 Value *IsClean = IRB.CreateIsNull(Arg: SElem, Name: "_msdpp");
4016 Value *DstMaskV = createDppMask(Width, Mask: DstMask);
4017
4018 return IRB.CreateSelect(
4019 C: IsClean, True: Constant::getNullValue(Ty: DstMaskV->getType()), False: DstMaskV);
4020 }
4021
4022 // See `Intel Intrinsics Guide` for `_dp_p*` instructions.
4023 //
4024 // 2 and 4 element versions produce single scalar of dot product, and then
4025 // puts it into elements of output vector, selected by 4 lowest bits of the
4026 // mask. Top 4 bits of the mask control which elements of input to use for dot
4027 // product.
4028 //
4029 // 8 element version mask still has only 4 bit for input, and 4 bit for output
4030 // mask. According to the spec it just operates as 4 element version on first
4031 // 4 elements of inputs and output, and then on last 4 elements of inputs and
4032 // output.
4033 void handleDppIntrinsic(IntrinsicInst &I) {
4034 IRBuilder<> IRB(&I);
4035
4036 Value *S0 = getShadow(I: &I, i: 0);
4037 Value *S1 = getShadow(I: &I, i: 1);
4038 Value *S = IRB.CreateOr(LHS: S0, RHS: S1);
4039
4040 const unsigned Width =
4041 cast<FixedVectorType>(Val: S->getType())->getNumElements();
4042 assert(Width == 2 || Width == 4 || Width == 8);
4043
4044 const unsigned Mask = cast<ConstantInt>(Val: I.getArgOperand(i: 2))->getZExtValue();
4045 const unsigned SrcMask = Mask >> 4;
4046 const unsigned DstMask = Mask & 0xf;
4047
4048 // Calculate shadow as `<n x i1>`.
4049 Value *SI1 = findDppPoisonedOutput(IRB, S, SrcMask, DstMask);
4050 if (Width == 8) {
4051 // First 4 elements of shadow are already calculated. `makeDppShadow`
4052 // operats on 32 bit masks, so we can just shift masks, and repeat.
4053 SI1 = IRB.CreateOr(
4054 LHS: SI1, RHS: findDppPoisonedOutput(IRB, S, SrcMask: SrcMask << 4, DstMask: DstMask << 4));
4055 }
4056 // Extend to real size of shadow, poisoning either all or none bits of an
4057 // element.
4058 S = IRB.CreateSExt(V: SI1, DestTy: S->getType(), Name: "_msdpp");
4059
4060 setShadow(V: &I, SV: S);
4061 setOriginForNaryOp(I);
4062 }
4063
4064 Value *convertBlendvToSelectMask(IRBuilder<> &IRB, Value *C) {
4065 C = CreateAppToShadowCast(IRB, V: C);
4066 FixedVectorType *FVT = cast<FixedVectorType>(Val: C->getType());
4067 unsigned ElSize = FVT->getElementType()->getPrimitiveSizeInBits();
4068 C = IRB.CreateAShr(LHS: C, RHS: ElSize - 1);
4069 FVT = FixedVectorType::get(ElementType: IRB.getInt1Ty(), NumElts: FVT->getNumElements());
4070 return IRB.CreateTrunc(V: C, DestTy: FVT);
4071 }
4072
4073 // `blendv(f, t, c)` is effectively `select(c[top_bit], t, f)`.
4074 void handleBlendvIntrinsic(IntrinsicInst &I) {
4075 Value *C = I.getOperand(i_nocapture: 2);
4076 Value *T = I.getOperand(i_nocapture: 1);
4077 Value *F = I.getOperand(i_nocapture: 0);
4078
4079 Value *Sc = getShadow(I: &I, i: 2);
4080 Value *Oc = MS.TrackOrigins ? getOrigin(V: C) : nullptr;
4081
4082 {
4083 IRBuilder<> IRB(&I);
4084 // Extract top bit from condition and its shadow.
4085 C = convertBlendvToSelectMask(IRB, C);
4086 Sc = convertBlendvToSelectMask(IRB, C: Sc);
4087
4088 setShadow(V: C, SV: Sc);
4089 setOrigin(V: C, Origin: Oc);
4090 }
4091
4092 handleSelectLikeInst(I, B: C, C: T, D: F);
4093 }
4094
4095 // Instrument sum-of-absolute-differences intrinsic.
4096 void handleVectorSadIntrinsic(IntrinsicInst &I, bool IsMMX = false) {
4097 const unsigned SignificantBitsPerResultElement = 16;
4098 Type *ResTy = IsMMX ? IntegerType::get(C&: *MS.C, NumBits: 64) : I.getType();
4099 unsigned ZeroBitsPerResultElement =
4100 ResTy->getScalarSizeInBits() - SignificantBitsPerResultElement;
4101
4102 IRBuilder<> IRB(&I);
4103 auto *Shadow0 = getShadow(I: &I, i: 0);
4104 auto *Shadow1 = getShadow(I: &I, i: 1);
4105 Value *S = IRB.CreateOr(LHS: Shadow0, RHS: Shadow1);
4106 S = IRB.CreateBitCast(V: S, DestTy: ResTy);
4107 S = IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: S, RHS: Constant::getNullValue(Ty: ResTy)),
4108 DestTy: ResTy);
4109 S = IRB.CreateLShr(LHS: S, RHS: ZeroBitsPerResultElement);
4110 S = IRB.CreateBitCast(V: S, DestTy: getShadowTy(V: &I));
4111 setShadow(V: &I, SV: S);
4112 setOriginForNaryOp(I);
4113 }
4114
4115 // Instrument dot-product / multiply-add(-accumulate)? intrinsics.
4116 //
4117 // e.g., Two operands:
4118 // <4 x i32> @llvm.x86.sse2.pmadd.wd(<8 x i16> %a, <8 x i16> %b)
4119 //
4120 // Two operands which require an EltSizeInBits override:
4121 // <1 x i64> @llvm.x86.mmx.pmadd.wd(<1 x i64> %a, <1 x i64> %b)
4122 //
4123 // Three operands:
4124 // <4 x i32> @llvm.x86.avx512.vpdpbusd.128
4125 // (<4 x i32> %s, <16 x i8> %a, <16 x i8> %b)
4126 // <2 x float> @llvm.aarch64.neon.bfdot.v2f32.v4bf16
4127 // (<2 x float> %acc, <4 x bfloat> %a, <4 x bfloat> %b)
4128 // (these are equivalent to multiply-add on %a and %b, followed by
4129 // adding/"accumulating" %s. "Accumulation" stores the result in one
4130 // of the source registers, but this accumulate vs. add distinction
4131 // is lost when dealing with LLVM intrinsics.)
4132 //
4133 // ZeroPurifies means that multiplying a known-zero with an uninitialized
4134 // value results in an initialized value. This is applicable for integer
4135 // multiplication, but not floating-point (counter-example: NaN).
4136 void handleVectorDotProductIntrinsic(IntrinsicInst &I,
4137 unsigned ReductionFactor,
4138 bool ZeroPurifies,
4139 unsigned EltSizeInBits,
4140 enum OddOrEvenLanes Lanes) {
4141 IRBuilder<> IRB(&I);
4142
4143 [[maybe_unused]] FixedVectorType *ReturnType =
4144 cast<FixedVectorType>(Val: I.getType());
4145 assert(isa<FixedVectorType>(ReturnType));
4146
4147 // Vectors A and B, and shadows
4148 Value *Va = nullptr;
4149 Value *Vb = nullptr;
4150 Value *Sa = nullptr;
4151 Value *Sb = nullptr;
4152
4153 assert(I.arg_size() == 2 || I.arg_size() == 3);
4154 if (I.arg_size() == 2) {
4155 assert(Lanes == kBothLanes);
4156
4157 Va = I.getOperand(i_nocapture: 0);
4158 Vb = I.getOperand(i_nocapture: 1);
4159
4160 Sa = getShadow(I: &I, i: 0);
4161 Sb = getShadow(I: &I, i: 1);
4162 } else if (I.arg_size() == 3) {
4163 // Operand 0 is the accumulator. We will deal with that below.
4164 Va = I.getOperand(i_nocapture: 1);
4165 Vb = I.getOperand(i_nocapture: 2);
4166
4167 Sa = getShadow(I: &I, i: 1);
4168 Sb = getShadow(I: &I, i: 2);
4169
4170 if (Lanes == kEvenLanes || Lanes == kOddLanes) {
4171 // Convert < S0, S1, S2, S3, S4, S5, S6, S7 >
4172 // to < S0, S0, S2, S2, S4, S4, S6, S6 > (if even)
4173 // to < S1, S1, S3, S3, S5, S5, S7, S7 > (if odd)
4174 //
4175 // Note: for aarch64.neon.bfmlalb/t, the odd/even-indexed values are
4176 // zeroed, not duplicated. However, for shadow propagation, this
4177 // distinction is unimportant because Step 1 below will squeeze
4178 // each pair of elements (e.g., [S0, S0]) into a single bit, and
4179 // we only care if it is fully initialized.
4180
4181 FixedVectorType *InputShadowType = cast<FixedVectorType>(Val: Sa->getType());
4182 unsigned Width = InputShadowType->getNumElements();
4183
4184 Sa = IRB.CreateShuffleVector(
4185 V: Sa, Mask: getPclmulMask(Width, /*OddElements=*/Lanes == kOddLanes));
4186 Sb = IRB.CreateShuffleVector(
4187 V: Sb, Mask: getPclmulMask(Width, /*OddElements=*/Lanes == kOddLanes));
4188 }
4189 }
4190
4191 FixedVectorType *ParamType = cast<FixedVectorType>(Val: Va->getType());
4192 assert(ParamType == Vb->getType());
4193
4194 assert(ParamType->getPrimitiveSizeInBits() ==
4195 ReturnType->getPrimitiveSizeInBits());
4196
4197 if (I.arg_size() == 3) {
4198 [[maybe_unused]] auto *AccumulatorType =
4199 cast<FixedVectorType>(Val: I.getOperand(i_nocapture: 0)->getType());
4200 assert(AccumulatorType == ReturnType);
4201 }
4202
4203 FixedVectorType *ImplicitReturnType =
4204 cast<FixedVectorType>(Val: getShadowTy(OrigTy: ReturnType));
4205 // Step 1: instrument multiplication of corresponding vector elements
4206 if (EltSizeInBits) {
4207 ImplicitReturnType = cast<FixedVectorType>(
4208 Val: getMMXVectorTy(EltSizeInBits: EltSizeInBits * ReductionFactor,
4209 X86_MMXSizeInBits: ParamType->getPrimitiveSizeInBits()));
4210 ParamType = cast<FixedVectorType>(
4211 Val: getMMXVectorTy(EltSizeInBits, X86_MMXSizeInBits: ParamType->getPrimitiveSizeInBits()));
4212
4213 Va = IRB.CreateBitCast(V: Va, DestTy: ParamType);
4214 Vb = IRB.CreateBitCast(V: Vb, DestTy: ParamType);
4215
4216 Sa = IRB.CreateBitCast(V: Sa, DestTy: getShadowTy(OrigTy: ParamType));
4217 Sb = IRB.CreateBitCast(V: Sb, DestTy: getShadowTy(OrigTy: ParamType));
4218 } else {
4219 assert(ParamType->getNumElements() ==
4220 ReturnType->getNumElements() * ReductionFactor);
4221 }
4222
4223 // Each element of the vector is represented by a single bit (poisoned or
4224 // not) e.g., <8 x i1>.
4225 Value *SaNonZero = IRB.CreateIsNotNull(Arg: Sa);
4226 Value *SbNonZero = IRB.CreateIsNotNull(Arg: Sb);
4227 Value *And;
4228 if (ZeroPurifies) {
4229 // Multiplying an *initialized* zero by an uninitialized element results
4230 // in an initialized zero element.
4231 //
4232 // This is analogous to bitwise AND, where "AND" of 0 and a poisoned value
4233 // results in an unpoisoned value.
4234 Value *VaInt = Va;
4235 Value *VbInt = Vb;
4236 if (!Va->getType()->isIntegerTy()) {
4237 VaInt = CreateAppToShadowCast(IRB, V: Va);
4238 VbInt = CreateAppToShadowCast(IRB, V: Vb);
4239 }
4240
4241 // We check for non-zero on a per-element basis, not per-bit.
4242 Value *VaNonZero = IRB.CreateIsNotNull(Arg: VaInt);
4243 Value *VbNonZero = IRB.CreateIsNotNull(Arg: VbInt);
4244
4245 And = handleBitwiseAnd(IRB, V1: VaNonZero, V2: VbNonZero, S1: SaNonZero, S2: SbNonZero);
4246 } else {
4247 And = IRB.CreateOr(Ops: {SaNonZero, SbNonZero});
4248 }
4249
4250 // Extend <8 x i1> to <8 x i16>.
4251 // (The real pmadd intrinsic would have computed intermediate values of
4252 // <8 x i32>, but that is irrelevant for our shadow purposes because we
4253 // consider each element to be either fully initialized or fully
4254 // uninitialized.)
4255 And = IRB.CreateSExt(V: And, DestTy: Sa->getType());
4256
4257 // Step 2: instrument horizontal add
4258 // We don't need bit-precise horizontalReduce because we only want to check
4259 // if each pair/quad of elements is fully zero.
4260 // Cast to <4 x i32>.
4261 Value *Horizontal = IRB.CreateBitCast(V: And, DestTy: ImplicitReturnType);
4262
4263 // Compute <4 x i1>, then extend back to <4 x i32>.
4264 Value *OutShadow = IRB.CreateSExt(
4265 V: IRB.CreateICmpNE(LHS: Horizontal,
4266 RHS: Constant::getNullValue(Ty: Horizontal->getType())),
4267 DestTy: ImplicitReturnType);
4268
4269 // Cast it back to the required fake return type (if MMX: <1 x i64>; for
4270 // AVX, it is already correct).
4271 if (EltSizeInBits)
4272 OutShadow = CreateShadowCast(IRB, V: OutShadow, dstTy: getShadowTy(V: &I));
4273
4274 // Step 3 (if applicable): instrument accumulator
4275 if (I.arg_size() == 3)
4276 OutShadow = IRB.CreateOr(LHS: OutShadow, RHS: getShadow(I: &I, i: 0));
4277
4278 setShadow(V: &I, SV: OutShadow);
4279 setOriginForNaryOp(I);
4280 }
4281
4282 // Instrument compare-packed intrinsic.
4283 //
4284 // x86 has the predicate as the third operand, which is ImmArg e.g.,
4285 // - <4 x double> @llvm.x86.avx.cmp.pd.256(<4 x double>, <4 x double>, i8)
4286 // - <2 x double> @llvm.x86.sse2.cmp.pd(<2 x double>, <2 x double>, i8)
4287 //
4288 // while Arm has separate intrinsics for >= and > e.g.,
4289 // - <2 x i32> @llvm.aarch64.neon.facge.v2i32.v2f32
4290 // (<2 x float> %A, <2 x float>)
4291 // - <2 x i32> @llvm.aarch64.neon.facgt.v2i32.v2f32
4292 // (<2 x float> %A, <2 x float>)
4293 //
4294 // Bonus: this also handles scalar cases e.g.,
4295 // - i32 @llvm.aarch64.neon.facgt.i32.f32(float %A, float %B)
4296 void handleVectorComparePackedIntrinsic(IntrinsicInst &I,
4297 bool PredicateAsOperand) {
4298 if (PredicateAsOperand) {
4299 assert(I.arg_size() == 3);
4300 assert(I.paramHasAttr(2, Attribute::ImmArg));
4301 } else
4302 assert(I.arg_size() == 2);
4303
4304 IRBuilder<> IRB(&I);
4305
4306 // Basically, an or followed by sext(icmp ne 0) to end up with all-zeros or
4307 // all-ones shadow.
4308 Type *ResTy = getShadowTy(V: &I);
4309 auto *Shadow0 = getShadow(I: &I, i: 0);
4310 auto *Shadow1 = getShadow(I: &I, i: 1);
4311 Value *S0 = IRB.CreateOr(LHS: Shadow0, RHS: Shadow1);
4312 Value *S = IRB.CreateSExt(
4313 V: IRB.CreateICmpNE(LHS: S0, RHS: Constant::getNullValue(Ty: ResTy)), DestTy: ResTy);
4314 setShadow(V: &I, SV: S);
4315 setOriginForNaryOp(I);
4316 }
4317
4318 // Instrument compare-scalar intrinsic.
4319 // This handles both cmp* intrinsics which return the result in the first
4320 // element of a vector, and comi* which return the result as i32.
4321 void handleVectorCompareScalarIntrinsic(IntrinsicInst &I) {
4322 IRBuilder<> IRB(&I);
4323 auto *Shadow0 = getShadow(I: &I, i: 0);
4324 auto *Shadow1 = getShadow(I: &I, i: 1);
4325 Value *S0 = IRB.CreateOr(LHS: Shadow0, RHS: Shadow1);
4326 Value *S = LowerElementShadowExtend(IRB, S: S0, T: getShadowTy(V: &I));
4327 setShadow(V: &I, SV: S);
4328 setOriginForNaryOp(I);
4329 }
4330
4331 // Instrument generic vector reduction intrinsics
4332 // by ORing together all their fields.
4333 //
4334 // If AllowShadowCast is true, the return type does not need to be the same
4335 // type as the fields
4336 // e.g., declare i32 @llvm.aarch64.neon.uaddv.i32.v16i8(<16 x i8>)
4337 void handleVectorReduceIntrinsic(IntrinsicInst &I, bool AllowShadowCast) {
4338 assert(I.arg_size() == 1);
4339
4340 IRBuilder<> IRB(&I);
4341 Value *S = IRB.CreateOrReduce(Src: getShadow(I: &I, i: 0));
4342 if (AllowShadowCast)
4343 S = CreateShadowCast(IRB, V: S, dstTy: getShadowTy(V: &I));
4344 else
4345 assert(S->getType() == getShadowTy(&I));
4346 setShadow(V: &I, SV: S);
4347 setOriginForNaryOp(I);
4348 }
4349
4350 // Similar to handleVectorReduceIntrinsic but with an initial starting value.
4351 // e.g., call float @llvm.vector.reduce.fadd.f32.v2f32(float %a0, <2 x float>
4352 // %a1)
4353 // shadow = shadow[a0] | shadow[a1.0] | shadow[a1.1]
4354 //
4355 // The type of the return value, initial starting value, and elements of the
4356 // vector must be identical.
4357 void handleVectorReduceWithStarterIntrinsic(IntrinsicInst &I) {
4358 assert(I.arg_size() == 2);
4359
4360 IRBuilder<> IRB(&I);
4361 Value *Shadow0 = getShadow(I: &I, i: 0);
4362 Value *Shadow1 = IRB.CreateOrReduce(Src: getShadow(I: &I, i: 1));
4363 assert(Shadow0->getType() == Shadow1->getType());
4364 Value *S = IRB.CreateOr(LHS: Shadow0, RHS: Shadow1);
4365 assert(S->getType() == getShadowTy(&I));
4366 setShadow(V: &I, SV: S);
4367 setOriginForNaryOp(I);
4368 }
4369
4370 // Instrument vector.reduce.or intrinsic.
4371 // Valid (non-poisoned) set bits in the operand pull low the
4372 // corresponding shadow bits.
4373 void handleVectorReduceOrIntrinsic(IntrinsicInst &I) {
4374 assert(I.arg_size() == 1);
4375
4376 IRBuilder<> IRB(&I);
4377 Value *OperandShadow = getShadow(I: &I, i: 0);
4378 Value *OperandUnsetBits = IRB.CreateNot(V: I.getOperand(i_nocapture: 0));
4379 Value *OperandUnsetOrPoison = IRB.CreateOr(LHS: OperandUnsetBits, RHS: OperandShadow);
4380 // Bit N is clean if any field's bit N is 1 and unpoison
4381 Value *OutShadowMask = IRB.CreateAndReduce(Src: OperandUnsetOrPoison);
4382 // Otherwise, it is clean if every field's bit N is unpoison
4383 Value *OrShadow = IRB.CreateOrReduce(Src: OperandShadow);
4384 Value *S = IRB.CreateAnd(LHS: OutShadowMask, RHS: OrShadow);
4385
4386 setShadow(V: &I, SV: S);
4387 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
4388 }
4389
4390 // Instrument vector.reduce.and intrinsic.
4391 // Valid (non-poisoned) unset bits in the operand pull down the
4392 // corresponding shadow bits.
4393 void handleVectorReduceAndIntrinsic(IntrinsicInst &I) {
4394 assert(I.arg_size() == 1);
4395
4396 IRBuilder<> IRB(&I);
4397 Value *OperandShadow = getShadow(I: &I, i: 0);
4398 Value *OperandSetOrPoison = IRB.CreateOr(LHS: I.getOperand(i_nocapture: 0), RHS: OperandShadow);
4399 // Bit N is clean if any field's bit N is 0 and unpoison
4400 Value *OutShadowMask = IRB.CreateAndReduce(Src: OperandSetOrPoison);
4401 // Otherwise, it is clean if every field's bit N is unpoison
4402 Value *OrShadow = IRB.CreateOrReduce(Src: OperandShadow);
4403 Value *S = IRB.CreateAnd(LHS: OutShadowMask, RHS: OrShadow);
4404
4405 setShadow(V: &I, SV: S);
4406 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
4407 }
4408
4409 void handleStmxcsr(IntrinsicInst &I) {
4410 IRBuilder<> IRB(&I);
4411 Value *Addr = I.getArgOperand(i: 0);
4412 Type *Ty = IRB.getInt32Ty();
4413 Value *ShadowPtr =
4414 getShadowOriginPtr(Addr, IRB, ShadowTy: Ty, Alignment: Align(1), /*isStore*/ true).first;
4415
4416 IRB.CreateStore(Val: getCleanShadow(OrigTy: Ty), Ptr: ShadowPtr);
4417
4418 if (ClCheckAccessAddress)
4419 insertCheckShadowOf(Val: Addr, OrigIns: &I);
4420 }
4421
4422 void handleLdmxcsr(IntrinsicInst &I) {
4423 if (!InsertChecks)
4424 return;
4425
4426 IRBuilder<> IRB(&I);
4427 Value *Addr = I.getArgOperand(i: 0);
4428 Type *Ty = IRB.getInt32Ty();
4429 const Align Alignment = Align(1);
4430 Value *ShadowPtr, *OriginPtr;
4431 std::tie(args&: ShadowPtr, args&: OriginPtr) =
4432 getShadowOriginPtr(Addr, IRB, ShadowTy: Ty, Alignment, /*isStore*/ false);
4433
4434 if (ClCheckAccessAddress)
4435 insertCheckShadowOf(Val: Addr, OrigIns: &I);
4436
4437 Value *Shadow = IRB.CreateAlignedLoad(Ty, Ptr: ShadowPtr, Align: Alignment, Name: "_ldmxcsr");
4438 Value *Origin = MS.TrackOrigins ? IRB.CreateLoad(Ty: MS.OriginTy, Ptr: OriginPtr)
4439 : getCleanOrigin();
4440 insertCheckShadow(Shadow, Origin, OrigIns: &I);
4441 }
4442
4443 void handleMaskedExpandLoad(IntrinsicInst &I) {
4444 IRBuilder<> IRB(&I);
4445 Value *Ptr = I.getArgOperand(i: 0);
4446 MaybeAlign Align = I.getParamAlign(ArgNo: 0);
4447 Value *Mask = I.getArgOperand(i: 1);
4448 Value *PassThru = I.getArgOperand(i: 2);
4449
4450 if (ClCheckAccessAddress) {
4451 insertCheckShadowOf(Val: Ptr, OrigIns: &I);
4452 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4453 }
4454
4455 if (!PropagateShadow) {
4456 setShadow(V: &I, SV: getCleanShadow(V: &I));
4457 setOrigin(V: &I, Origin: getCleanOrigin());
4458 return;
4459 }
4460
4461 Type *ShadowTy = getShadowTy(V: &I);
4462 Type *ElementShadowTy = cast<VectorType>(Val: ShadowTy)->getElementType();
4463 auto [ShadowPtr, OriginPtr] =
4464 getShadowOriginPtr(Addr: Ptr, IRB, ShadowTy: ElementShadowTy, Alignment: Align, /*isStore*/ false);
4465
4466 Value *Shadow =
4467 IRB.CreateMaskedExpandLoad(Ty: ShadowTy, Ptr: ShadowPtr, Align, Mask,
4468 PassThru: getShadow(V: PassThru), Name: "_msmaskedexpload");
4469
4470 setShadow(V: &I, SV: Shadow);
4471
4472 // TODO: Store origins.
4473 setOrigin(V: &I, Origin: getCleanOrigin());
4474 }
4475
4476 void handleMaskedCompressStore(IntrinsicInst &I) {
4477 IRBuilder<> IRB(&I);
4478 Value *Values = I.getArgOperand(i: 0);
4479 Value *Ptr = I.getArgOperand(i: 1);
4480 MaybeAlign Align = I.getParamAlign(ArgNo: 1);
4481 Value *Mask = I.getArgOperand(i: 2);
4482
4483 if (ClCheckAccessAddress) {
4484 insertCheckShadowOf(Val: Ptr, OrigIns: &I);
4485 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4486 }
4487
4488 Value *Shadow = getShadow(V: Values);
4489 Type *ElementShadowTy =
4490 getShadowTy(OrigTy: cast<VectorType>(Val: Values->getType())->getElementType());
4491 auto [ShadowPtr, OriginPtrs] =
4492 getShadowOriginPtr(Addr: Ptr, IRB, ShadowTy: ElementShadowTy, Alignment: Align, /*isStore*/ true);
4493
4494 IRB.CreateMaskedCompressStore(Val: Shadow, Ptr: ShadowPtr, Align, Mask);
4495
4496 // TODO: Store origins.
4497 }
4498
4499 void handleMaskedGather(IntrinsicInst &I) {
4500 IRBuilder<> IRB(&I);
4501 Value *Ptrs = I.getArgOperand(i: 0);
4502 const Align Alignment = I.getParamAlign(ArgNo: 0).valueOrOne();
4503 Value *Mask = I.getArgOperand(i: 1);
4504 Value *PassThru = I.getArgOperand(i: 2);
4505
4506 Type *PtrsShadowTy = getShadowTy(V: Ptrs);
4507 if (ClCheckAccessAddress) {
4508 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4509 Value *MaskedPtrShadow = IRB.CreateSelect(
4510 C: Mask, True: getShadow(V: Ptrs), False: Constant::getNullValue(Ty: (PtrsShadowTy)),
4511 Name: "_msmaskedptrs");
4512 insertCheckShadow(Shadow: MaskedPtrShadow, Origin: getOrigin(V: Ptrs), OrigIns: &I);
4513 }
4514
4515 if (!PropagateShadow) {
4516 setShadow(V: &I, SV: getCleanShadow(V: &I));
4517 setOrigin(V: &I, Origin: getCleanOrigin());
4518 return;
4519 }
4520
4521 Type *ShadowTy = getShadowTy(V: &I);
4522 Type *ElementShadowTy = cast<VectorType>(Val: ShadowTy)->getElementType();
4523 auto [ShadowPtrs, OriginPtrs] = getShadowOriginPtr(
4524 Addr: Ptrs, IRB, ShadowTy: ElementShadowTy, Alignment, /*isStore*/ false);
4525
4526 Value *Shadow =
4527 IRB.CreateMaskedGather(Ty: ShadowTy, Ptrs: ShadowPtrs, Alignment, Mask,
4528 PassThru: getShadow(V: PassThru), Name: "_msmaskedgather");
4529
4530 setShadow(V: &I, SV: Shadow);
4531
4532 // TODO: Store origins.
4533 setOrigin(V: &I, Origin: getCleanOrigin());
4534 }
4535
4536 void handleMaskedScatter(IntrinsicInst &I) {
4537 IRBuilder<> IRB(&I);
4538 Value *Values = I.getArgOperand(i: 0);
4539 Value *Ptrs = I.getArgOperand(i: 1);
4540 const Align Alignment = I.getParamAlign(ArgNo: 1).valueOrOne();
4541 Value *Mask = I.getArgOperand(i: 2);
4542
4543 Type *PtrsShadowTy = getShadowTy(V: Ptrs);
4544 if (ClCheckAccessAddress) {
4545 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4546 Value *MaskedPtrShadow = IRB.CreateSelect(
4547 C: Mask, True: getShadow(V: Ptrs), False: Constant::getNullValue(Ty: (PtrsShadowTy)),
4548 Name: "_msmaskedptrs");
4549 insertCheckShadow(Shadow: MaskedPtrShadow, Origin: getOrigin(V: Ptrs), OrigIns: &I);
4550 }
4551
4552 Value *Shadow = getShadow(V: Values);
4553 Type *ElementShadowTy =
4554 getShadowTy(OrigTy: cast<VectorType>(Val: Values->getType())->getElementType());
4555 auto [ShadowPtrs, OriginPtrs] = getShadowOriginPtr(
4556 Addr: Ptrs, IRB, ShadowTy: ElementShadowTy, Alignment, /*isStore*/ true);
4557
4558 IRB.CreateMaskedScatter(Val: Shadow, Ptrs: ShadowPtrs, Alignment, Mask);
4559
4560 // TODO: Store origin.
4561 }
4562
4563 // Intrinsic::masked_store
4564 //
4565 // Note: handleAVXMaskedStore handles AVX/AVX2 variants, though AVX512 masked
4566 // stores are lowered to Intrinsic::masked_store.
4567 void handleMaskedStore(IntrinsicInst &I) {
4568 IRBuilder<> IRB(&I);
4569 Value *V = I.getArgOperand(i: 0);
4570 Value *Ptr = I.getArgOperand(i: 1);
4571 const Align Alignment = I.getParamAlign(ArgNo: 1).valueOrOne();
4572 Value *Mask = I.getArgOperand(i: 2);
4573 Value *Shadow = getShadow(V);
4574
4575 if (ClCheckAccessAddress) {
4576 insertCheckShadowOf(Val: Ptr, OrigIns: &I);
4577 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4578 }
4579
4580 Value *ShadowPtr;
4581 Value *OriginPtr;
4582 std::tie(args&: ShadowPtr, args&: OriginPtr) = getShadowOriginPtr(
4583 Addr: Ptr, IRB, ShadowTy: Shadow->getType(), Alignment, /*isStore*/ true);
4584
4585 IRB.CreateMaskedStore(Val: Shadow, Ptr: ShadowPtr, Alignment, Mask);
4586
4587 if (!MS.TrackOrigins)
4588 return;
4589
4590 auto &DL = F.getDataLayout();
4591 paintOrigin(IRB, Origin: getOrigin(V), OriginPtr,
4592 TS: DL.getTypeStoreSize(Ty: Shadow->getType()),
4593 Alignment: std::max(a: Alignment, b: kMinOriginAlignment));
4594 }
4595
4596 // Intrinsic::masked_load
4597 //
4598 // Note: handleAVXMaskedLoad handles AVX/AVX2 variants, though AVX512 masked
4599 // loads are lowered to Intrinsic::masked_load.
4600 void handleMaskedLoad(IntrinsicInst &I) {
4601 IRBuilder<> IRB(&I);
4602 Value *Ptr = I.getArgOperand(i: 0);
4603 const Align Alignment = I.getParamAlign(ArgNo: 0).valueOrOne();
4604 Value *Mask = I.getArgOperand(i: 1);
4605 Value *PassThru = I.getArgOperand(i: 2);
4606
4607 if (ClCheckAccessAddress) {
4608 insertCheckShadowOf(Val: Ptr, OrigIns: &I);
4609 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4610 }
4611
4612 if (!PropagateShadow) {
4613 setShadow(V: &I, SV: getCleanShadow(V: &I));
4614 setOrigin(V: &I, Origin: getCleanOrigin());
4615 return;
4616 }
4617
4618 Type *ShadowTy = getShadowTy(V: &I);
4619 Value *ShadowPtr, *OriginPtr;
4620 std::tie(args&: ShadowPtr, args&: OriginPtr) =
4621 getShadowOriginPtr(Addr: Ptr, IRB, ShadowTy, Alignment, /*isStore*/ false);
4622 setShadow(V: &I, SV: IRB.CreateMaskedLoad(Ty: ShadowTy, Ptr: ShadowPtr, Alignment, Mask,
4623 PassThru: getShadow(V: PassThru), Name: "_msmaskedld"));
4624
4625 if (!MS.TrackOrigins)
4626 return;
4627
4628 // Choose between PassThru's and the loaded value's origins.
4629 Value *MaskedPassThruShadow = IRB.CreateAnd(
4630 LHS: getShadow(V: PassThru), RHS: IRB.CreateSExt(V: IRB.CreateNeg(V: Mask), DestTy: ShadowTy));
4631
4632 Value *NotNull = convertToBool(V: MaskedPassThruShadow, IRB, name: "_mscmp");
4633
4634 Value *PtrOrigin = IRB.CreateLoad(Ty: MS.OriginTy, Ptr: OriginPtr);
4635 Value *Origin = IRB.CreateSelect(C: NotNull, True: getOrigin(V: PassThru), False: PtrOrigin);
4636
4637 setOrigin(V: &I, Origin);
4638 }
4639
4640 // e.g., <4 x i32> @llvm.masked.udiv.v4i32(<4 x i32> %dividend,
4641 // <4 x i32> %divisor,
4642 // <4 x i1> %mask)
4643 //
4644 // As handleIntegerDiv(), but per-lane: strict on the divisor and propagating
4645 // the dividend, both only on the enabled lanes. Disabled lanes cannot cause
4646 // undefined behaviour, and their result is poison.
4647 void handleMaskedIntegerDivRem(IntrinsicInst &I) {
4648 assert(I.arg_size() == 3);
4649 IRBuilder<> IRB(&I);
4650 Value *Dividend = I.getArgOperand(i: 0);
4651 Value *Divisor = I.getArgOperand(i: 1);
4652 Value *Mask = I.getArgOperand(i: 2);
4653
4654 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4655
4656 Value *MaskedDivisorShadow = IRB.CreateSelect(
4657 C: Mask, True: getShadow(V: Divisor), False: getCleanShadow(V: Divisor), Name: "_msmaskeddivisor");
4658 insertCheckShadow(Shadow: MaskedDivisorShadow, Origin: getOrigin(V: Divisor), OrigIns: &I);
4659
4660 if (!PropagateShadow) {
4661 setShadow(V: &I, SV: getCleanShadow(V: &I));
4662 setOrigin(V: &I, Origin: getCleanOrigin());
4663 return;
4664 }
4665
4666 setShadow(V: &I, SV: IRB.CreateSelect(C: Mask, True: getShadow(V: Dividend),
4667 False: getPoisonedShadow(V: &I), Name: "_msmaskeddiv"));
4668 setOrigin(V: &I, Origin: getOrigin(V: Dividend));
4669 }
4670
4671 // e.g., void @llvm.x86.avx.maskstore.ps.256(ptr, <8 x i32>, <8 x float>)
4672 // dst mask src
4673 //
4674 // AVX512 masked stores are lowered to Intrinsic::masked_load and are handled
4675 // by handleMaskedStore.
4676 //
4677 // This function handles AVX and AVX2 masked stores; these use the MSBs of a
4678 // vector of integers, unlike the LLVM masked intrinsics, which require a
4679 // vector of booleans. X86InstCombineIntrinsic.cpp::simplifyX86MaskedLoad
4680 // mentions that the x86 backend does not know how to efficiently convert
4681 // from a vector of booleans back into the AVX mask format; therefore, they
4682 // (and we) do not reduce AVX/AVX2 masked intrinsics into LLVM masked
4683 // intrinsics.
4684 void handleAVXMaskedStore(IntrinsicInst &I) {
4685 assert(I.arg_size() == 3);
4686
4687 IRBuilder<> IRB(&I);
4688
4689 Value *Dst = I.getArgOperand(i: 0);
4690 assert(Dst->getType()->isPointerTy() && "Destination is not a pointer!");
4691
4692 Value *Mask = I.getArgOperand(i: 1);
4693 assert(isa<VectorType>(Mask->getType()) && "Mask is not a vector!");
4694
4695 Value *Src = I.getArgOperand(i: 2);
4696 assert(isa<VectorType>(Src->getType()) && "Source is not a vector!");
4697
4698 const Align Alignment = Align(1);
4699
4700 Value *SrcShadow = getShadow(V: Src);
4701
4702 if (ClCheckAccessAddress) {
4703 insertCheckShadowOf(Val: Dst, OrigIns: &I);
4704 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4705 }
4706
4707 Value *DstShadowPtr;
4708 Value *DstOriginPtr;
4709 std::tie(args&: DstShadowPtr, args&: DstOriginPtr) = getShadowOriginPtr(
4710 Addr: Dst, IRB, ShadowTy: SrcShadow->getType(), Alignment, /*isStore*/ true);
4711
4712 SmallVector<Value *, 2> ShadowArgs;
4713 ShadowArgs.append(NumInputs: 1, Elt: DstShadowPtr);
4714 ShadowArgs.append(NumInputs: 1, Elt: Mask);
4715 // The intrinsic may require floating-point but shadows can be arbitrary
4716 // bit patterns, of which some would be interpreted as "invalid"
4717 // floating-point values (NaN etc.); we assume the intrinsic will happily
4718 // copy them.
4719 ShadowArgs.append(NumInputs: 1, Elt: IRB.CreateBitCast(V: SrcShadow, DestTy: Src->getType()));
4720
4721 CallInst *CI = IRB.CreateIntrinsicWithoutFolding(
4722 RetTy: IRB.getVoidTy(), ID: I.getIntrinsicID(), Args: ShadowArgs);
4723 setShadow(V: &I, SV: CI);
4724
4725 if (!MS.TrackOrigins)
4726 return;
4727
4728 // Approximation only
4729 auto &DL = F.getDataLayout();
4730 paintOrigin(IRB, Origin: getOrigin(V: Src), OriginPtr: DstOriginPtr,
4731 TS: DL.getTypeStoreSize(Ty: SrcShadow->getType()),
4732 Alignment: std::max(a: Alignment, b: kMinOriginAlignment));
4733 }
4734
4735 // e.g., <8 x float> @llvm.x86.avx.maskload.ps.256(ptr, <8 x i32>)
4736 // return src mask
4737 //
4738 // Masked-off values are replaced with 0, which conveniently also represents
4739 // initialized memory.
4740 //
4741 // AVX512 masked stores are lowered to Intrinsic::masked_load and are handled
4742 // by handleMaskedStore.
4743 //
4744 // We do not combine this with handleMaskedLoad; see comment in
4745 // handleAVXMaskedStore for the rationale.
4746 //
4747 // This is subtly different than handleIntrinsicByApplyingToShadow(I, 1)
4748 // because we need to apply getShadowOriginPtr, not getShadow, to the first
4749 // parameter.
4750 void handleAVXMaskedLoad(IntrinsicInst &I) {
4751 assert(I.arg_size() == 2);
4752
4753 IRBuilder<> IRB(&I);
4754
4755 Value *Src = I.getArgOperand(i: 0);
4756 assert(Src->getType()->isPointerTy() && "Source is not a pointer!");
4757
4758 Value *Mask = I.getArgOperand(i: 1);
4759 assert(isa<VectorType>(Mask->getType()) && "Mask is not a vector!");
4760
4761 const Align Alignment = Align(1);
4762
4763 if (ClCheckAccessAddress) {
4764 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4765 }
4766
4767 Type *SrcShadowTy = getShadowTy(V: Src);
4768 Value *SrcShadowPtr, *SrcOriginPtr;
4769 std::tie(args&: SrcShadowPtr, args&: SrcOriginPtr) =
4770 getShadowOriginPtr(Addr: Src, IRB, ShadowTy: SrcShadowTy, Alignment, /*isStore*/ false);
4771
4772 SmallVector<Value *, 2> ShadowArgs;
4773 ShadowArgs.append(NumInputs: 1, Elt: SrcShadowPtr);
4774 ShadowArgs.append(NumInputs: 1, Elt: Mask);
4775
4776 CallInst *CI = IRB.CreateIntrinsicWithoutFolding(
4777 RetTy: I.getType(), ID: I.getIntrinsicID(), Args: ShadowArgs);
4778 // The AVX masked load intrinsics do not have integer variants. We use the
4779 // floating-point variants, which will happily copy the shadows even if
4780 // they are interpreted as "invalid" floating-point values (NaN etc.).
4781 setShadow(V: &I, SV: IRB.CreateBitCast(V: CI, DestTy: getShadowTy(V: &I)));
4782
4783 if (!MS.TrackOrigins)
4784 return;
4785
4786 // The "pass-through" value is always zero (initialized). To the extent
4787 // that that results in initialized aligned 4-byte chunks, the origin value
4788 // is ignored. It is therefore correct to simply copy the origin from src.
4789 Value *PtrSrcOrigin = IRB.CreateLoad(Ty: MS.OriginTy, Ptr: SrcOriginPtr);
4790 setOrigin(V: &I, Origin: PtrSrcOrigin);
4791 }
4792
4793 // Test whether the mask indices are initialized, only checking the bits that
4794 // are actually used.
4795 //
4796 // e.g., if Idx is <32 x i16>, only (log2(32) == 5) bits of each index are
4797 // used/checked.
4798 void maskedCheckAVXIndexShadow(IRBuilder<> &IRB, Value *Idx, Instruction *I) {
4799 assert(isFixedIntVector(Idx));
4800 auto IdxVectorSize =
4801 cast<FixedVectorType>(Val: Idx->getType())->getNumElements();
4802 assert(isPowerOf2_64(IdxVectorSize));
4803
4804 // Compiler isn't smart enough, let's help it
4805 if (isa<Constant>(Val: Idx))
4806 return;
4807
4808 auto *IdxShadow = getShadow(V: Idx);
4809 Value *Truncated = IRB.CreateTrunc(
4810 V: IdxShadow,
4811 DestTy: FixedVectorType::get(ElementType: Type::getIntNTy(C&: *MS.C, N: Log2_64(Value: IdxVectorSize)),
4812 NumElts: IdxVectorSize));
4813 insertCheckShadow(Shadow: Truncated, Origin: getOrigin(V: Idx), OrigIns: I);
4814 }
4815
4816 // Instrument AVX permutation intrinsic.
4817 // We apply the same permutation (argument index 1) to the shadow.
4818 void handleAVXVpermilvar(IntrinsicInst &I) {
4819 IRBuilder<> IRB(&I);
4820 Value *Shadow = getShadow(I: &I, i: 0);
4821 maskedCheckAVXIndexShadow(IRB, Idx: I.getArgOperand(i: 1), I: &I);
4822
4823 // Shadows are integer-ish types but some intrinsics require a
4824 // different (e.g., floating-point) type.
4825 Shadow = IRB.CreateBitCast(V: Shadow, DestTy: I.getArgOperand(i: 0)->getType());
4826 CallInst *CI = IRB.CreateIntrinsicWithoutFolding(
4827 RetTy: I.getType(), ID: I.getIntrinsicID(), Args: {Shadow, I.getArgOperand(i: 1)});
4828
4829 setShadow(V: &I, SV: IRB.CreateBitCast(V: CI, DestTy: getShadowTy(V: &I)));
4830 setOriginForNaryOp(I);
4831 }
4832
4833 // Instrument AVX permutation intrinsic.
4834 // We apply the same permutation (argument index 1) to the shadows.
4835 void handleAVXVpermi2var(IntrinsicInst &I) {
4836 assert(I.arg_size() == 3);
4837 assert(isa<FixedVectorType>(I.getArgOperand(0)->getType()));
4838 assert(isa<FixedVectorType>(I.getArgOperand(1)->getType()));
4839 assert(isa<FixedVectorType>(I.getArgOperand(2)->getType()));
4840 [[maybe_unused]] auto ArgVectorSize =
4841 cast<FixedVectorType>(Val: I.getArgOperand(i: 0)->getType())->getNumElements();
4842 assert(cast<FixedVectorType>(I.getArgOperand(1)->getType())
4843 ->getNumElements() == ArgVectorSize);
4844 assert(cast<FixedVectorType>(I.getArgOperand(2)->getType())
4845 ->getNumElements() == ArgVectorSize);
4846 assert(I.getArgOperand(0)->getType() == I.getArgOperand(2)->getType());
4847 assert(I.getType() == I.getArgOperand(0)->getType());
4848 assert(I.getArgOperand(1)->getType()->isIntOrIntVectorTy());
4849 IRBuilder<> IRB(&I);
4850 Value *AShadow = getShadow(I: &I, i: 0);
4851 Value *Idx = I.getArgOperand(i: 1);
4852 Value *BShadow = getShadow(I: &I, i: 2);
4853
4854 maskedCheckAVXIndexShadow(IRB, Idx, I: &I);
4855
4856 // Shadows are integer-ish types but some intrinsics require a
4857 // different (e.g., floating-point) type.
4858 AShadow = IRB.CreateBitCast(V: AShadow, DestTy: I.getArgOperand(i: 0)->getType());
4859 BShadow = IRB.CreateBitCast(V: BShadow, DestTy: I.getArgOperand(i: 2)->getType());
4860 CallInst *CI = IRB.CreateIntrinsicWithoutFolding(
4861 RetTy: I.getType(), ID: I.getIntrinsicID(), Args: {AShadow, Idx, BShadow});
4862 setShadow(V: &I, SV: IRB.CreateBitCast(V: CI, DestTy: getShadowTy(V: &I)));
4863 setOriginForNaryOp(I);
4864 }
4865
4866 [[maybe_unused]] static bool isFixedIntVectorTy(const Type *T) {
4867 return isa<FixedVectorType>(Val: T) && T->isIntOrIntVectorTy();
4868 }
4869
4870 [[maybe_unused]] static bool isFixedFPVectorTy(const Type *T) {
4871 return isa<FixedVectorType>(Val: T) && T->isFPOrFPVectorTy();
4872 }
4873
4874 [[maybe_unused]] static bool isFixedIntVector(const Value *V) {
4875 return isFixedIntVectorTy(T: V->getType());
4876 }
4877
4878 [[maybe_unused]] static bool isFixedFPVector(const Value *V) {
4879 return isFixedFPVectorTy(T: V->getType());
4880 }
4881
4882 // e.g., <16 x i32> @llvm.x86.avx512.mask.cvtps2dq.512
4883 // (<16 x float> a, <16 x i32> writethru, i16 mask,
4884 // i32 rounding)
4885 //
4886 // Inconveniently, some similar intrinsics have a different operand order:
4887 // <16 x i16> @llvm.x86.avx512.mask.vcvtps2ph.512
4888 // (<16 x float> a, i32 rounding, <16 x i16> writethru,
4889 // i16 mask)
4890 //
4891 // If the return type has more elements than A, the excess elements are
4892 // zeroed (and the corresponding shadow is initialized).
4893 // <8 x i16> @llvm.x86.avx512.mask.vcvtps2ph.128
4894 // (<4 x float> a, i32 rounding, <8 x i16> writethru,
4895 // i8 mask)
4896 //
4897 // dst[i] = mask[i] ? convert(a[i]) : writethru[i]
4898 // dst_shadow[i] = mask[i] ? all_or_nothing(a_shadow[i]) : writethru_shadow[i]
4899 // where all_or_nothing(x) is fully uninitialized if x has any
4900 // uninitialized bits
4901 void handleAVX512VectorConvertFPToInt(IntrinsicInst &I, bool LastMask) {
4902 IRBuilder<> IRB(&I);
4903
4904 assert(I.arg_size() == 4);
4905 Value *A = I.getOperand(i_nocapture: 0);
4906 Value *WriteThrough;
4907 Value *Mask;
4908 Value *RoundingMode;
4909 if (LastMask) {
4910 WriteThrough = I.getOperand(i_nocapture: 2);
4911 Mask = I.getOperand(i_nocapture: 3);
4912 RoundingMode = I.getOperand(i_nocapture: 1);
4913 } else {
4914 WriteThrough = I.getOperand(i_nocapture: 1);
4915 Mask = I.getOperand(i_nocapture: 2);
4916 RoundingMode = I.getOperand(i_nocapture: 3);
4917 }
4918
4919 assert(isFixedFPVector(A));
4920 assert(isFixedIntVector(WriteThrough));
4921
4922 unsigned ANumElements =
4923 cast<FixedVectorType>(Val: A->getType())->getNumElements();
4924 [[maybe_unused]] unsigned WriteThruNumElements =
4925 cast<FixedVectorType>(Val: WriteThrough->getType())->getNumElements();
4926 assert(ANumElements == WriteThruNumElements ||
4927 ANumElements * 2 == WriteThruNumElements);
4928
4929 assert(Mask->getType()->isIntegerTy());
4930 unsigned MaskNumElements = Mask->getType()->getScalarSizeInBits();
4931 assert(ANumElements == MaskNumElements ||
4932 ANumElements * 2 == MaskNumElements);
4933
4934 assert(WriteThruNumElements == MaskNumElements);
4935
4936 // Some bits of the mask may be unused, though it's unusual to have partly
4937 // uninitialized bits.
4938 insertCheckShadowOf(Val: Mask, OrigIns: &I);
4939
4940 assert(RoundingMode->getType()->isIntegerTy());
4941 // Only some bits of the rounding mode are used, though it's very
4942 // unusual to have uninitialized bits there (more commonly, it's a
4943 // constant).
4944 insertCheckShadowOf(Val: RoundingMode, OrigIns: &I);
4945
4946 assert(I.getType() == WriteThrough->getType());
4947
4948 Value *AShadow = getShadow(V: A);
4949 AShadow = maybeExtendVectorShadowWithZeros(Shadow: AShadow, I);
4950
4951 if (ANumElements * 2 == MaskNumElements) {
4952 // Ensure that the irrelevant bits of the mask are zero, hence selecting
4953 // from the zeroed shadow instead of the writethrough's shadow.
4954 Mask =
4955 IRB.CreateTrunc(V: Mask, DestTy: IRB.getIntNTy(N: ANumElements), Name: "_ms_mask_trunc");
4956 Mask =
4957 IRB.CreateZExt(V: Mask, DestTy: IRB.getIntNTy(N: MaskNumElements), Name: "_ms_mask_zext");
4958 }
4959
4960 // Convert i16 mask to <16 x i1>
4961 Mask = IRB.CreateBitCast(
4962 V: Mask, DestTy: FixedVectorType::get(ElementType: IRB.getInt1Ty(), NumElts: MaskNumElements),
4963 Name: "_ms_mask_bitcast");
4964
4965 /// For floating-point to integer conversion, the output is:
4966 /// - fully uninitialized if *any* bit of the input is uninitialized
4967 /// - fully ininitialized if all bits of the input are ininitialized
4968 /// We apply the same principle on a per-element basis for vectors.
4969 ///
4970 /// We use the scalar width of the return type instead of A's.
4971 AShadow = IRB.CreateSExt(
4972 V: IRB.CreateICmpNE(LHS: AShadow, RHS: getCleanShadow(OrigTy: AShadow->getType())),
4973 DestTy: getShadowTy(V: &I), Name: "_ms_a_shadow");
4974
4975 Value *WriteThroughShadow = getShadow(V: WriteThrough);
4976 Value *Shadow = IRB.CreateSelect(C: Mask, True: AShadow, False: WriteThroughShadow,
4977 Name: "_ms_writethru_select");
4978
4979 setShadow(V: &I, SV: Shadow);
4980 setOriginForNaryOp(I);
4981 }
4982
4983 static SmallVector<int, 8> getPclmulMask(unsigned Width, bool OddElements) {
4984 SmallVector<int, 8> Mask;
4985 for (unsigned X = OddElements ? 1 : 0; X < Width; X += 2) {
4986 Mask.append(NumInputs: 2, Elt: X);
4987 }
4988 return Mask;
4989 }
4990
4991 // Instrument pclmul intrinsics.
4992 // These intrinsics operate either on odd or on even elements of the input
4993 // vectors, depending on the constant in the 3rd argument, ignoring the rest.
4994 // Replace the unused elements with copies of the used ones, ex:
4995 // (0, 1, 2, 3) -> (0, 0, 2, 2) (even case)
4996 // or
4997 // (0, 1, 2, 3) -> (1, 1, 3, 3) (odd case)
4998 // and then apply the usual shadow combining logic.
4999 void handlePclmulIntrinsic(IntrinsicInst &I) {
5000 IRBuilder<> IRB(&I);
5001 unsigned Width =
5002 cast<FixedVectorType>(Val: I.getArgOperand(i: 0)->getType())->getNumElements();
5003 assert(isa<ConstantInt>(I.getArgOperand(2)) &&
5004 "pclmul 3rd operand must be a constant");
5005 unsigned Imm = cast<ConstantInt>(Val: I.getArgOperand(i: 2))->getZExtValue();
5006 Value *Shuf0 = IRB.CreateShuffleVector(V: getShadow(I: &I, i: 0),
5007 Mask: getPclmulMask(Width, OddElements: Imm & 0x01));
5008 Value *Shuf1 = IRB.CreateShuffleVector(V: getShadow(I: &I, i: 1),
5009 Mask: getPclmulMask(Width, OddElements: Imm & 0x10));
5010 ShadowAndOriginCombiner SOC(this, IRB);
5011 SOC.Add(OpShadow: Shuf0, OpOrigin: getOrigin(I: &I, i: 0));
5012 SOC.Add(OpShadow: Shuf1, OpOrigin: getOrigin(I: &I, i: 1));
5013 SOC.Done(I: &I);
5014 }
5015
5016 // Instrument _mm_*_sd|ss intrinsics
5017 void handleUnarySdSsIntrinsic(IntrinsicInst &I) {
5018 IRBuilder<> IRB(&I);
5019 unsigned Width =
5020 cast<FixedVectorType>(Val: I.getArgOperand(i: 0)->getType())->getNumElements();
5021 Value *First = getShadow(I: &I, i: 0);
5022 Value *Second = getShadow(I: &I, i: 1);
5023 // First element of second operand, remaining elements of first operand
5024 SmallVector<int, 16> Mask;
5025 Mask.push_back(Elt: Width);
5026 for (unsigned i = 1; i < Width; i++)
5027 Mask.push_back(Elt: i);
5028 Value *Shadow = IRB.CreateShuffleVector(V1: First, V2: Second, Mask);
5029
5030 setShadow(V: &I, SV: Shadow);
5031 setOriginForNaryOp(I);
5032 }
5033
5034 void handleVtestIntrinsic(IntrinsicInst &I) {
5035 IRBuilder<> IRB(&I);
5036 Value *Shadow0 = getShadow(I: &I, i: 0);
5037 Value *Shadow1 = getShadow(I: &I, i: 1);
5038 Value *Or = IRB.CreateOr(LHS: Shadow0, RHS: Shadow1);
5039 Value *NZ = IRB.CreateICmpNE(LHS: Or, RHS: Constant::getNullValue(Ty: Or->getType()));
5040 Value *Scalar = convertShadowToScalar(V: NZ, IRB);
5041 Value *Shadow = IRB.CreateZExt(V: Scalar, DestTy: getShadowTy(V: &I));
5042
5043 setShadow(V: &I, SV: Shadow);
5044 setOriginForNaryOp(I);
5045 }
5046
5047 void handleBinarySdSsIntrinsic(IntrinsicInst &I) {
5048 IRBuilder<> IRB(&I);
5049 unsigned Width =
5050 cast<FixedVectorType>(Val: I.getArgOperand(i: 0)->getType())->getNumElements();
5051 Value *First = getShadow(I: &I, i: 0);
5052 Value *Second = getShadow(I: &I, i: 1);
5053 Value *OrShadow = IRB.CreateOr(LHS: First, RHS: Second);
5054 // First element of both OR'd together, remaining elements of first operand
5055 SmallVector<int, 16> Mask;
5056 Mask.push_back(Elt: Width);
5057 for (unsigned i = 1; i < Width; i++)
5058 Mask.push_back(Elt: i);
5059 Value *Shadow = IRB.CreateShuffleVector(V1: First, V2: OrShadow, Mask);
5060
5061 setShadow(V: &I, SV: Shadow);
5062 setOriginForNaryOp(I);
5063 }
5064
5065 // _mm_round_ps / _mm_round_ps.
5066 // Similar to maybeHandleSimpleNomemIntrinsic except
5067 // the second argument is guaranteed to be a constant integer.
5068 void handleRoundPdPsIntrinsic(IntrinsicInst &I) {
5069 assert(I.getArgOperand(0)->getType() == I.getType());
5070 assert(I.arg_size() == 2);
5071 assert(isa<ConstantInt>(I.getArgOperand(1)));
5072
5073 IRBuilder<> IRB(&I);
5074 ShadowAndOriginCombiner SC(this, IRB);
5075 SC.Add(V: I.getArgOperand(i: 0));
5076 SC.Done(I: &I);
5077 }
5078
5079 // Instrument @llvm.abs intrinsic.
5080 //
5081 // e.g., i32 @llvm.abs.i32 (i32 <Src>, i1 <is_int_min_poison>)
5082 // <4 x i32> @llvm.abs.v4i32(<4 x i32> <Src>, i1 <is_int_min_poison>)
5083 void handleAbsIntrinsic(IntrinsicInst &I) {
5084 assert(I.arg_size() == 2);
5085 Value *Src = I.getArgOperand(i: 0);
5086 Value *IsIntMinPoison = I.getArgOperand(i: 1);
5087
5088 assert(I.getType()->isIntOrIntVectorTy());
5089
5090 assert(Src->getType() == I.getType());
5091
5092 assert(IsIntMinPoison->getType()->isIntegerTy());
5093 assert(IsIntMinPoison->getType()->getIntegerBitWidth() == 1);
5094
5095 IRBuilder<> IRB(&I);
5096 Value *SrcShadow = getShadow(V: Src);
5097
5098 APInt MinVal =
5099 APInt::getSignedMinValue(numBits: Src->getType()->getScalarSizeInBits());
5100 Value *MinValVec = ConstantInt::get(Ty: Src->getType(), V: MinVal);
5101 Value *SrcIsMin = IRB.CreateICmp(P: CmpInst::ICMP_EQ, LHS: Src, RHS: MinValVec);
5102
5103 Value *PoisonedShadow = getPoisonedShadow(V: Src);
5104 Value *PoisonedIfIntMinShadow =
5105 IRB.CreateSelect(C: SrcIsMin, True: PoisonedShadow, False: SrcShadow);
5106 Value *Shadow =
5107 IRB.CreateSelect(C: IsIntMinPoison, True: PoisonedIfIntMinShadow, False: SrcShadow);
5108
5109 setShadow(V: &I, SV: Shadow);
5110 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
5111 }
5112
5113 void handleIsFpClass(IntrinsicInst &I) {
5114 IRBuilder<> IRB(&I);
5115 Value *Shadow = getShadow(I: &I, i: 0);
5116 setShadow(V: &I, SV: IRB.CreateICmpNE(LHS: Shadow, RHS: getCleanShadow(V: Shadow)));
5117 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
5118 }
5119
5120 void handleArithmeticWithOverflow(IntrinsicInst &I) {
5121 IRBuilder<> IRB(&I);
5122 Value *Shadow0 = getShadow(I: &I, i: 0);
5123 Value *Shadow1 = getShadow(I: &I, i: 1);
5124 Value *ShadowElt0 = IRB.CreateOr(LHS: Shadow0, RHS: Shadow1);
5125 Value *ShadowElt1 =
5126 IRB.CreateICmpNE(LHS: ShadowElt0, RHS: getCleanShadow(V: ShadowElt0));
5127
5128 Value *Shadow = PoisonValue::get(T: getShadowTy(V: &I));
5129 Shadow = IRB.CreateInsertValue(Agg: Shadow, Val: ShadowElt0, Idxs: 0);
5130 Shadow = IRB.CreateInsertValue(Agg: Shadow, Val: ShadowElt1, Idxs: 1);
5131
5132 setShadow(V: &I, SV: Shadow);
5133 setOriginForNaryOp(I);
5134 }
5135
5136 void handleModfOrSincos(IntrinsicInst &I) {
5137 IRBuilder<> IRB(&I);
5138 Value *ArgShadow = getShadow(I: &I, i: 0);
5139 Value *Shadow = PoisonValue::get(T: getShadowTy(V: &I));
5140 Shadow = IRB.CreateInsertValue(Agg: Shadow, Val: ArgShadow, Idxs: 0);
5141 Shadow = IRB.CreateInsertValue(Agg: Shadow, Val: ArgShadow, Idxs: 1);
5142 setShadow(V: &I, SV: Shadow);
5143 setOrigin(V: &I, Origin: getOrigin(I: &I, i: 0));
5144 }
5145
5146 Value *extractLowerShadow(IRBuilder<> &IRB, Value *V) {
5147 assert(isa<FixedVectorType>(V->getType()));
5148 assert(cast<FixedVectorType>(V->getType())->getNumElements() > 0);
5149 Value *Shadow = getShadow(V);
5150 return IRB.CreateExtractElement(Vec: Shadow,
5151 Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: 0));
5152 }
5153
5154 // Handle llvm.x86.avx512.mask.pmov{,s,us}.*.{128,256,512}
5155 //
5156 // e.g., call <16 x i8> @llvm.x86.avx512.mask.pmov.qb.512
5157 // (<8 x i64>, <16 x i8>, i8)
5158 // A WriteThru Mask
5159 //
5160 // call <16 x i8> @llvm.x86.avx512.mask.pmovs.db.512
5161 // (<16 x i32>, <16 x i8>, i16)
5162 //
5163 // Dst[i] = Mask[i] ? truncate_or_saturate(A[i]) : WriteThru[i]
5164 // Dst_shadow[i] = Mask[i] ? truncate(A_shadow[i]) : WriteThru_shadow[i]
5165 //
5166 // If Dst has more elements than A, the excess elements are zeroed (and the
5167 // corresponding shadow is initialized).
5168 //
5169 // Note: for PMOV (truncation), handleIntrinsicByApplyingToShadow is precise
5170 // and is much faster than this handler.
5171 void handleAVX512VectorDownConvert(IntrinsicInst &I) {
5172 IRBuilder<> IRB(&I);
5173
5174 assert(I.arg_size() == 3);
5175 Value *A = I.getOperand(i_nocapture: 0);
5176 Value *WriteThrough = I.getOperand(i_nocapture: 1);
5177 Value *Mask = I.getOperand(i_nocapture: 2);
5178
5179 assert(isFixedIntVector(A));
5180 assert(isFixedIntVector(WriteThrough));
5181
5182 unsigned ANumElements =
5183 cast<FixedVectorType>(Val: A->getType())->getNumElements();
5184 unsigned OutputNumElements =
5185 cast<FixedVectorType>(Val: WriteThrough->getType())->getNumElements();
5186 assert(ANumElements == OutputNumElements ||
5187 ANumElements * 2 == OutputNumElements);
5188 // N.B. some PMOV{,S,US} instructions have a 4x or even 8x ratio in the
5189 // number of elements e.g.,
5190 // <16 x i8> @llvm.x86.avx512.mask.pmovs.qb.256
5191 // (<4 x i64>, <16 x i8>, i8)
5192 // <16 x i8> @llvm.x86.avx512.mask.pmovs.qb.128
5193 // (<2 x i64>, <16 x i8>, i8)
5194 // However, we currently handle those elsewhere.
5195
5196 assert(Mask->getType()->isIntegerTy());
5197 insertCheckShadowOf(Val: Mask, OrigIns: &I);
5198
5199 // The mask has 1 bit per element of A, but a minimum of 8 bits.
5200 if (Mask->getType()->getScalarSizeInBits() == 8 && OutputNumElements < 8)
5201 Mask = IRB.CreateTrunc(V: Mask, DestTy: Type::getIntNTy(C&: *MS.C, N: OutputNumElements));
5202 assert(Mask->getType()->getScalarSizeInBits() == ANumElements);
5203
5204 assert(I.getType() == WriteThrough->getType());
5205
5206 // Widen the mask, if necessary, to have one bit per element of the output
5207 // vector.
5208 // We want the extra bits to have '1's, so that the CreateSelect will
5209 // select the values from AShadow instead of WriteThroughShadow ("maskless"
5210 // versions of the intrinsics are sometimes implemented using an all-1's
5211 // mask and an undefined value for WriteThroughShadow). We accomplish this
5212 // by using bitwise NOT before and after the ZExt.
5213 if (ANumElements != OutputNumElements) {
5214 Mask = IRB.CreateNot(V: Mask);
5215 Mask = IRB.CreateZExt(V: Mask, DestTy: Type::getIntNTy(C&: *MS.C, N: OutputNumElements),
5216 Name: "_ms_widen_mask");
5217 Mask = IRB.CreateNot(V: Mask);
5218 }
5219 Mask = IRB.CreateBitCast(
5220 V: Mask, DestTy: FixedVectorType::get(ElementType: IRB.getInt1Ty(), NumElts: OutputNumElements));
5221
5222 Value *AShadow = getShadow(V: A);
5223
5224 // The return type might have more elements than the input.
5225 // Temporarily shrink the return type's number of elements.
5226 VectorType *ShadowType = maybeShrinkVectorShadowType(Src: A, I);
5227
5228 // PMOV truncates; PMOVS/PMOVUS uses signed/unsigned saturation.
5229 // This handler treats them all as truncation, which leads to some rare
5230 // false positives in the cases where the truncated bytes could
5231 // unambiguously saturate the value e.g., if A = ??????10 ????????
5232 // (big-endian), the unsigned saturated byte conversion is 11111111 i.e.,
5233 // fully defined, but the truncated byte is ????????.
5234 //
5235 // TODO: use GetMinMaxUnsigned() to handle saturation precisely.
5236 AShadow = IRB.CreateTrunc(V: AShadow, DestTy: ShadowType, Name: "_ms_trunc_shadow");
5237 AShadow = maybeExtendVectorShadowWithZeros(Shadow: AShadow, I);
5238
5239 Value *WriteThroughShadow = getShadow(V: WriteThrough);
5240
5241 Value *Shadow = IRB.CreateSelect(C: Mask, True: AShadow, False: WriteThroughShadow);
5242 setShadow(V: &I, SV: Shadow);
5243 setOriginForNaryOp(I);
5244 }
5245
5246 // Handle llvm.x86.avx512.* instructions that take vector(s) of floating-point
5247 // values and perform an operation whose shadow propagation should be handled
5248 // as all-or-nothing [*], with masking provided by a vector and a mask
5249 // supplied as an integer.
5250 //
5251 // [*] if all bits of a vector element are initialized, the output is fully
5252 // initialized; otherwise, the output is fully uninitialized
5253 //
5254 // e.g., <16 x float> @llvm.x86.avx512.rsqrt14.ps.512
5255 // (<16 x float>, <16 x float>, i16)
5256 // A WriteThru Mask
5257 //
5258 // <2 x double> @llvm.x86.avx512.rcp14.pd.128
5259 // (<2 x double>, <2 x double>, i8)
5260 // A WriteThru Mask
5261 //
5262 // <8 x double> @llvm.x86.avx512.mask.rndscale.pd.512
5263 // (<8 x double>, i32, <8 x double>, i8, i32)
5264 // A Imm WriteThru Mask Rounding
5265 //
5266 // <16 x float> @llvm.x86.avx512.mask.scalef.ps.512
5267 // (<16 x float>, <16 x float>, <16 x float>, i16, i32)
5268 // WriteThru A B Mask Rnd
5269 //
5270 // All operands other than A, B, ..., and WriteThru (e.g., Mask, Imm,
5271 // Rounding) must be fully initialized.
5272 //
5273 // Dst[i] = Mask[i] ? some_op(A[i], B[i], ...)
5274 // : WriteThru[i]
5275 // Dst_shadow[i] = Mask[i] ? all_or_nothing(A_shadow[i] | B_shadow[i] | ...)
5276 // : WriteThru_shadow[i]
5277 void handleAVX512VectorGenericMaskedFP(IntrinsicInst &I,
5278 SmallVector<unsigned, 4> DataIndices,
5279 unsigned WriteThruIndex,
5280 unsigned MaskIndex) {
5281 IRBuilder<> IRB(&I);
5282
5283 unsigned NumArgs = I.arg_size();
5284
5285 assert(WriteThruIndex < NumArgs);
5286 assert(MaskIndex < NumArgs);
5287 assert(WriteThruIndex != MaskIndex);
5288 Value *WriteThru = I.getOperand(i_nocapture: WriteThruIndex);
5289
5290 unsigned OutputNumElements =
5291 cast<FixedVectorType>(Val: WriteThru->getType())->getNumElements();
5292
5293 assert(DataIndices.size() > 0);
5294
5295 bool isData[16] = {false};
5296 assert(NumArgs <= 16);
5297 for (unsigned i : DataIndices) {
5298 assert(i < NumArgs);
5299 assert(i != WriteThruIndex);
5300 assert(i != MaskIndex);
5301
5302 isData[i] = true;
5303
5304 Value *A = I.getOperand(i_nocapture: i);
5305 assert(isFixedFPVector(A));
5306 [[maybe_unused]] unsigned ANumElements =
5307 cast<FixedVectorType>(Val: A->getType())->getNumElements();
5308 assert(ANumElements == OutputNumElements);
5309 }
5310
5311 Value *Mask = I.getOperand(i_nocapture: MaskIndex);
5312
5313 assert(isFixedFPVector(WriteThru));
5314
5315 for (unsigned i = 0; i < NumArgs; ++i) {
5316 if (!isData[i] && i != WriteThruIndex) {
5317 // Imm, Mask, Rounding etc. are "control" data, hence we require that
5318 // they be fully initialized.
5319 assert(I.getOperand(i)->getType()->isIntegerTy());
5320 insertCheckShadowOf(Val: I.getOperand(i_nocapture: i), OrigIns: &I);
5321 }
5322 }
5323
5324 // The mask has 1 bit per element of A, but a minimum of 8 bits.
5325 if (Mask->getType()->getScalarSizeInBits() == 8 && OutputNumElements < 8)
5326 Mask = IRB.CreateTrunc(V: Mask, DestTy: Type::getIntNTy(C&: *MS.C, N: OutputNumElements));
5327 assert(Mask->getType()->getScalarSizeInBits() == OutputNumElements);
5328
5329 assert(I.getType() == WriteThru->getType());
5330
5331 Mask = IRB.CreateBitCast(
5332 V: Mask, DestTy: FixedVectorType::get(ElementType: IRB.getInt1Ty(), NumElts: OutputNumElements));
5333
5334 Value *DataShadow = nullptr;
5335 for (unsigned i : DataIndices) {
5336 Value *A = I.getOperand(i_nocapture: i);
5337 if (DataShadow)
5338 DataShadow = IRB.CreateOr(LHS: DataShadow, RHS: getShadow(V: A));
5339 else
5340 DataShadow = getShadow(V: A);
5341 }
5342
5343 // All-or-nothing shadow
5344 DataShadow =
5345 IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: DataShadow, RHS: getCleanShadow(V: DataShadow)),
5346 DestTy: DataShadow->getType());
5347
5348 Value *WriteThruShadow = getShadow(V: WriteThru);
5349
5350 Value *Shadow = IRB.CreateSelect(C: Mask, True: DataShadow, False: WriteThruShadow);
5351 setShadow(V: &I, SV: Shadow);
5352
5353 setOriginForNaryOp(I);
5354 }
5355
5356 // AVX512 Floating-Point Classification
5357 //
5358 // e.g.,
5359 // - < 8 x i1> @llvm.x86.avx512.fpclass.pd.512
5360 // (<8 x double> %input, i32 %classifiers)
5361 // - <16 x i1> @llvm.x86.avx512.fpclass.ps.512
5362 // (<16 x float> %input, i32 %classifiers)
5363 void handleAVX512FPClass(IntrinsicInst &I) {
5364 IRBuilder<> IRB(&I);
5365
5366 assert(I.arg_size() == 2);
5367
5368 Value *Input = I.getOperand(i_nocapture: 0);
5369 assert(isFixedFPVector(Input));
5370 [[maybe_unused]] FixedVectorType *InputType = cast<FixedVectorType>(Val: Input->getType());
5371
5372 Value *Classifiers = I.getOperand(i_nocapture: 1);
5373 assert(isa<ConstantInt>(Classifiers));
5374 // No shadow check needed for constants
5375
5376 assert(isFixedIntVectorTy(I.getType()));
5377 FixedVectorType *OutputType = cast<FixedVectorType>(Val: I.getType());
5378 assert(OutputType->getScalarSizeInBits() == 1);
5379
5380 assert(OutputType->getNumElements() == InputType->getNumElements());
5381
5382 Value *OutputShadow;
5383 if (cast<ConstantInt>(Val: Classifiers)->isZero())
5384 // Each bit specifies whether a particular classifier is enabled.
5385 // If Classifiers == 0, the output is trivially known to be zero, thus
5386 // the output is fully initialized.
5387 OutputShadow = getCleanShadow(OrigTy: OutputType);
5388 else
5389 // Approximate each bit of the output shadow based on whether the
5390 // corresponding input element is fully initialized. It is only
5391 // approximate because some classifications do not rely on all bits of
5392 // the input element.
5393 OutputShadow = IRB.CreateICmpNE(LHS: getShadow(V: Input), RHS: getCleanShadow(V: Input));
5394
5395 setShadow(V: &I, SV: OutputShadow);
5396
5397 setOriginForNaryOp(I);
5398 }
5399
5400 // For sh.* compiler intrinsics:
5401 // llvm.x86.avx512fp16.mask.{add/sub/mul/div/max/min}.sh.round
5402 // (<8 x half>, <8 x half>, <8 x half>, i8, i32)
5403 // A B WriteThru Mask RoundingMode
5404 //
5405 // DstShadow[0] = Mask[0] ? (AShadow[0] | BShadow[0]) : WriteThruShadow[0]
5406 // DstShadow[1..7] = AShadow[1..7]
5407 void visitGenericScalarHalfwordInst(IntrinsicInst &I) {
5408 IRBuilder<> IRB(&I);
5409
5410 assert(I.arg_size() == 5);
5411 Value *A = I.getOperand(i_nocapture: 0);
5412 Value *B = I.getOperand(i_nocapture: 1);
5413 Value *WriteThrough = I.getOperand(i_nocapture: 2);
5414 Value *Mask = I.getOperand(i_nocapture: 3);
5415 Value *RoundingMode = I.getOperand(i_nocapture: 4);
5416
5417 // Technically, we could probably just check whether the LSB is
5418 // initialized, but intuitively it feels like a partly uninitialized mask
5419 // is unintended, and we should warn the user immediately.
5420 insertCheckShadowOf(Val: Mask, OrigIns: &I);
5421 insertCheckShadowOf(Val: RoundingMode, OrigIns: &I);
5422
5423 assert(isa<FixedVectorType>(A->getType()));
5424 unsigned NumElements =
5425 cast<FixedVectorType>(Val: A->getType())->getNumElements();
5426 assert(NumElements == 8);
5427 assert(A->getType() == B->getType());
5428 assert(B->getType() == WriteThrough->getType());
5429 assert(Mask->getType()->getPrimitiveSizeInBits() == NumElements);
5430 assert(RoundingMode->getType()->isIntegerTy());
5431
5432 Value *ALowerShadow = extractLowerShadow(IRB, V: A);
5433 Value *BLowerShadow = extractLowerShadow(IRB, V: B);
5434
5435 Value *ABLowerShadow = IRB.CreateOr(LHS: ALowerShadow, RHS: BLowerShadow);
5436
5437 Value *WriteThroughLowerShadow = extractLowerShadow(IRB, V: WriteThrough);
5438
5439 Mask = IRB.CreateBitCast(
5440 V: Mask, DestTy: FixedVectorType::get(ElementType: IRB.getInt1Ty(), NumElts: NumElements));
5441 Value *MaskLower =
5442 IRB.CreateExtractElement(Vec: Mask, Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: 0));
5443
5444 Value *AShadow = getShadow(V: A);
5445 Value *DstLowerShadow =
5446 IRB.CreateSelect(C: MaskLower, True: ABLowerShadow, False: WriteThroughLowerShadow);
5447 Value *DstShadow = IRB.CreateInsertElement(
5448 Vec: AShadow, NewElt: DstLowerShadow, Idx: ConstantInt::get(Ty: IRB.getInt32Ty(), V: 0),
5449 Name: "_msprop");
5450
5451 setShadow(V: &I, SV: DstShadow);
5452 setOriginForNaryOp(I);
5453 }
5454
5455 // Approximately handle AVX Galois Field Affine Transformation
5456 //
5457 // e.g.,
5458 // <16 x i8> @llvm.x86.vgf2p8affineqb.128(<16 x i8>, <16 x i8>, i8)
5459 // <32 x i8> @llvm.x86.vgf2p8affineqb.256(<32 x i8>, <32 x i8>, i8)
5460 // <64 x i8> @llvm.x86.vgf2p8affineqb.512(<64 x i8>, <64 x i8>, i8)
5461 // Out A x b
5462 // where A and x are packed matrices, b is a vector,
5463 // Out = A * x + b in GF(2)
5464 //
5465 // Multiplication in GF(2) is equivalent to bitwise AND. However, the matrix
5466 // computation also includes a parity calculation.
5467 //
5468 // For the bitwise AND of bits V1 and V2, the exact shadow is:
5469 // Out_Shadow = (V1_Shadow & V2_Shadow)
5470 // | (V1 & V2_Shadow)
5471 // | (V1_Shadow & V2 )
5472 //
5473 // We approximate the shadow of gf2p8affineqb using:
5474 // Out_Shadow = gf2p8affineqb(x_Shadow, A_shadow, 0)
5475 // | gf2p8affineqb(x, A_shadow, 0)
5476 // | gf2p8affineqb(x_Shadow, A, 0)
5477 // | set1_epi8(b_Shadow)
5478 //
5479 // This approximation has false negatives: if an intermediate dot-product
5480 // contains an even number of 1's, the parity is 0.
5481 // It has no false positives.
5482 void handleAVXGF2P8Affine(IntrinsicInst &I) {
5483 IRBuilder<> IRB(&I);
5484
5485 assert(I.arg_size() == 3);
5486 Value *A = I.getOperand(i_nocapture: 0);
5487 Value *X = I.getOperand(i_nocapture: 1);
5488 Value *B = I.getOperand(i_nocapture: 2);
5489
5490 assert(isFixedIntVector(A));
5491 assert(cast<VectorType>(A->getType())
5492 ->getElementType()
5493 ->getScalarSizeInBits() == 8);
5494
5495 assert(A->getType() == X->getType());
5496
5497 assert(B->getType()->isIntegerTy());
5498 assert(B->getType()->getScalarSizeInBits() == 8);
5499
5500 assert(I.getType() == A->getType());
5501
5502 Value *AShadow = getShadow(V: A);
5503 Value *XShadow = getShadow(V: X);
5504 Value *BZeroShadow = getCleanShadow(V: B);
5505
5506 Value *AShadowXShadow = IRB.CreateIntrinsic(
5507 RetTy: I.getType(), ID: I.getIntrinsicID(), Args: {XShadow, AShadow, BZeroShadow});
5508 Value *AShadowX = IRB.CreateIntrinsic(RetTy: I.getType(), ID: I.getIntrinsicID(),
5509 Args: {X, AShadow, BZeroShadow});
5510 Value *XShadowA = IRB.CreateIntrinsic(RetTy: I.getType(), ID: I.getIntrinsicID(),
5511 Args: {XShadow, A, BZeroShadow});
5512
5513 unsigned NumElements = cast<FixedVectorType>(Val: I.getType())->getNumElements();
5514 Value *BShadow = getShadow(V: B);
5515 Value *BBroadcastShadow = getCleanShadow(V: AShadow);
5516 // There is no LLVM IR intrinsic for _mm512_set1_epi8.
5517 // This loop generates a lot of LLVM IR, which we expect that CodeGen will
5518 // lower appropriately (e.g., VPBROADCASTB).
5519 // Besides, b is often a constant, in which case it is fully initialized.
5520 for (unsigned i = 0; i < NumElements; i++)
5521 BBroadcastShadow = IRB.CreateInsertElement(Vec: BBroadcastShadow, NewElt: BShadow, Idx: i);
5522
5523 setShadow(V: &I, SV: IRB.CreateOr(
5524 Ops: {AShadowXShadow, AShadowX, XShadowA, BBroadcastShadow}));
5525 setOriginForNaryOp(I);
5526 }
5527
5528 // Handle Arm NEON vector load intrinsics (vld*).
5529 //
5530 // The WithLane instructions (ld[234]lane) are similar to:
5531 // call {<4 x i32>, <4 x i32>, <4 x i32>}
5532 // @llvm.aarch64.neon.ld3lane.v4i32.p0
5533 // (<4 x i32> %L1, <4 x i32> %L2, <4 x i32> %L3, i64 %lane, ptr
5534 // %A)
5535 //
5536 // The non-WithLane instructions (ld[234], ld1x[234], ld[234]r) are similar
5537 // to:
5538 // call {<8 x i8>, <8 x i8>} @llvm.aarch64.neon.ld2.v8i8.p0(ptr %A)
5539 void handleNEONVectorLoad(IntrinsicInst &I, bool WithLane) {
5540 unsigned int numArgs = I.arg_size();
5541
5542 // Return type is a struct of vectors of integers or floating-point
5543 assert(I.getType()->isStructTy());
5544 [[maybe_unused]] StructType *RetTy = cast<StructType>(Val: I.getType());
5545 assert(RetTy->getNumElements() > 0);
5546 assert(RetTy->getElementType(0)->isIntOrIntVectorTy() ||
5547 RetTy->getElementType(0)->isFPOrFPVectorTy());
5548 for (unsigned int i = 0; i < RetTy->getNumElements(); i++)
5549 assert(RetTy->getElementType(i) == RetTy->getElementType(0));
5550
5551 if (WithLane) {
5552 // 2, 3 or 4 vectors, plus lane number, plus input pointer
5553 assert(4 <= numArgs && numArgs <= 6);
5554
5555 // Return type is a struct of the input vectors
5556 assert(RetTy->getNumElements() + 2 == numArgs);
5557 for (unsigned int i = 0; i < RetTy->getNumElements(); i++)
5558 assert(I.getArgOperand(i)->getType() == RetTy->getElementType(0));
5559 } else {
5560 assert(numArgs == 1);
5561 }
5562
5563 IRBuilder<> IRB(&I);
5564
5565 SmallVector<Value *, 6> ShadowArgs;
5566 if (WithLane) {
5567 for (unsigned int i = 0; i < numArgs - 2; i++)
5568 ShadowArgs.push_back(Elt: getShadow(V: I.getArgOperand(i)));
5569
5570 // Lane number, passed verbatim
5571 Value *LaneNumber = I.getArgOperand(i: numArgs - 2);
5572 ShadowArgs.push_back(Elt: LaneNumber);
5573
5574 // TODO: blend shadow of lane number into output shadow?
5575 insertCheckShadowOf(Val: LaneNumber, OrigIns: &I);
5576 }
5577
5578 Value *Src = I.getArgOperand(i: numArgs - 1);
5579 assert(Src->getType()->isPointerTy() && "Source is not a pointer!");
5580
5581 Type *SrcShadowTy = getShadowTy(V: Src);
5582 auto [SrcShadowPtr, SrcOriginPtr] =
5583 getShadowOriginPtr(Addr: Src, IRB, ShadowTy: SrcShadowTy, Alignment: Align(1), /*isStore*/ false);
5584 ShadowArgs.push_back(Elt: SrcShadowPtr);
5585
5586 // The NEON vector load instructions handled by this function all have
5587 // integer variants. It is easier to use those rather than trying to cast
5588 // a struct of vectors of floats into a struct of vectors of integers.
5589 CallInst *CI = IRB.CreateIntrinsicWithoutFolding(
5590 RetTy: getShadowTy(V: &I), ID: I.getIntrinsicID(), Args: ShadowArgs);
5591 setShadow(V: &I, SV: CI);
5592
5593 if (!MS.TrackOrigins)
5594 return;
5595
5596 Value *PtrSrcOrigin = IRB.CreateLoad(Ty: MS.OriginTy, Ptr: SrcOriginPtr);
5597 setOrigin(V: &I, Origin: PtrSrcOrigin);
5598 }
5599
5600 /// Handle Arm NEON vector store intrinsics (vst{2,3,4}, vst1x_{2,3,4},
5601 /// and vst{2,3,4}lane).
5602 ///
5603 /// Arm NEON vector store intrinsics have the output address (pointer) as the
5604 /// last argument, with the initial arguments being the inputs (and lane
5605 /// number for vst{2,3,4}lane). They return void.
5606 ///
5607 /// - st4 interleaves the output e.g., st4 (inA, inB, inC, inD, outP) writes
5608 /// abcdabcdabcdabcd... into *outP
5609 /// - st1_x4 is non-interleaved e.g., st1_x4 (inA, inB, inC, inD, outP)
5610 /// writes aaaa...bbbb...cccc...dddd... into *outP
5611 /// - st4lane has arguments of (inA, inB, inC, inD, lane, outP)
5612 /// These instructions can all be instrumented with essentially the same
5613 /// MSan logic, simply by applying the corresponding intrinsic to the shadow.
5614 void handleNEONVectorStoreIntrinsic(IntrinsicInst &I, bool useLane) {
5615 IRBuilder<> IRB(&I);
5616
5617 // Don't use getNumOperands() because it includes the callee
5618 int numArgOperands = I.arg_size();
5619
5620 // The last arg operand is the output (pointer)
5621 assert(numArgOperands >= 1);
5622 Value *Addr = I.getArgOperand(i: numArgOperands - 1);
5623 assert(Addr->getType()->isPointerTy());
5624 int skipTrailingOperands = 1;
5625
5626 if (ClCheckAccessAddress)
5627 insertCheckShadowOf(Val: Addr, OrigIns: &I);
5628
5629 // Second-last operand is the lane number (for vst{2,3,4}lane)
5630 if (useLane) {
5631 skipTrailingOperands++;
5632 assert(numArgOperands >= static_cast<int>(skipTrailingOperands));
5633 assert(isa<IntegerType>(
5634 I.getArgOperand(numArgOperands - skipTrailingOperands)->getType()));
5635 }
5636
5637 SmallVector<Value *, 8> ShadowArgs;
5638 // All the initial operands are the inputs
5639 for (int i = 0; i < numArgOperands - skipTrailingOperands; i++) {
5640 assert(isa<FixedVectorType>(I.getArgOperand(i)->getType()));
5641 Value *Shadow = getShadow(I: &I, i);
5642 ShadowArgs.append(NumInputs: 1, Elt: Shadow);
5643 }
5644
5645 // MSan's GetShadowTy assumes the LHS is the type we want the shadow for
5646 // e.g., for:
5647 // [[TMP5:%.*]] = bitcast <16 x i8> [[TMP2]] to i128
5648 // we know the type of the output (and its shadow) is <16 x i8>.
5649 //
5650 // Arm NEON VST is unusual because the last argument is the output address:
5651 // define void @st2_16b(<16 x i8> %A, <16 x i8> %B, ptr %P) {
5652 // call void @llvm.aarch64.neon.st2.v16i8.p0
5653 // (<16 x i8> [[A]], <16 x i8> [[B]], ptr [[P]])
5654 // and we have no type information about P's operand. We must manually
5655 // compute the type (<16 x i8> x 2).
5656 FixedVectorType *OutputVectorTy = FixedVectorType::get(
5657 ElementType: cast<FixedVectorType>(Val: I.getArgOperand(i: 0)->getType())->getElementType(),
5658 NumElts: cast<FixedVectorType>(Val: I.getArgOperand(i: 0)->getType())->getNumElements() *
5659 (numArgOperands - skipTrailingOperands));
5660 Type *OutputShadowTy = getShadowTy(OrigTy: OutputVectorTy);
5661
5662 if (useLane)
5663 ShadowArgs.append(NumInputs: 1,
5664 Elt: I.getArgOperand(i: numArgOperands - skipTrailingOperands));
5665
5666 Value *OutputShadowPtr, *OutputOriginPtr;
5667 // AArch64 NEON does not need alignment (unless OS requires it)
5668 std::tie(args&: OutputShadowPtr, args&: OutputOriginPtr) = getShadowOriginPtr(
5669 Addr, IRB, ShadowTy: OutputShadowTy, Alignment: Align(1), /*isStore*/ true);
5670 ShadowArgs.append(NumInputs: 1, Elt: OutputShadowPtr);
5671
5672 CallInst *CI = IRB.CreateIntrinsicWithoutFolding(
5673 RetTy: IRB.getVoidTy(), ID: I.getIntrinsicID(), Args: ShadowArgs);
5674 setShadow(V: &I, SV: CI);
5675
5676 if (MS.TrackOrigins) {
5677 // TODO: if we modelled the vst* instruction more precisely, we could
5678 // more accurately track the origins (e.g., if both inputs are
5679 // uninitialized for vst2, we currently blame the second input, even
5680 // though part of the output depends only on the first input).
5681 //
5682 // This is particularly imprecise for vst{2,3,4}lane, since only one
5683 // lane of each input is actually copied to the output.
5684 OriginCombiner OC(this, IRB);
5685 for (int i = 0; i < numArgOperands - skipTrailingOperands; i++)
5686 OC.Add(V: I.getArgOperand(i));
5687
5688 const DataLayout &DL = F.getDataLayout();
5689 OC.DoneAndStoreOrigin(TS: DL.getTypeStoreSize(Ty: OutputVectorTy),
5690 OriginPtr: OutputOriginPtr);
5691 }
5692 }
5693
5694 // Integer matrix multiplication:
5695 // - <4 x i32> @llvm.aarch64.neon.{s,u,us}mmla.v4i32.v16i8
5696 // (<4 x i32> %R, <16 x i8> %A, <16 x i8> %B)
5697 // - <4 x i32> is a 2x2 matrix
5698 // - <16 x i8> %A and %B are 2x8 and 8x2 matrices respectively
5699 //
5700 // Floating-point matrix multiplication:
5701 // - <4 x float> @llvm.aarch64.neon.bfmmla
5702 // (<4 x float> %R, <8 x bfloat> %A, <8 x bfloat> %B)
5703 // - <4 x float> is a 2x2 matrix
5704 // - <8 x bfloat> %A and %B are 2x4 and 4x2 matrices respectively
5705 //
5706 // The general shadow propagation approach is:
5707 // 1) get the shadows of the input matrices %A and %B
5708 // 2) map each shadow value to 0x1 if the corresponding value is fully
5709 // initialized, and 0x0 otherwise
5710 // 3) perform a matrix multiplication on the shadows of %A and %B [*].
5711 // The output will be a 2x2 matrix. For each element, a value of 0x8
5712 // (for {s,u,us}mmla) or 0x4 (for bfmmla) means all the corresponding
5713 // inputs were clean; if so, set the shadow to zero, otherwise set to -1.
5714 // 4) blend in the shadow of %R
5715 //
5716 // [*] Since shadows are integral, the obvious approach is to always apply
5717 // ummla to the shadows. Unfortunately, Armv8.2+bf16 supports bfmmla,
5718 // but not ummla. Thus, for bfmmla, our instrumentation reuses bfmmla.
5719 //
5720 // TODO: consider allowing multiplication of zero with an uninitialized value
5721 // to result in an initialized value.
5722 void handleNEONMatrixMultiply(IntrinsicInst &I) {
5723 IRBuilder<> IRB(&I);
5724
5725 assert(I.arg_size() == 3);
5726 Value *R = I.getArgOperand(i: 0);
5727 Value *A = I.getArgOperand(i: 1);
5728 Value *B = I.getArgOperand(i: 2);
5729
5730 assert(I.getType() == R->getType());
5731
5732 assert(isa<FixedVectorType>(R->getType()));
5733 assert(isa<FixedVectorType>(A->getType()));
5734 assert(isa<FixedVectorType>(B->getType()));
5735
5736 FixedVectorType *RTy = cast<FixedVectorType>(Val: R->getType());
5737 FixedVectorType *ATy = cast<FixedVectorType>(Val: A->getType());
5738 FixedVectorType *BTy = cast<FixedVectorType>(Val: B->getType());
5739 assert(ATy->getElementType() == BTy->getElementType());
5740
5741 if (RTy->getElementType()->isIntegerTy()) {
5742 // <4 x i32> @llvm.aarch64.neon.ummla.v4i32.v16i8
5743 // (<4 x i32> %R, <16 x i8> %X, <16 x i8> %Y)
5744 assert(RTy == FixedVectorType::get(IntegerType::get(*MS.C, 32), 4));
5745 assert(ATy == FixedVectorType::get(IntegerType::get(*MS.C, 8), 16));
5746 assert(BTy == FixedVectorType::get(IntegerType::get(*MS.C, 8), 16));
5747 } else {
5748 // <4 x float> @llvm.aarch64.neon.bfmmla
5749 // (<4 x float> %R, <8 x bfloat> %X, <8 x bfloat> %Y)
5750 assert(RTy == FixedVectorType::get(Type::getFloatTy(*MS.C), 4));
5751 assert(ATy == FixedVectorType::get(Type::getBFloatTy(*MS.C), 8));
5752 assert(BTy == FixedVectorType::get(Type::getBFloatTy(*MS.C), 8));
5753 }
5754
5755 Value *ShadowR = getShadow(I: &I, i: 0);
5756 Value *ShadowA = getShadow(I: &I, i: 1);
5757 Value *ShadowB = getShadow(I: &I, i: 2);
5758
5759 Value *ShadowAB;
5760 Value *FullyInit;
5761
5762 if (RTy->getElementType()->isIntegerTy()) {
5763 // If the value is fully initialized, the shadow will be 000...001.
5764 // Otherwise, the shadow will be all zero.
5765 // (This is the opposite of how we typically handle shadows.)
5766 ShadowA = IRB.CreateZExt(V: IRB.CreateICmpEQ(LHS: ShadowA, RHS: getCleanShadow(OrigTy: ATy)),
5767 DestTy: getShadowTy(OrigTy: ATy));
5768 ShadowB = IRB.CreateZExt(V: IRB.CreateICmpEQ(LHS: ShadowB, RHS: getCleanShadow(OrigTy: BTy)),
5769 DestTy: getShadowTy(OrigTy: BTy));
5770 // TODO: the CreateSelect approach used below for floating-point is more
5771 // generic than CreateZExt. Investigate whether it is worthwhile
5772 // unifying the two approaches.
5773
5774 ShadowAB = IRB.CreateIntrinsic(RetTy: RTy, ID: Intrinsic::aarch64_neon_ummla,
5775 Args: {getCleanShadow(OrigTy: RTy), ShadowA, ShadowB});
5776
5777 // ummla multiplies a 2x8 matrix with an 8x2 matrix. If all entries of the
5778 // input matrices are equal to 0x1, all entries of the output matrix will
5779 // be 0x8.
5780 FullyInit = ConstantVector::getSplat(
5781 EC: RTy->getElementCount(), Elt: ConstantInt::get(Ty: RTy->getElementType(), V: 0x8));
5782
5783 ShadowAB = IRB.CreateICmpNE(LHS: ShadowAB, RHS: FullyInit);
5784 } else {
5785 Constant *ABZeros = ConstantVector::getSplat(
5786 EC: ATy->getElementCount(), Elt: ConstantFP::get(Ty: ATy->getElementType(), V: 0));
5787 Constant *ABOnes = ConstantVector::getSplat(
5788 EC: ATy->getElementCount(), Elt: ConstantFP::get(Ty: ATy->getElementType(), V: 1));
5789
5790 // As per the integer case, if the shadow is clean, we store 0x1,
5791 // otherwise we store 0x0 (the opposite of usual shadow arithmetic).
5792 ShadowA = IRB.CreateSelect(C: IRB.CreateICmpEQ(LHS: ShadowA, RHS: getCleanShadow(OrigTy: ATy)),
5793 True: ABOnes, False: ABZeros);
5794 ShadowB = IRB.CreateSelect(C: IRB.CreateICmpEQ(LHS: ShadowB, RHS: getCleanShadow(OrigTy: BTy)),
5795 True: ABOnes, False: ABZeros);
5796
5797 Constant *RZeros = ConstantVector::getSplat(
5798 EC: RTy->getElementCount(), Elt: ConstantFP::get(Ty: RTy->getElementType(), V: 0));
5799
5800 ShadowAB = IRB.CreateIntrinsic(RetTy: RTy, ID: Intrinsic::aarch64_neon_bfmmla,
5801 Args: {RZeros, ShadowA, ShadowB});
5802
5803 // bfmmla multiplies a 2x4 matrix with an 4x2 matrix. If all entries of
5804 // the input matrices are equal to 0x1, all entries of the output matrix
5805 // will be 4.0. (To avoid floating-point error, we check if each entry
5806 // < 3.5.)
5807 FullyInit = ConstantVector::getSplat(
5808 EC: RTy->getElementCount(), Elt: ConstantFP::get(Ty: RTy->getElementType(), V: 3.5));
5809
5810 // FCmpULT: "yields true if either operand is a QNAN or op1 is less than"
5811 // op2"
5812 ShadowAB = IRB.CreateFCmpULT(LHS: ShadowAB, RHS: FullyInit);
5813 }
5814
5815 ShadowR = IRB.CreateICmpNE(LHS: ShadowR, RHS: getCleanShadow(OrigTy: RTy));
5816 ShadowR = IRB.CreateOr(LHS: ShadowAB, RHS: ShadowR);
5817
5818 setShadow(V: &I, SV: IRB.CreateSExt(V: ShadowR, DestTy: getShadowTy(OrigTy: RTy)));
5819
5820 setOriginForNaryOp(I);
5821 }
5822
5823 /// Handle intrinsics by applying the intrinsic to the shadows.
5824 ///
5825 /// For example, this can be applied to the Arm NEON vector table intrinsics
5826 /// (tbl{1,2,3,4}).
5827 ///
5828 /// Typically, shadowIntrinsicID will be specified by the caller to be
5829 /// I.getIntrinsicID(), but the caller can choose to replace it with another
5830 /// intrinsic of the same type.
5831 ///
5832 /// The trailing arguments are passed verbatim to the intrinsic, though any
5833 /// uninitialized trailing arguments can also taint the shadow e.g., for an
5834 /// intrinsic with one trailing verbatim argument:
5835 /// out = intrinsic(var1, var2, opType)
5836 /// we compute:
5837 /// shadow[out] =
5838 /// intrinsic(shadow[var1], shadow[var2], opType) | shadow[opType]
5839 ///
5840 /// If an intrinsic is called with floating-point arguments, we will
5841 /// typically cast the shadows to floating-point, apply the intrinsic [*],
5842 /// then cast the result back to integer/shadow.
5843 ///
5844 /// In cases where we know the intrinsic is compatible with integer
5845 /// arguments, 'forceIntegerIntrinsic' will apply the integer variant, even
5846 /// if the arguments are floating-point, thus avoiding unnecessary casts
5847 /// e.g., if I is:
5848 /// <16 x float> @llvm.x86.avx512.mask.compress
5849 /// (<16 x float>, <16 x float>, <16 x i1> %mask)
5850 /// we would prefer to compute the shadows using:
5851 /// <16 x i32> @llvm.x86.avx512.mask.compress
5852 /// (<16 x i32>, <16 x i32>, <16 x i1> %mask)
5853 ///
5854 /// [*] CAUTION: this assumes that the intrinsic will handle arbitrary
5855 /// bit-patterns (for example, if the intrinsic accepts floats
5856 /// for var1, we require that it doesn't care if inputs are
5857 /// NaNs).
5858 ///
5859 /// The origin is approximated using setOriginForNaryOp.
5860 void handleIntrinsicByApplyingToShadow(IntrinsicInst &I,
5861 Intrinsic::ID shadowIntrinsicID,
5862 unsigned int trailingVerbatimArgs,
5863 bool forceIntegerIntrinsic) {
5864 IRBuilder<> IRB(&I);
5865
5866 assert(trailingVerbatimArgs < I.arg_size());
5867
5868 SmallVector<Value *, 8> ShadowArgs;
5869 // Don't use getNumOperands() because it includes the callee
5870 for (unsigned int i = 0; i < I.arg_size() - trailingVerbatimArgs; i++) {
5871 Value *Shadow = getShadow(I: &I, i);
5872
5873 if (forceIntegerIntrinsic)
5874 ShadowArgs.push_back(Elt: Shadow);
5875 else
5876 ShadowArgs.push_back(
5877 Elt: IRB.CreateBitCast(V: Shadow, DestTy: I.getArgOperand(i)->getType()));
5878 }
5879
5880 for (unsigned int i = I.arg_size() - trailingVerbatimArgs; i < I.arg_size();
5881 i++) {
5882 Value *Arg = I.getArgOperand(i);
5883 if (forceIntegerIntrinsic)
5884 assert(Arg->getType()->isIntOrIntVectorTy());
5885 ShadowArgs.push_back(Elt: Arg);
5886 }
5887
5888 Value *CombinedShadow;
5889 if (forceIntegerIntrinsic) {
5890 CombinedShadow =
5891 IRB.CreateIntrinsic(RetTy: getShadowTy(V: &I), ID: shadowIntrinsicID, Args: ShadowArgs);
5892 } else {
5893 Value *CI =
5894 IRB.CreateIntrinsic(RetTy: I.getType(), ID: shadowIntrinsicID, Args: ShadowArgs);
5895 CombinedShadow = IRB.CreateBitCast(V: CI, DestTy: getShadowTy(V: &I));
5896 }
5897
5898 // Combine the computed shadow with the shadow of trailing args
5899 for (unsigned int i = I.arg_size() - trailingVerbatimArgs; i < I.arg_size();
5900 i++) {
5901 Value *Shadow =
5902 CreateShadowCast(IRB, V: getShadow(I: &I, i), dstTy: CombinedShadow->getType());
5903 CombinedShadow = IRB.CreateOr(LHS: Shadow, RHS: CombinedShadow, Name: "_msprop");
5904 }
5905
5906 setShadow(V: &I, SV: CombinedShadow);
5907
5908 setOriginForNaryOp(I);
5909 }
5910
5911 // Approximation only
5912 //
5913 // e.g., <16 x i8> @llvm.aarch64.neon.pmull64(i64, i64)
5914 void handleNEONVectorMultiplyIntrinsic(IntrinsicInst &I) {
5915 assert(I.arg_size() == 2);
5916
5917 handleShadowOr(I);
5918 }
5919
5920 // Handles:
5921 // <4 x half> @llvm.aarch64.neon.fp8.fdot2.lane
5922 // (<4 x half>, <8 x i8>, <16 x i8>, i32)
5923 // accumulator A B lane
5924 //
5925 // <8 x half> @llvm.aarch64.neon.fp8.fdot2.lane
5926 // (<8 x half>, <16 x i8>, <16 x i8>, i32)
5927 // <2 x float> @llvm.aarch64.neon.fp8.fdot4.lane
5928 // (<2 x float>, <8 x i8>, <16 x i8>, i32)
5929 // <4 x float> @llvm.aarch64.neon.fp8.fdot4.lane
5930 // (<4 x float>, <16 x i8>, <16 x i8>, i32)
5931 //
5932 // The lane specifies which pair (fdot2) or quad (fdot4) of numbers to
5933 // extract from B, which is then splatted before being used in the dot
5934 // products e.g., for
5935 // <4 x half> @llvm.aarch64.neon.fp8.fdot2.lane:
5936 // (<4 x half>, <8 x i8>, <16 x i8>, 1)
5937 //
5938 // acc[0] acc[1] acc[2] acc[3]
5939 // + + + + + + + +
5940 // A[0] A[1] A[2] A[3] A[4] A[5] A[6] A[7]
5941 // * * * * * * * *
5942 // B[2] B[3] B[2] B[3] B[2] B[3] B[2] B[3]
5943 //
5944 // Notice that if any bit of B[2] or B[3] is uninitialized, every accumulator
5945 // value will become tainted; we approximate this by marking the output as
5946 // fully uninitialized. This permits a 'Select' optimization.
5947 //
5948 // This function is separate from handleVectorDotProductIntrinsic(), because
5949 // the non-overlapping features (e.g., odd/even lanes vs. numbered lanes,
5950 // ZeroPurifies, EltSizeInBits) and optimizations make clean code reuse
5951 // difficult.
5952 void handleNEONDotProductLaneIntrinsic(IntrinsicInst &I,
5953 unsigned ReductionFactor) {
5954 IRBuilder<> IRB(&I);
5955 assert(I.arg_size() == 4);
5956
5957 [[maybe_unused]] Value *VAcc = I.getOperand(i_nocapture: 0);
5958 [[maybe_unused]] Value *Va = I.getOperand(i_nocapture: 1);
5959 [[maybe_unused]] Value *Vb = I.getOperand(i_nocapture: 2);
5960 Value *Lane = I.getOperand(i_nocapture: 3);
5961
5962 assert(isa<FixedVectorType>(VAcc->getType()));
5963 assert(VAcc->getType() == I.getType());
5964
5965 assert(isa<FixedVectorType>(Va->getType()));
5966 assert(Va->getType()->getPrimitiveSizeInBits() ==
5967 I.getType()->getPrimitiveSizeInBits());
5968
5969 assert(cast<FixedVectorType>(Va->getType())->getNumElements() ==
5970 cast<FixedVectorType>(I.getType())->getNumElements() *
5971 ReductionFactor);
5972
5973 assert(isa<FixedVectorType>(Vb->getType()));
5974 // Deliberately not strict equality
5975 assert(Vb->getType()->getPrimitiveSizeInBits() >=
5976 I.getType()->getPrimitiveSizeInBits());
5977
5978 assert(Lane->getType()->isIntegerTy());
5979
5980 // (<4 x 16>, <8 x i8>, <16 x i8>)
5981 // SAcc Sa Sb
5982 Value *SAcc = getShadow(I: &I, i: 0);
5983 Value *Sa = getShadow(I: &I, i: 1);
5984 Value *Sb = getShadow(I: &I, i: 2);
5985
5986 // Cast the shadows to:
5987 // (<4 x i16>, <4 x i16>, <8 x i16>)
5988 // SAcc Sa Sb
5989 Sa = IRB.CreateBitCast(V: Sa, DestTy: SAcc->getType());
5990 Sb = IRB.CreateBitCast(
5991 V: Sb, DestTy: FixedVectorType::getWithSizeAndScalar(
5992 SizeTy: cast<FixedVectorType>(Val: Sb->getType()),
5993 EltTy: cast<FixedVectorType>(Val: SAcc->getType())->getElementType()));
5994
5995 // All-or-nothing shadows
5996 Sa =
5997 IRB.CreateSExt(V: IRB.CreateICmpNE(LHS: Sa, RHS: getCleanShadow(V: Sa)), DestTy: Sa->getType());
5998
5999 // Extract the specific lane from Sb to get i16, then turn it into a single
6000 // bit representing if it is fully initialized.
6001 Sb = IRB.CreateExtractElement(Vec: Sb, Idx: Lane);
6002 Value *SbClean = IRB.CreateIsNull(Arg: Sb);
6003
6004 Value *SOutput = IRB.CreateOr(LHS: SAcc, RHS: Sa);
6005
6006 // Select is cheaper than broadcasting Sb into <4 x i16>.
6007 SOutput = IRB.CreateSelect(C: SbClean, True: SOutput, False: getPoisonedShadow(V: SOutput));
6008
6009 setShadow(V: &I, SV: SOutput);
6010 setOriginForNaryOp(I);
6011 }
6012
6013 bool maybeHandleCrossPlatformIntrinsic(IntrinsicInst &I) {
6014 switch (I.getIntrinsicID()) {
6015 case Intrinsic::uadd_with_overflow:
6016 case Intrinsic::sadd_with_overflow:
6017 case Intrinsic::usub_with_overflow:
6018 case Intrinsic::ssub_with_overflow:
6019 case Intrinsic::umul_with_overflow:
6020 case Intrinsic::smul_with_overflow:
6021 handleArithmeticWithOverflow(I);
6022 break;
6023 case Intrinsic::modf:
6024 case Intrinsic::sincos:
6025 case Intrinsic::sincospi:
6026 handleModfOrSincos(I);
6027 break;
6028 case Intrinsic::abs:
6029 handleAbsIntrinsic(I);
6030 break;
6031 case Intrinsic::bitreverse:
6032 handleIntrinsicByApplyingToShadow(I, shadowIntrinsicID: I.getIntrinsicID(),
6033 /*trailingVerbatimArgs=*/0,
6034 /*forceIntegerIntrinsic=*/false);
6035 break;
6036 case Intrinsic::is_fpclass:
6037 handleIsFpClass(I);
6038 break;
6039 case Intrinsic::lifetime_start:
6040 handleLifetimeStart(I);
6041 break;
6042 case Intrinsic::launder_invariant_group:
6043 handleInvariantGroup(I);
6044 break;
6045 case Intrinsic::bswap:
6046 handleBswap(I);
6047 break;
6048 case Intrinsic::ctlz:
6049 case Intrinsic::cttz:
6050 handleCountLeadingTrailingZeros(I);
6051 break;
6052 case Intrinsic::masked_compressstore:
6053 handleMaskedCompressStore(I);
6054 break;
6055 case Intrinsic::masked_expandload:
6056 handleMaskedExpandLoad(I);
6057 break;
6058 case Intrinsic::masked_gather:
6059 handleMaskedGather(I);
6060 break;
6061 case Intrinsic::masked_scatter:
6062 handleMaskedScatter(I);
6063 break;
6064 case Intrinsic::masked_store:
6065 handleMaskedStore(I);
6066 break;
6067 case Intrinsic::masked_load:
6068 handleMaskedLoad(I);
6069 break;
6070 case Intrinsic::masked_udiv:
6071 case Intrinsic::masked_sdiv:
6072 case Intrinsic::masked_urem:
6073 case Intrinsic::masked_srem:
6074 handleMaskedIntegerDivRem(I);
6075 break;
6076 case Intrinsic::vector_reduce_and:
6077 handleVectorReduceAndIntrinsic(I);
6078 break;
6079 case Intrinsic::vector_reduce_or:
6080 handleVectorReduceOrIntrinsic(I);
6081 break;
6082
6083 case Intrinsic::vector_reduce_add:
6084 case Intrinsic::vector_reduce_xor:
6085 case Intrinsic::vector_reduce_mul:
6086 // Signed/Unsigned Min/Max
6087 // TODO: handling similarly to AND/OR may be more precise.
6088 case Intrinsic::vector_reduce_smax:
6089 case Intrinsic::vector_reduce_smin:
6090 case Intrinsic::vector_reduce_umax:
6091 case Intrinsic::vector_reduce_umin:
6092 // TODO: this has no false positives, but arguably we should check that all
6093 // the bits are initialized.
6094 case Intrinsic::vector_reduce_fmax:
6095 case Intrinsic::vector_reduce_fmin:
6096 handleVectorReduceIntrinsic(I, /*AllowShadowCast=*/false);
6097 break;
6098
6099 case Intrinsic::vector_reduce_fadd:
6100 case Intrinsic::vector_reduce_fmul:
6101 handleVectorReduceWithStarterIntrinsic(I);
6102 break;
6103
6104 case Intrinsic::scmp:
6105 case Intrinsic::ucmp: {
6106 handleShadowOr(I);
6107 break;
6108 }
6109
6110 case Intrinsic::fshl:
6111 case Intrinsic::fshr:
6112 handleFunnelShift(I);
6113 break;
6114
6115 case Intrinsic::pdep:
6116 case Intrinsic::pext:
6117 handleGenericBitManipulation(I);
6118 break;
6119
6120 case Intrinsic::is_constant:
6121 // The result of llvm.is.constant() is always defined.
6122 setShadow(V: &I, SV: getCleanShadow(V: &I));
6123 setOrigin(V: &I, Origin: getCleanOrigin());
6124 break;
6125
6126 // The non-saturating versions are handled by visitFPTo[US]IInst().
6127 //
6128 // N.B. some platform-specific intrinsics, such as AArch64 fcvtz[us], are
6129 // lowered to these cross-platform intrinsics.
6130 case Intrinsic::fptosi_sat:
6131 case Intrinsic::fptoui_sat:
6132 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/false);
6133 break;
6134
6135 // e.g.,
6136 // notail call void (...) @llvm.fake.use(i64 %x)
6137 // notail call void (...) @llvm.fake.use(i32 %y)
6138 // notail call void (...) @llvm.fake.use(ptr %z)
6139 case Intrinsic::fake_use:
6140 assert(I.getType()->isVoidTy());
6141 // fake_uses aren't real, they can't hurt you. If the use isn't real, it
6142 // can't be a real use-of-uninitialized memory. Silently skip over
6143 // fake_use.
6144 return true;
6145
6146 default:
6147 return false;
6148 }
6149
6150 return true;
6151 }
6152
6153 bool maybeHandleX86SIMDIntrinsic(IntrinsicInst &I) {
6154 switch (I.getIntrinsicID()) {
6155 case Intrinsic::x86_sse_stmxcsr:
6156 handleStmxcsr(I);
6157 break;
6158 case Intrinsic::x86_sse_ldmxcsr:
6159 handleLdmxcsr(I);
6160 break;
6161
6162 // Convert Scalar Double Precision Floating-Point Value
6163 // to Unsigned Doubleword Integer
6164 // etc.
6165 case Intrinsic::x86_avx512_vcvtsd2usi64:
6166 case Intrinsic::x86_avx512_vcvtsd2usi32:
6167 case Intrinsic::x86_avx512_vcvtss2usi64:
6168 case Intrinsic::x86_avx512_vcvtss2usi32:
6169 case Intrinsic::x86_avx512_cvttss2usi64:
6170 case Intrinsic::x86_avx512_cvttss2usi:
6171 case Intrinsic::x86_avx512_cvttsd2usi64:
6172 case Intrinsic::x86_avx512_cvttsd2usi:
6173 case Intrinsic::x86_avx512_cvtusi2ss:
6174 case Intrinsic::x86_avx512_cvtusi642sd:
6175 case Intrinsic::x86_avx512_cvtusi642ss:
6176 handleSSEVectorConvertIntrinsic(I, NumUsedElements: 1, HasRoundingMode: true);
6177 break;
6178 case Intrinsic::x86_sse2_cvtsd2si64:
6179 case Intrinsic::x86_sse2_cvtsd2si:
6180 case Intrinsic::x86_sse2_cvtsd2ss:
6181 case Intrinsic::x86_sse2_cvttsd2si64:
6182 case Intrinsic::x86_sse2_cvttsd2si:
6183 case Intrinsic::x86_sse_cvtss2si64:
6184 case Intrinsic::x86_sse_cvtss2si:
6185 case Intrinsic::x86_sse_cvttss2si64:
6186 case Intrinsic::x86_sse_cvttss2si:
6187 handleSSEVectorConvertIntrinsic(I, NumUsedElements: 1);
6188 break;
6189 case Intrinsic::x86_sse_cvtps2pi:
6190 case Intrinsic::x86_sse_cvttps2pi:
6191 handleSSEVectorConvertIntrinsic(I, NumUsedElements: 2);
6192 break;
6193
6194 // TODO:
6195 // <1 x i64> @llvm.x86.sse.cvtpd2pi(<2 x double>)
6196 // <2 x double> @llvm.x86.sse.cvtpi2pd(<1 x i64>)
6197 // <4 x float> @llvm.x86.sse.cvtpi2ps(<4 x float>, <1 x i64>)
6198
6199 case Intrinsic::x86_vcvtps2ph_128:
6200 case Intrinsic::x86_vcvtps2ph_256: {
6201 handleSSEVectorConvertIntrinsicByProp(I, /*HasRoundingMode=*/true);
6202 break;
6203 }
6204
6205 // Convert Packed Single Precision Floating-Point Values
6206 // to Packed Signed Doubleword Integer Values
6207 //
6208 // <16 x i32> @llvm.x86.avx512.mask.cvtps2dq.512
6209 // (<16 x float>, <16 x i32>, i16, i32)
6210 case Intrinsic::x86_avx512_mask_cvtps2dq_512:
6211 handleAVX512VectorConvertFPToInt(I, /*LastMask=*/false);
6212 break;
6213
6214 // Convert Packed Double Precision Floating-Point Values
6215 // to Packed Single Precision Floating-Point Values
6216 case Intrinsic::x86_sse2_cvtpd2ps:
6217 case Intrinsic::x86_sse2_cvtps2dq:
6218 case Intrinsic::x86_sse2_cvtpd2dq:
6219 case Intrinsic::x86_sse2_cvttps2dq:
6220 case Intrinsic::x86_sse2_cvttpd2dq:
6221 case Intrinsic::x86_avx_cvt_pd2_ps_256:
6222 case Intrinsic::x86_avx_cvt_ps2dq_256:
6223 case Intrinsic::x86_avx_cvt_pd2dq_256:
6224 case Intrinsic::x86_avx_cvtt_ps2dq_256:
6225 case Intrinsic::x86_avx_cvtt_pd2dq_256: {
6226 handleSSEVectorConvertIntrinsicByProp(I, /*HasRoundingMode=*/false);
6227 break;
6228 }
6229
6230 // Convert Single-Precision FP Value to 16-bit FP Value
6231 // <16 x i16> @llvm.x86.avx512.mask.vcvtps2ph.512
6232 // (<16 x float>, i32, <16 x i16>, i16)
6233 // <8 x i16> @llvm.x86.avx512.mask.vcvtps2ph.128
6234 // (<4 x float>, i32, <8 x i16>, i8)
6235 // <8 x i16> @llvm.x86.avx512.mask.vcvtps2ph.256
6236 // (<8 x float>, i32, <8 x i16>, i8)
6237 case Intrinsic::x86_avx512_mask_vcvtps2ph_512:
6238 case Intrinsic::x86_avx512_mask_vcvtps2ph_256:
6239 case Intrinsic::x86_avx512_mask_vcvtps2ph_128:
6240 handleAVX512VectorConvertFPToInt(I, /*LastMask=*/true);
6241 break;
6242
6243 // Shift Packed Data (Left Logical, Right Arithmetic, Right Logical)
6244 case Intrinsic::x86_avx512_psll_w_512:
6245 case Intrinsic::x86_avx512_psll_d_512:
6246 case Intrinsic::x86_avx512_psll_q_512:
6247 case Intrinsic::x86_avx512_pslli_w_512:
6248 case Intrinsic::x86_avx512_pslli_d_512:
6249 case Intrinsic::x86_avx512_pslli_q_512:
6250 case Intrinsic::x86_avx512_psrl_w_512:
6251 case Intrinsic::x86_avx512_psrl_d_512:
6252 case Intrinsic::x86_avx512_psrl_q_512:
6253 case Intrinsic::x86_avx512_psra_w_512:
6254 case Intrinsic::x86_avx512_psra_d_512:
6255 case Intrinsic::x86_avx512_psra_q_512:
6256 case Intrinsic::x86_avx512_psrli_w_512:
6257 case Intrinsic::x86_avx512_psrli_d_512:
6258 case Intrinsic::x86_avx512_psrli_q_512:
6259 case Intrinsic::x86_avx512_psrai_w_512:
6260 case Intrinsic::x86_avx512_psrai_d_512:
6261 case Intrinsic::x86_avx512_psrai_q_512:
6262 case Intrinsic::x86_avx512_psra_q_256:
6263 case Intrinsic::x86_avx512_psra_q_128:
6264 case Intrinsic::x86_avx512_psrai_q_256:
6265 case Intrinsic::x86_avx512_psrai_q_128:
6266 case Intrinsic::x86_avx2_psll_w:
6267 case Intrinsic::x86_avx2_psll_d:
6268 case Intrinsic::x86_avx2_psll_q:
6269 case Intrinsic::x86_avx2_pslli_w:
6270 case Intrinsic::x86_avx2_pslli_d:
6271 case Intrinsic::x86_avx2_pslli_q:
6272 case Intrinsic::x86_avx2_psrl_w:
6273 case Intrinsic::x86_avx2_psrl_d:
6274 case Intrinsic::x86_avx2_psrl_q:
6275 case Intrinsic::x86_avx2_psra_w:
6276 case Intrinsic::x86_avx2_psra_d:
6277 case Intrinsic::x86_avx2_psrli_w:
6278 case Intrinsic::x86_avx2_psrli_d:
6279 case Intrinsic::x86_avx2_psrli_q:
6280 case Intrinsic::x86_avx2_psrai_w:
6281 case Intrinsic::x86_avx2_psrai_d:
6282 case Intrinsic::x86_sse2_psll_w:
6283 case Intrinsic::x86_sse2_psll_d:
6284 case Intrinsic::x86_sse2_psll_q:
6285 case Intrinsic::x86_sse2_pslli_w:
6286 case Intrinsic::x86_sse2_pslli_d:
6287 case Intrinsic::x86_sse2_pslli_q:
6288 case Intrinsic::x86_sse2_psrl_w:
6289 case Intrinsic::x86_sse2_psrl_d:
6290 case Intrinsic::x86_sse2_psrl_q:
6291 case Intrinsic::x86_sse2_psra_w:
6292 case Intrinsic::x86_sse2_psra_d:
6293 case Intrinsic::x86_sse2_psrli_w:
6294 case Intrinsic::x86_sse2_psrli_d:
6295 case Intrinsic::x86_sse2_psrli_q:
6296 case Intrinsic::x86_sse2_psrai_w:
6297 case Intrinsic::x86_sse2_psrai_d:
6298 case Intrinsic::x86_mmx_psll_w:
6299 case Intrinsic::x86_mmx_psll_d:
6300 case Intrinsic::x86_mmx_psll_q:
6301 case Intrinsic::x86_mmx_pslli_w:
6302 case Intrinsic::x86_mmx_pslli_d:
6303 case Intrinsic::x86_mmx_pslli_q:
6304 case Intrinsic::x86_mmx_psrl_w:
6305 case Intrinsic::x86_mmx_psrl_d:
6306 case Intrinsic::x86_mmx_psrl_q:
6307 case Intrinsic::x86_mmx_psra_w:
6308 case Intrinsic::x86_mmx_psra_d:
6309 case Intrinsic::x86_mmx_psrli_w:
6310 case Intrinsic::x86_mmx_psrli_d:
6311 case Intrinsic::x86_mmx_psrli_q:
6312 case Intrinsic::x86_mmx_psrai_w:
6313 case Intrinsic::x86_mmx_psrai_d:
6314 handleVectorShiftIntrinsic(I, /* Variable */ false);
6315 break;
6316 case Intrinsic::x86_avx2_psllv_d:
6317 case Intrinsic::x86_avx2_psllv_d_256:
6318 case Intrinsic::x86_avx512_psllv_d_512:
6319 case Intrinsic::x86_avx2_psllv_q:
6320 case Intrinsic::x86_avx2_psllv_q_256:
6321 case Intrinsic::x86_avx512_psllv_q_512:
6322 case Intrinsic::x86_avx2_psrlv_d:
6323 case Intrinsic::x86_avx2_psrlv_d_256:
6324 case Intrinsic::x86_avx512_psrlv_d_512:
6325 case Intrinsic::x86_avx2_psrlv_q:
6326 case Intrinsic::x86_avx2_psrlv_q_256:
6327 case Intrinsic::x86_avx512_psrlv_q_512:
6328 case Intrinsic::x86_avx2_psrav_d:
6329 case Intrinsic::x86_avx2_psrav_d_256:
6330 case Intrinsic::x86_avx512_psrav_d_512:
6331 case Intrinsic::x86_avx512_psrav_q_128:
6332 case Intrinsic::x86_avx512_psrav_q_256:
6333 case Intrinsic::x86_avx512_psrav_q_512:
6334 handleVectorShiftIntrinsic(I, /* Variable */ true);
6335 break;
6336
6337 // Pack with Signed/Unsigned Saturation
6338 case Intrinsic::x86_sse2_packsswb_128:
6339 case Intrinsic::x86_sse2_packssdw_128:
6340 case Intrinsic::x86_sse2_packuswb_128:
6341 case Intrinsic::x86_sse41_packusdw:
6342 case Intrinsic::x86_avx2_packsswb:
6343 case Intrinsic::x86_avx2_packssdw:
6344 case Intrinsic::x86_avx2_packuswb:
6345 case Intrinsic::x86_avx2_packusdw:
6346 // e.g., <64 x i8> @llvm.x86.avx512.packsswb.512
6347 // (<32 x i16> %a, <32 x i16> %b)
6348 // <32 x i16> @llvm.x86.avx512.packssdw.512
6349 // (<16 x i32> %a, <16 x i32> %b)
6350 // Note: AVX512 masked variants are auto-upgraded by LLVM.
6351 case Intrinsic::x86_avx512_packsswb_512:
6352 case Intrinsic::x86_avx512_packssdw_512:
6353 case Intrinsic::x86_avx512_packuswb_512:
6354 case Intrinsic::x86_avx512_packusdw_512:
6355 handleVectorPackIntrinsic(I);
6356 break;
6357
6358 case Intrinsic::x86_sse41_pblendvb:
6359 case Intrinsic::x86_sse41_blendvpd:
6360 case Intrinsic::x86_sse41_blendvps:
6361 case Intrinsic::x86_avx_blendv_pd_256:
6362 case Intrinsic::x86_avx_blendv_ps_256:
6363 case Intrinsic::x86_avx2_pblendvb:
6364 handleBlendvIntrinsic(I);
6365 break;
6366
6367 case Intrinsic::x86_avx_dp_ps_256:
6368 case Intrinsic::x86_sse41_dppd:
6369 case Intrinsic::x86_sse41_dpps:
6370 handleDppIntrinsic(I);
6371 break;
6372
6373 case Intrinsic::x86_mmx_packsswb:
6374 case Intrinsic::x86_mmx_packuswb:
6375 handleVectorPackIntrinsic(I, MMXEltSizeInBits: 16);
6376 break;
6377
6378 case Intrinsic::x86_mmx_packssdw:
6379 handleVectorPackIntrinsic(I, MMXEltSizeInBits: 32);
6380 break;
6381
6382 case Intrinsic::x86_mmx_psad_bw:
6383 handleVectorSadIntrinsic(I, IsMMX: true);
6384 break;
6385 case Intrinsic::x86_sse2_psad_bw:
6386 case Intrinsic::x86_avx2_psad_bw:
6387 handleVectorSadIntrinsic(I);
6388 break;
6389
6390 // Multiply and Add Packed Words
6391 // < 4 x i32> @llvm.x86.sse2.pmadd.wd(<8 x i16>, <8 x i16>)
6392 // < 8 x i32> @llvm.x86.avx2.pmadd.wd(<16 x i16>, <16 x i16>)
6393 // <16 x i32> @llvm.x86.avx512.pmaddw.d.512(<32 x i16>, <32 x i16>)
6394 //
6395 // Multiply and Add Packed Signed and Unsigned Bytes
6396 // < 8 x i16> @llvm.x86.ssse3.pmadd.ub.sw.128(<16 x i8>, <16 x i8>)
6397 // <16 x i16> @llvm.x86.avx2.pmadd.ub.sw(<32 x i8>, <32 x i8>)
6398 // <32 x i16> @llvm.x86.avx512.pmaddubs.w.512(<64 x i8>, <64 x i8>)
6399 //
6400 // These intrinsics are auto-upgraded into non-masked forms:
6401 // < 4 x i32> @llvm.x86.avx512.mask.pmaddw.d.128
6402 // (<8 x i16>, <8 x i16>, <4 x i32>, i8)
6403 // < 8 x i32> @llvm.x86.avx512.mask.pmaddw.d.256
6404 // (<16 x i16>, <16 x i16>, <8 x i32>, i8)
6405 // <16 x i32> @llvm.x86.avx512.mask.pmaddw.d.512
6406 // (<32 x i16>, <32 x i16>, <16 x i32>, i16)
6407 // < 8 x i16> @llvm.x86.avx512.mask.pmaddubs.w.128
6408 // (<16 x i8>, <16 x i8>, <8 x i16>, i8)
6409 // <16 x i16> @llvm.x86.avx512.mask.pmaddubs.w.256
6410 // (<32 x i8>, <32 x i8>, <16 x i16>, i16)
6411 // <32 x i16> @llvm.x86.avx512.mask.pmaddubs.w.512
6412 // (<64 x i8>, <64 x i8>, <32 x i16>, i32)
6413 case Intrinsic::x86_sse2_pmadd_wd:
6414 case Intrinsic::x86_avx2_pmadd_wd:
6415 case Intrinsic::x86_avx512_pmaddw_d_512:
6416 case Intrinsic::x86_ssse3_pmadd_ub_sw_128:
6417 case Intrinsic::x86_avx2_pmadd_ub_sw:
6418 case Intrinsic::x86_avx512_pmaddubs_w_512:
6419 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
6420 /*ZeroPurifies=*/true,
6421 /*EltSizeInBits=*/0,
6422 /*Lanes=*/kBothLanes);
6423 break;
6424
6425 // <1 x i64> @llvm.x86.ssse3.pmadd.ub.sw(<1 x i64>, <1 x i64>)
6426 case Intrinsic::x86_ssse3_pmadd_ub_sw:
6427 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
6428 /*ZeroPurifies=*/true,
6429 /*EltSizeInBits=*/8,
6430 /*Lanes=*/kBothLanes);
6431 break;
6432
6433 // <1 x i64> @llvm.x86.mmx.pmadd.wd(<1 x i64>, <1 x i64>)
6434 case Intrinsic::x86_mmx_pmadd_wd:
6435 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
6436 /*ZeroPurifies=*/true,
6437 /*EltSizeInBits=*/16,
6438 /*Lanes=*/kBothLanes);
6439 break;
6440
6441 // BFloat16 multiply-add to single-precision
6442 // <4 x float> llvm.aarch64.neon.bfmlalt
6443 // (<4 x float>, <8 x bfloat>, <8 x bfloat>)
6444 case Intrinsic::aarch64_neon_bfmlalt:
6445 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
6446 /*ZeroPurifies=*/false,
6447 /*EltSizeInBits=*/0,
6448 /*Lanes=*/kOddLanes);
6449 break;
6450
6451 // <4 x float> llvm.aarch64.neon.bfmlalb
6452 // (<4 x float>, <8 x bfloat>, <8 x bfloat>)
6453 case Intrinsic::aarch64_neon_bfmlalb:
6454 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
6455 /*ZeroPurifies=*/false,
6456 /*EltSizeInBits=*/0,
6457 /*Lanes=*/kEvenLanes);
6458 break;
6459
6460 // AVX Vector Neural Network Instructions: bytes
6461 //
6462 // Multiply and Add Signed Bytes
6463 // < 4 x i32> @llvm.x86.avx2.vpdpbssd.128
6464 // (< 4 x i32>, <16 x i8>, <16 x i8>)
6465 // < 8 x i32> @llvm.x86.avx2.vpdpbssd.256
6466 // (< 8 x i32>, <32 x i8>, <32 x i8>)
6467 // <16 x i32> @llvm.x86.avx10.vpdpbssd.512
6468 // (<16 x i32>, <64 x i8>, <64 x i8>)
6469 //
6470 // Multiply and Add Signed Bytes With Saturation
6471 // < 4 x i32> @llvm.x86.avx2.vpdpbssds.128
6472 // (< 4 x i32>, <16 x i8>, <16 x i8>)
6473 // < 8 x i32> @llvm.x86.avx2.vpdpbssds.256
6474 // (< 8 x i32>, <32 x i8>, <32 x i8>)
6475 // <16 x i32> @llvm.x86.avx10.vpdpbssds.512
6476 // (<16 x i32>, <64 x i8>, <64 x i8>)
6477 //
6478 // Multiply and Add Signed and Unsigned Bytes
6479 // < 4 x i32> @llvm.x86.avx2.vpdpbsud.128
6480 // (< 4 x i32>, <16 x i8>, <16 x i8>)
6481 // < 8 x i32> @llvm.x86.avx2.vpdpbsud.256
6482 // (< 8 x i32>, <32 x i8>, <32 x i8>)
6483 // <16 x i32> @llvm.x86.avx10.vpdpbsud.512
6484 // (<16 x i32>, <64 x i8>, <64 x i8>)
6485 //
6486 // Multiply and Add Signed and Unsigned Bytes With Saturation
6487 // < 4 x i32> @llvm.x86.avx2.vpdpbsuds.128
6488 // (< 4 x i32>, <16 x i8>, <16 x i8>)
6489 // < 8 x i32> @llvm.x86.avx2.vpdpbsuds.256
6490 // (< 8 x i32>, <32 x i8>, <32 x i8>)
6491 // <16 x i32> @llvm.x86.avx512.vpdpbusds.512
6492 // (<16 x i32>, <64 x i8>, <64 x i8>)
6493 //
6494 // Multiply and Add Unsigned and Signed Bytes
6495 // < 4 x i32> @llvm.x86.avx512.vpdpbusd.128
6496 // (< 4 x i32>, <16 x i8>, <16 x i8>)
6497 // < 8 x i32> @llvm.x86.avx512.vpdpbusd.256
6498 // (< 8 x i32>, <32 x i8>, <32 x i8>)
6499 // <16 x i32> @llvm.x86.avx512.vpdpbusd.512
6500 // (<16 x i32>, <64 x i8>, <64 x i8>)
6501 //
6502 // Multiply and Add Unsigned and Signed Bytes With Saturation
6503 // < 4 x i32> @llvm.x86.avx512.vpdpbusds.128
6504 // (< 4 x i32>, <16 x i8>, <16 x i8>)
6505 // < 8 x i32> @llvm.x86.avx512.vpdpbusds.256
6506 // (< 8 x i32>, <32 x i8>, <32 x i8>)
6507 // <16 x i32> @llvm.x86.avx10.vpdpbsuds.512
6508 // (<16 x i32>, <64 x i8>, <64 x i8>)
6509 //
6510 // Multiply and Add Unsigned Bytes
6511 // < 4 x i32> @llvm.x86.avx2.vpdpbuud.128
6512 // (< 4 x i32>, <16 x i8>, <16 x i8>)
6513 // < 8 x i32> @llvm.x86.avx2.vpdpbuud.256
6514 // (< 8 x i32>, <32 x i8>, <32 x i8>)
6515 // <16 x i32> @llvm.x86.avx10.vpdpbuud.512
6516 // (<16 x i32>, <64 x i8>, <64 x i8>)
6517 //
6518 // Multiply and Add Unsigned Bytes With Saturation
6519 // < 4 x i32> @llvm.x86.avx2.vpdpbuuds.128
6520 // (< 4 x i32>, <16 x i8>, <16 x i8>)
6521 // < 8 x i32> @llvm.x86.avx2.vpdpbuuds.256
6522 // (< 8 x i32>, <32 x i8>, <32 x i8>)
6523 // <16 x i32> @llvm.x86.avx10.vpdpbuuds.512
6524 // (<16 x i32>, <64 x i8>, <64 x i8>)
6525 //
6526 // These intrinsics are auto-upgraded into non-masked forms:
6527 // <4 x i32> @llvm.x86.avx512.mask.vpdpbusd.128
6528 // (<4 x i32>, <16 x i8>, <16 x i8>, i8)
6529 // <4 x i32> @llvm.x86.avx512.maskz.vpdpbusd.128
6530 // (<4 x i32>, <16 x i8>, <16 x i8>, i8)
6531 // <8 x i32> @llvm.x86.avx512.mask.vpdpbusd.256
6532 // (<8 x i32>, <32 x i8>, <32 x i8>, i8)
6533 // <8 x i32> @llvm.x86.avx512.maskz.vpdpbusd.256
6534 // (<8 x i32>, <32 x i8>, <32 x i8>, i8)
6535 // <16 x i32> @llvm.x86.avx512.mask.vpdpbusd.512
6536 // (<16 x i32>, <64 x i8>, <64 x i8>, i16)
6537 // <16 x i32> @llvm.x86.avx512.maskz.vpdpbusd.512
6538 // (<16 x i32>, <64 x i8>, <64 x i8>, i16)
6539 //
6540 // <4 x i32> @llvm.x86.avx512.mask.vpdpbusds.128
6541 // (<4 x i32>, <16 x i8>, <16 x i8>, i8)
6542 // <4 x i32> @llvm.x86.avx512.maskz.vpdpbusds.128
6543 // (<4 x i32>, <16 x i8>, <16 x i8>, i8)
6544 // <8 x i32> @llvm.x86.avx512.mask.vpdpbusds.256
6545 // (<8 x i32>, <32 x i8>, <32 x i8>, i8)
6546 // <8 x i32> @llvm.x86.avx512.maskz.vpdpbusds.256
6547 // (<8 x i32>, <32 x i8>, <32 x i8>, i8)
6548 // <16 x i32> @llvm.x86.avx512.mask.vpdpbusds.512
6549 // (<16 x i32>, <64 x i8>, <64 x i8>, i16)
6550 // <16 x i32> @llvm.x86.avx512.maskz.vpdpbusds.512
6551 // (<16 x i32>, <64 x i8>, <64 x i8>, i16)
6552 case Intrinsic::x86_avx512_vpdpbusd_128:
6553 case Intrinsic::x86_avx512_vpdpbusd_256:
6554 case Intrinsic::x86_avx512_vpdpbusd_512:
6555 case Intrinsic::x86_avx512_vpdpbusds_128:
6556 case Intrinsic::x86_avx512_vpdpbusds_256:
6557 case Intrinsic::x86_avx512_vpdpbusds_512:
6558 case Intrinsic::x86_avx2_vpdpbssd_128:
6559 case Intrinsic::x86_avx2_vpdpbssd_256:
6560 case Intrinsic::x86_avx10_vpdpbssd_512:
6561 case Intrinsic::x86_avx2_vpdpbssds_128:
6562 case Intrinsic::x86_avx2_vpdpbssds_256:
6563 case Intrinsic::x86_avx10_vpdpbssds_512:
6564 case Intrinsic::x86_avx2_vpdpbsud_128:
6565 case Intrinsic::x86_avx2_vpdpbsud_256:
6566 case Intrinsic::x86_avx10_vpdpbsud_512:
6567 case Intrinsic::x86_avx2_vpdpbsuds_128:
6568 case Intrinsic::x86_avx2_vpdpbsuds_256:
6569 case Intrinsic::x86_avx10_vpdpbsuds_512:
6570 case Intrinsic::x86_avx2_vpdpbuud_128:
6571 case Intrinsic::x86_avx2_vpdpbuud_256:
6572 case Intrinsic::x86_avx10_vpdpbuud_512:
6573 case Intrinsic::x86_avx2_vpdpbuuds_128:
6574 case Intrinsic::x86_avx2_vpdpbuuds_256:
6575 case Intrinsic::x86_avx10_vpdpbuuds_512:
6576 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/4,
6577 /*ZeroPurifies=*/true,
6578 /*EltSizeInBits=*/0,
6579 /*Lanes=*/kBothLanes);
6580 break;
6581
6582 // AVX Vector Neural Network Instructions: words
6583 //
6584 // Multiply and Add Signed Word Integers
6585 // < 4 x i32> @llvm.x86.avx512.vpdpwssd.128
6586 // (< 4 x i32>, < 8 x i16>, < 8 x i16>)
6587 // < 8 x i32> @llvm.x86.avx512.vpdpwssd.256
6588 // (< 8 x i32>, <16 x i16>, <16 x i16>)
6589 // <16 x i32> @llvm.x86.avx512.vpdpwssd.512
6590 // (<16 x i32>, <32 x i16>, <32 x i16>)
6591 //
6592 // Multiply and Add Signed Word Integers With Saturation
6593 // < 4 x i32> @llvm.x86.avx512.vpdpwssds.128
6594 // (< 4 x i32>, < 8 x i16>, < 8 x i16>)
6595 // < 8 x i32> @llvm.x86.avx512.vpdpwssds.256
6596 // (< 8 x i32>, <16 x i16>, <16 x i16>)
6597 // <16 x i32> @llvm.x86.avx512.vpdpwssds.512
6598 // (<16 x i32>, <32 x i16>, <32 x i16>)
6599 //
6600 // Multiply and Add Signed and Unsigned Word Integers
6601 // < 4 x i32> @llvm.x86.avx2.vpdpwsud.128
6602 // (< 4 x i32>, < 8 x i16>, < 8 x i16>)
6603 // < 8 x i32> @llvm.x86.avx2.vpdpwsud.256
6604 // (< 8 x i32>, <16 x i16>, <16 x i16>)
6605 // <16 x i32> @llvm.x86.avx10.vpdpwsud.512
6606 // (<16 x i32>, <32 x i16>, <32 x i16>)
6607 //
6608 // Multiply and Add Signed and Unsigned Word Integers With Saturation
6609 // < 4 x i32> @llvm.x86.avx2.vpdpwsuds.128
6610 // (< 4 x i32>, < 8 x i16>, < 8 x i16>)
6611 // < 8 x i32> @llvm.x86.avx2.vpdpwsuds.256
6612 // (< 8 x i32>, <16 x i16>, <16 x i16>)
6613 // <16 x i32> @llvm.x86.avx10.vpdpwsuds.512
6614 // (<16 x i32>, <32 x i16>, <32 x i16>)
6615 //
6616 // Multiply and Add Unsigned and Signed Word Integers
6617 // < 4 x i32> @llvm.x86.avx2.vpdpwusd.128
6618 // (< 4 x i32>, < 8 x i16>, < 8 x i16>)
6619 // < 8 x i32> @llvm.x86.avx2.vpdpwusd.256
6620 // (< 8 x i32>, <16 x i16>, <16 x i16>)
6621 // <16 x i32> @llvm.x86.avx10.vpdpwusd.512
6622 // (<16 x i32>, <32 x i16>, <32 x i16>)
6623 //
6624 // Multiply and Add Unsigned and Signed Word Integers With Saturation
6625 // < 4 x i32> @llvm.x86.avx2.vpdpwusds.128
6626 // (< 4 x i32>, < 8 x i16>, < 8 x i16>)
6627 // < 8 x i32> @llvm.x86.avx2.vpdpwusds.256
6628 // (< 8 x i32>, <16 x i16>, <16 x i16>)
6629 // <16 x i32> @llvm.x86.avx10.vpdpwusds.512
6630 // (<16 x i32>, <32 x i16>, <32 x i16>)
6631 //
6632 // Multiply and Add Unsigned and Unsigned Word Integers
6633 // < 4 x i32> @llvm.x86.avx2.vpdpwuud.128
6634 // (< 4 x i32>, < 8 x i16>, < 8 x i16>)
6635 // < 8 x i32> @llvm.x86.avx2.vpdpwuud.256
6636 // (< 8 x i32>, <16 x i16>, <16 x i16>)
6637 // <16 x i32> @llvm.x86.avx10.vpdpwuud.512
6638 // (<16 x i32>, <32 x i16>, <32 x i16>)
6639 //
6640 // Multiply and Add Unsigned and Unsigned Word Integers With Saturation
6641 // < 4 x i32> @llvm.x86.avx2.vpdpwuuds.128
6642 // (< 4 x i32>, < 8 x i16>, < 8 x i16>)
6643 // < 8 x i32> @llvm.x86.avx2.vpdpwuuds.256
6644 // (< 8 x i32>, <16 x i16>, <16 x i16>)
6645 // <16 x i32> @llvm.x86.avx10.vpdpwuuds.512
6646 // (<16 x i32>, <32 x i16>, <32 x i16>)
6647 //
6648 // These intrinsics are auto-upgraded into non-masked forms:
6649 // <4 x i32> @llvm.x86.avx512.mask.vpdpwssd.128
6650 // (<4 x i32>, <8 x i16>, <8 x i16>, i8)
6651 // <4 x i32> @llvm.x86.avx512.maskz.vpdpwssd.128
6652 // (<4 x i32>, <8 x i16>, <8 x i16>, i8)
6653 // <8 x i32> @llvm.x86.avx512.mask.vpdpwssd.256
6654 // (<8 x i32>, <16 x i16>, <16 x i16>, i8)
6655 // <8 x i32> @llvm.x86.avx512.maskz.vpdpwssd.256
6656 // (<8 x i32>, <16 x i16>, <16 x i16>, i8)
6657 // <16 x i32> @llvm.x86.avx512.mask.vpdpwssd.512
6658 // (<16 x i32>, <32 x i16>, <32 x i16>, i16)
6659 // <16 x i32> @llvm.x86.avx512.maskz.vpdpwssd.512
6660 // (<16 x i32>, <32 x i16>, <32 x i16>, i16)
6661 //
6662 // <4 x i32> @llvm.x86.avx512.mask.vpdpwssds.128
6663 // (<4 x i32>, <8 x i16>, <8 x i16>, i8)
6664 // <4 x i32> @llvm.x86.avx512.maskz.vpdpwssds.128
6665 // (<4 x i32>, <8 x i16>, <8 x i16>, i8)
6666 // <8 x i32> @llvm.x86.avx512.mask.vpdpwssds.256
6667 // (<8 x i32>, <16 x i16>, <16 x i16>, i8)
6668 // <8 x i32> @llvm.x86.avx512.maskz.vpdpwssds.256
6669 // (<8 x i32>, <16 x i16>, <16 x i16>, i8)
6670 // <16 x i32> @llvm.x86.avx512.mask.vpdpwssds.512
6671 // (<16 x i32>, <32 x i16>, <32 x i16>, i16)
6672 // <16 x i32> @llvm.x86.avx512.maskz.vpdpwssds.512
6673 // (<16 x i32>, <32 x i16>, <32 x i16>, i16)
6674 case Intrinsic::x86_avx512_vpdpwssd_128:
6675 case Intrinsic::x86_avx512_vpdpwssd_256:
6676 case Intrinsic::x86_avx512_vpdpwssd_512:
6677 case Intrinsic::x86_avx512_vpdpwssds_128:
6678 case Intrinsic::x86_avx512_vpdpwssds_256:
6679 case Intrinsic::x86_avx512_vpdpwssds_512:
6680 case Intrinsic::x86_avx2_vpdpwsud_128:
6681 case Intrinsic::x86_avx2_vpdpwsud_256:
6682 case Intrinsic::x86_avx10_vpdpwsud_512:
6683 case Intrinsic::x86_avx2_vpdpwsuds_128:
6684 case Intrinsic::x86_avx2_vpdpwsuds_256:
6685 case Intrinsic::x86_avx10_vpdpwsuds_512:
6686 case Intrinsic::x86_avx2_vpdpwusd_128:
6687 case Intrinsic::x86_avx2_vpdpwusd_256:
6688 case Intrinsic::x86_avx10_vpdpwusd_512:
6689 case Intrinsic::x86_avx2_vpdpwusds_128:
6690 case Intrinsic::x86_avx2_vpdpwusds_256:
6691 case Intrinsic::x86_avx10_vpdpwusds_512:
6692 case Intrinsic::x86_avx2_vpdpwuud_128:
6693 case Intrinsic::x86_avx2_vpdpwuud_256:
6694 case Intrinsic::x86_avx10_vpdpwuud_512:
6695 case Intrinsic::x86_avx2_vpdpwuuds_128:
6696 case Intrinsic::x86_avx2_vpdpwuuds_256:
6697 case Intrinsic::x86_avx10_vpdpwuuds_512:
6698 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
6699 /*ZeroPurifies=*/true,
6700 /*EltSizeInBits=*/0,
6701 /*Lanes=*/kBothLanes);
6702 break;
6703
6704 // Dot Product of BF16 Pairs Accumulated Into Packed Single
6705 // Precision
6706 // <4 x float> @llvm.x86.avx512bf16.dpbf16ps.128
6707 // (<4 x float>, <8 x bfloat>, <8 x bfloat>)
6708 // <8 x float> @llvm.x86.avx512bf16.dpbf16ps.256
6709 // (<8 x float>, <16 x bfloat>, <16 x bfloat>)
6710 // <16 x float> @llvm.x86.avx512bf16.dpbf16ps.512
6711 // (<16 x float>, <32 x bfloat>, <32 x bfloat>)
6712 case Intrinsic::x86_avx512bf16_dpbf16ps_128:
6713 case Intrinsic::x86_avx512bf16_dpbf16ps_256:
6714 case Intrinsic::x86_avx512bf16_dpbf16ps_512:
6715 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
6716 /*ZeroPurifies=*/false,
6717 /*EltSizeInBits=*/0,
6718 /*Lanes=*/kBothLanes);
6719 break;
6720
6721 case Intrinsic::x86_sse_cmp_ss:
6722 case Intrinsic::x86_sse2_cmp_sd:
6723 case Intrinsic::x86_sse_comieq_ss:
6724 case Intrinsic::x86_sse_comilt_ss:
6725 case Intrinsic::x86_sse_comile_ss:
6726 case Intrinsic::x86_sse_comigt_ss:
6727 case Intrinsic::x86_sse_comige_ss:
6728 case Intrinsic::x86_sse_comineq_ss:
6729 case Intrinsic::x86_sse_ucomieq_ss:
6730 case Intrinsic::x86_sse_ucomilt_ss:
6731 case Intrinsic::x86_sse_ucomile_ss:
6732 case Intrinsic::x86_sse_ucomigt_ss:
6733 case Intrinsic::x86_sse_ucomige_ss:
6734 case Intrinsic::x86_sse_ucomineq_ss:
6735 case Intrinsic::x86_sse2_comieq_sd:
6736 case Intrinsic::x86_sse2_comilt_sd:
6737 case Intrinsic::x86_sse2_comile_sd:
6738 case Intrinsic::x86_sse2_comigt_sd:
6739 case Intrinsic::x86_sse2_comige_sd:
6740 case Intrinsic::x86_sse2_comineq_sd:
6741 case Intrinsic::x86_sse2_ucomieq_sd:
6742 case Intrinsic::x86_sse2_ucomilt_sd:
6743 case Intrinsic::x86_sse2_ucomile_sd:
6744 case Intrinsic::x86_sse2_ucomigt_sd:
6745 case Intrinsic::x86_sse2_ucomige_sd:
6746 case Intrinsic::x86_sse2_ucomineq_sd:
6747 handleVectorCompareScalarIntrinsic(I);
6748 break;
6749
6750 case Intrinsic::x86_avx_cmp_pd_256:
6751 case Intrinsic::x86_avx_cmp_ps_256:
6752 case Intrinsic::x86_sse2_cmp_pd:
6753 case Intrinsic::x86_sse_cmp_ps:
6754 handleVectorComparePackedIntrinsic(I, /*PredicateAsOperand=*/true);
6755 break;
6756
6757 case Intrinsic::x86_bmi_bextr_32:
6758 case Intrinsic::x86_bmi_bextr_64:
6759 case Intrinsic::x86_bmi_bzhi_32:
6760 case Intrinsic::x86_bmi_bzhi_64:
6761 handleGenericBitManipulation(I);
6762 break;
6763
6764 case Intrinsic::x86_pclmulqdq:
6765 case Intrinsic::x86_pclmulqdq_256:
6766 case Intrinsic::x86_pclmulqdq_512:
6767 handlePclmulIntrinsic(I);
6768 break;
6769
6770 case Intrinsic::x86_avx_round_pd_256:
6771 case Intrinsic::x86_avx_round_ps_256:
6772 case Intrinsic::x86_sse41_round_pd:
6773 case Intrinsic::x86_sse41_round_ps:
6774 handleRoundPdPsIntrinsic(I);
6775 break;
6776
6777 case Intrinsic::x86_sse41_round_sd:
6778 case Intrinsic::x86_sse41_round_ss:
6779 handleUnarySdSsIntrinsic(I);
6780 break;
6781
6782 case Intrinsic::x86_sse2_max_sd:
6783 case Intrinsic::x86_sse_max_ss:
6784 case Intrinsic::x86_sse2_min_sd:
6785 case Intrinsic::x86_sse_min_ss:
6786 handleBinarySdSsIntrinsic(I);
6787 break;
6788
6789 case Intrinsic::x86_avx_vtestc_pd:
6790 case Intrinsic::x86_avx_vtestc_pd_256:
6791 case Intrinsic::x86_avx_vtestc_ps:
6792 case Intrinsic::x86_avx_vtestc_ps_256:
6793 case Intrinsic::x86_avx_vtestnzc_pd:
6794 case Intrinsic::x86_avx_vtestnzc_pd_256:
6795 case Intrinsic::x86_avx_vtestnzc_ps:
6796 case Intrinsic::x86_avx_vtestnzc_ps_256:
6797 case Intrinsic::x86_avx_vtestz_pd:
6798 case Intrinsic::x86_avx_vtestz_pd_256:
6799 case Intrinsic::x86_avx_vtestz_ps:
6800 case Intrinsic::x86_avx_vtestz_ps_256:
6801 case Intrinsic::x86_avx_ptestc_256:
6802 case Intrinsic::x86_avx_ptestnzc_256:
6803 case Intrinsic::x86_avx_ptestz_256:
6804 case Intrinsic::x86_sse41_ptestc:
6805 case Intrinsic::x86_sse41_ptestnzc:
6806 case Intrinsic::x86_sse41_ptestz:
6807 handleVtestIntrinsic(I);
6808 break;
6809
6810 // Packed Horizontal Add/Subtract
6811 case Intrinsic::x86_ssse3_phadd_w:
6812 case Intrinsic::x86_ssse3_phadd_w_128:
6813 case Intrinsic::x86_ssse3_phsub_w:
6814 case Intrinsic::x86_ssse3_phsub_w_128:
6815 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/1,
6816 /*ReinterpretElemWidth=*/16);
6817 break;
6818
6819 case Intrinsic::x86_avx2_phadd_w:
6820 case Intrinsic::x86_avx2_phsub_w:
6821 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/2,
6822 /*ReinterpretElemWidth=*/16);
6823 break;
6824
6825 // Packed Horizontal Add/Subtract
6826 case Intrinsic::x86_ssse3_phadd_d:
6827 case Intrinsic::x86_ssse3_phadd_d_128:
6828 case Intrinsic::x86_ssse3_phsub_d:
6829 case Intrinsic::x86_ssse3_phsub_d_128:
6830 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/1,
6831 /*ReinterpretElemWidth=*/32);
6832 break;
6833
6834 case Intrinsic::x86_avx2_phadd_d:
6835 case Intrinsic::x86_avx2_phsub_d:
6836 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/2,
6837 /*ReinterpretElemWidth=*/32);
6838 break;
6839
6840 // Packed Horizontal Add/Subtract and Saturate
6841 case Intrinsic::x86_ssse3_phadd_sw:
6842 case Intrinsic::x86_ssse3_phadd_sw_128:
6843 case Intrinsic::x86_ssse3_phsub_sw:
6844 case Intrinsic::x86_ssse3_phsub_sw_128:
6845 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/1,
6846 /*ReinterpretElemWidth=*/16);
6847 break;
6848
6849 case Intrinsic::x86_avx2_phadd_sw:
6850 case Intrinsic::x86_avx2_phsub_sw:
6851 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/2,
6852 /*ReinterpretElemWidth=*/16);
6853 break;
6854
6855 // Packed Single/Double Precision Floating-Point Horizontal Add
6856 case Intrinsic::x86_sse3_hadd_ps:
6857 case Intrinsic::x86_sse3_hadd_pd:
6858 case Intrinsic::x86_sse3_hsub_ps:
6859 case Intrinsic::x86_sse3_hsub_pd:
6860 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/1);
6861 break;
6862
6863 case Intrinsic::x86_avx_hadd_pd_256:
6864 case Intrinsic::x86_avx_hadd_ps_256:
6865 case Intrinsic::x86_avx_hsub_pd_256:
6866 case Intrinsic::x86_avx_hsub_ps_256:
6867 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/2);
6868 break;
6869
6870 case Intrinsic::x86_avx_maskstore_ps:
6871 case Intrinsic::x86_avx_maskstore_pd:
6872 case Intrinsic::x86_avx_maskstore_ps_256:
6873 case Intrinsic::x86_avx_maskstore_pd_256:
6874 case Intrinsic::x86_avx2_maskstore_d:
6875 case Intrinsic::x86_avx2_maskstore_q:
6876 case Intrinsic::x86_avx2_maskstore_d_256:
6877 case Intrinsic::x86_avx2_maskstore_q_256: {
6878 handleAVXMaskedStore(I);
6879 break;
6880 }
6881
6882 case Intrinsic::x86_avx_maskload_ps:
6883 case Intrinsic::x86_avx_maskload_pd:
6884 case Intrinsic::x86_avx_maskload_ps_256:
6885 case Intrinsic::x86_avx_maskload_pd_256:
6886 case Intrinsic::x86_avx2_maskload_d:
6887 case Intrinsic::x86_avx2_maskload_q:
6888 case Intrinsic::x86_avx2_maskload_d_256:
6889 case Intrinsic::x86_avx2_maskload_q_256: {
6890 handleAVXMaskedLoad(I);
6891 break;
6892 }
6893
6894 // Packed
6895 case Intrinsic::x86_avx512fp16_add_ph_512:
6896 case Intrinsic::x86_avx512fp16_sub_ph_512:
6897 case Intrinsic::x86_avx512fp16_mul_ph_512:
6898 case Intrinsic::x86_avx512fp16_div_ph_512:
6899 case Intrinsic::x86_avx512fp16_max_ph_512:
6900 case Intrinsic::x86_avx512fp16_min_ph_512:
6901 case Intrinsic::x86_avx512_min_ps_512:
6902 case Intrinsic::x86_avx512_min_pd_512:
6903 case Intrinsic::x86_avx512_max_ps_512:
6904 case Intrinsic::x86_avx512_max_pd_512: {
6905 // These AVX512 variants contain the rounding mode as a trailing flag.
6906 // Earlier variants do not have a trailing flag and are already handled
6907 // by maybeHandleSimpleNomemIntrinsic(I, 0) via
6908 // maybeHandleUnknownIntrinsic.
6909 [[maybe_unused]] bool Success =
6910 maybeHandleSimpleNomemIntrinsic(I, /*trailingFlags=*/1);
6911 assert(Success);
6912 break;
6913 }
6914
6915 case Intrinsic::x86_avx_vpermilvar_pd:
6916 case Intrinsic::x86_avx_vpermilvar_pd_256:
6917 case Intrinsic::x86_avx512_vpermilvar_pd_512:
6918 case Intrinsic::x86_avx_vpermilvar_ps:
6919 case Intrinsic::x86_avx_vpermilvar_ps_256:
6920 case Intrinsic::x86_avx512_vpermilvar_ps_512: {
6921 handleAVXVpermilvar(I);
6922 break;
6923 }
6924
6925 case Intrinsic::x86_avx512_vpermi2var_d_128:
6926 case Intrinsic::x86_avx512_vpermi2var_d_256:
6927 case Intrinsic::x86_avx512_vpermi2var_d_512:
6928 case Intrinsic::x86_avx512_vpermi2var_hi_128:
6929 case Intrinsic::x86_avx512_vpermi2var_hi_256:
6930 case Intrinsic::x86_avx512_vpermi2var_hi_512:
6931 case Intrinsic::x86_avx512_vpermi2var_pd_128:
6932 case Intrinsic::x86_avx512_vpermi2var_pd_256:
6933 case Intrinsic::x86_avx512_vpermi2var_pd_512:
6934 case Intrinsic::x86_avx512_vpermi2var_ps_128:
6935 case Intrinsic::x86_avx512_vpermi2var_ps_256:
6936 case Intrinsic::x86_avx512_vpermi2var_ps_512:
6937 case Intrinsic::x86_avx512_vpermi2var_q_128:
6938 case Intrinsic::x86_avx512_vpermi2var_q_256:
6939 case Intrinsic::x86_avx512_vpermi2var_q_512:
6940 case Intrinsic::x86_avx512_vpermi2var_qi_128:
6941 case Intrinsic::x86_avx512_vpermi2var_qi_256:
6942 case Intrinsic::x86_avx512_vpermi2var_qi_512:
6943 handleAVXVpermi2var(I);
6944 break;
6945
6946 // Packed Shuffle
6947 // llvm.x86.sse.pshuf.w(<1 x i64>, i8)
6948 // llvm.x86.ssse3.pshuf.b(<1 x i64>, <1 x i64>)
6949 // llvm.x86.ssse3.pshuf.b.128(<16 x i8>, <16 x i8>)
6950 // llvm.x86.avx2.pshuf.b(<32 x i8>, <32 x i8>)
6951 // llvm.x86.avx512.pshuf.b.512(<64 x i8>, <64 x i8>)
6952 //
6953 // The following intrinsics are auto-upgraded:
6954 // llvm.x86.sse2.pshuf.d(<4 x i32>, i8)
6955 // llvm.x86.sse2.gpshufh.w(<8 x i16>, i8)
6956 // llvm.x86.sse2.pshufl.w(<8 x i16>, i8)
6957 case Intrinsic::x86_avx2_pshuf_b:
6958 case Intrinsic::x86_sse_pshuf_w:
6959 case Intrinsic::x86_ssse3_pshuf_b_128:
6960 case Intrinsic::x86_ssse3_pshuf_b:
6961 case Intrinsic::x86_avx512_pshuf_b_512:
6962 handleIntrinsicByApplyingToShadow(I, shadowIntrinsicID: I.getIntrinsicID(),
6963 /*trailingVerbatimArgs=*/1,
6964 /*forceIntegerIntrinsic=*/false);
6965 break;
6966
6967 // AVX512 PMOV: Packed MOV, with truncation
6968 // Precisely handled by applying the same intrinsic to the shadow
6969 case Intrinsic::x86_avx512_mask_pmov_dw_128:
6970 case Intrinsic::x86_avx512_mask_pmov_db_128:
6971 case Intrinsic::x86_avx512_mask_pmov_qb_128:
6972 case Intrinsic::x86_avx512_mask_pmov_qw_128:
6973 case Intrinsic::x86_avx512_mask_pmov_qd_128:
6974 case Intrinsic::x86_avx512_mask_pmov_wb_128:
6975 case Intrinsic::x86_avx512_mask_pmov_dw_256:
6976 case Intrinsic::x86_avx512_mask_pmov_db_256:
6977 case Intrinsic::x86_avx512_mask_pmov_qb_256:
6978 case Intrinsic::x86_avx512_mask_pmov_qw_256:
6979 case Intrinsic::x86_avx512_mask_pmov_dw_512:
6980 case Intrinsic::x86_avx512_mask_pmov_db_512:
6981 case Intrinsic::x86_avx512_mask_pmov_qb_512:
6982 case Intrinsic::x86_avx512_mask_pmov_qw_512: {
6983 // Intrinsic::x86_avx512_mask_pmov_{qd,wb}_{256,512} were removed in
6984 // f608dc1f5775ee880e8ea30e2d06ab5a4a935c22
6985 handleIntrinsicByApplyingToShadow(I, shadowIntrinsicID: I.getIntrinsicID(),
6986 /*trailingVerbatimArgs=*/1,
6987 /*forceIntegerIntrinsic=*/false);
6988 break;
6989 }
6990
6991 // AVX512 PMOV{S,US}: Packed MOV, with signed/unsigned saturation
6992 // Approximately handled using the corresponding truncation intrinsic
6993 // TODO: improve handleAVX512VectorDownConvert to precisely model saturation
6994 case Intrinsic::x86_avx512_mask_pmovs_dw_512:
6995 case Intrinsic::x86_avx512_mask_pmovus_dw_512: {
6996 handleIntrinsicByApplyingToShadow(
6997 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_dw_512,
6998 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
6999 break;
7000 }
7001
7002 case Intrinsic::x86_avx512_mask_pmovs_dw_256:
7003 case Intrinsic::x86_avx512_mask_pmovus_dw_256:
7004 handleIntrinsicByApplyingToShadow(
7005 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_dw_256,
7006 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7007 break;
7008
7009 case Intrinsic::x86_avx512_mask_pmovs_dw_128:
7010 case Intrinsic::x86_avx512_mask_pmovus_dw_128:
7011 handleIntrinsicByApplyingToShadow(
7012 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_dw_128,
7013 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7014 break;
7015
7016 case Intrinsic::x86_avx512_mask_pmovs_db_512:
7017 case Intrinsic::x86_avx512_mask_pmovus_db_512: {
7018 handleIntrinsicByApplyingToShadow(
7019 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_db_512,
7020 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7021 break;
7022 }
7023
7024 case Intrinsic::x86_avx512_mask_pmovs_db_256:
7025 case Intrinsic::x86_avx512_mask_pmovus_db_256:
7026 handleIntrinsicByApplyingToShadow(
7027 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_db_256,
7028 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7029 break;
7030
7031 case Intrinsic::x86_avx512_mask_pmovs_db_128:
7032 case Intrinsic::x86_avx512_mask_pmovus_db_128:
7033 handleIntrinsicByApplyingToShadow(
7034 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_db_128,
7035 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7036 break;
7037
7038 case Intrinsic::x86_avx512_mask_pmovs_qb_512:
7039 case Intrinsic::x86_avx512_mask_pmovus_qb_512: {
7040 handleIntrinsicByApplyingToShadow(
7041 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_qb_512,
7042 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7043 break;
7044 }
7045
7046 case Intrinsic::x86_avx512_mask_pmovs_qb_256:
7047 case Intrinsic::x86_avx512_mask_pmovus_qb_256:
7048 handleIntrinsicByApplyingToShadow(
7049 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_qb_256,
7050 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7051 break;
7052
7053 case Intrinsic::x86_avx512_mask_pmovs_qb_128:
7054 case Intrinsic::x86_avx512_mask_pmovus_qb_128:
7055 handleIntrinsicByApplyingToShadow(
7056 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_qb_128,
7057 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7058 break;
7059
7060 case Intrinsic::x86_avx512_mask_pmovs_qw_512:
7061 case Intrinsic::x86_avx512_mask_pmovus_qw_512: {
7062 handleIntrinsicByApplyingToShadow(
7063 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_qw_512,
7064 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7065 break;
7066 }
7067
7068 case Intrinsic::x86_avx512_mask_pmovs_qw_256:
7069 case Intrinsic::x86_avx512_mask_pmovus_qw_256:
7070 handleIntrinsicByApplyingToShadow(
7071 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_qw_256,
7072 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7073 break;
7074
7075 case Intrinsic::x86_avx512_mask_pmovs_qw_128:
7076 case Intrinsic::x86_avx512_mask_pmovus_qw_128:
7077 handleIntrinsicByApplyingToShadow(
7078 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_qw_128,
7079 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7080 break;
7081
7082 case Intrinsic::x86_avx512_mask_pmovs_qd_128:
7083 case Intrinsic::x86_avx512_mask_pmovus_qd_128:
7084 handleIntrinsicByApplyingToShadow(
7085 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_qd_128,
7086 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7087 break;
7088
7089 case Intrinsic::x86_avx512_mask_pmovs_wb_128:
7090 case Intrinsic::x86_avx512_mask_pmovus_wb_128:
7091 handleIntrinsicByApplyingToShadow(
7092 I, shadowIntrinsicID: Intrinsic::x86_avx512_mask_pmov_wb_128,
7093 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7094 break;
7095
7096 case Intrinsic::x86_avx512_mask_pmovs_qd_256:
7097 case Intrinsic::x86_avx512_mask_pmovus_qd_256:
7098 case Intrinsic::x86_avx512_mask_pmovs_wb_256:
7099 case Intrinsic::x86_avx512_mask_pmovus_wb_256:
7100 case Intrinsic::x86_avx512_mask_pmovs_qd_512:
7101 case Intrinsic::x86_avx512_mask_pmovus_qd_512:
7102 case Intrinsic::x86_avx512_mask_pmovs_wb_512:
7103 case Intrinsic::x86_avx512_mask_pmovus_wb_512: {
7104 // Since Intrinsic::x86_avx512_mask_pmov_{qd,wb}_{256,512} do not exist,
7105 // we cannot use handleIntrinsicByApplyingToShadow. Instead, we call the
7106 // slow-path handler.
7107 handleAVX512VectorDownConvert(I);
7108 break;
7109 }
7110
7111 // e.g.,
7112 // <16 x float> @llvm.x86.avx512.mask.compress
7113 // (<16 x float> %data, <16 x float> %passthru,
7114 // <16 x i1> %mask)
7115 // <16 x i32> @llvm.x86.avx512.mask.compress
7116 // (<16 x i32> %data, <16 x i32> %passthru,
7117 // <16 x i1> %mask)
7118 case Intrinsic::x86_avx512_mask_compress:
7119 handleIntrinsicByApplyingToShadow(I, shadowIntrinsicID: I.getIntrinsicID(),
7120 /*trailingVerbatimArgs=*/1,
7121 /*forceIntegerIntrinsic=*/true);
7122 break;
7123
7124 // AVX512/AVX10 Reciprocal
7125 // <16 x float> @llvm.x86.avx512.rsqrt14.ps.512
7126 // (<16 x float>, <16 x float>, i16)
7127 // <8 x float> @llvm.x86.avx512.rsqrt14.ps.256
7128 // (<8 x float>, <8 x float>, i8)
7129 // <4 x float> @llvm.x86.avx512.rsqrt14.ps.128
7130 // (<4 x float>, <4 x float>, i8)
7131 //
7132 // <8 x double> @llvm.x86.avx512.rsqrt14.pd.512
7133 // (<8 x double>, <8 x double>, i8)
7134 // <4 x double> @llvm.x86.avx512.rsqrt14.pd.256
7135 // (<4 x double>, <4 x double>, i8)
7136 // <2 x double> @llvm.x86.avx512.rsqrt14.pd.128
7137 // (<2 x double>, <2 x double>, i8)
7138 //
7139 // <32 x bfloat> @llvm.x86.avx10.mask.rsqrt.bf16.512
7140 // (<32 x bfloat>, <32 x bfloat>, i32)
7141 // <16 x bfloat> @llvm.x86.avx10.mask.rsqrt.bf16.256
7142 // (<16 x bfloat>, <16 x bfloat>, i16)
7143 // <8 x bfloat> @llvm.x86.avx10.mask.rsqrt.bf16.128
7144 // (<8 x bfloat>, <8 x bfloat>, i8)
7145 //
7146 // <32 x half> @llvm.x86.avx512fp16.mask.rsqrt.ph.512
7147 // (<32 x half>, <32 x half>, i32)
7148 // <16 x half> @llvm.x86.avx512fp16.mask.rsqrt.ph.256
7149 // (<16 x half>, <16 x half>, i16)
7150 // <8 x half> @llvm.x86.avx512fp16.mask.rsqrt.ph.128
7151 // (<8 x half>, <8 x half>, i8)
7152 //
7153 // TODO: 3-operand variants are not handled:
7154 // <2 x double> @llvm.x86.avx512.rsqrt14.sd
7155 // (<2 x double>, <2 x double>, <2 x double>, i8)
7156 // <4 x float> @llvm.x86.avx512.rsqrt14.ss
7157 // (<4 x float>, <4 x float>, <4 x float>, i8)
7158 // <8 x half> @llvm.x86.avx512fp16.mask.rsqrt.sh
7159 // (<8 x half>, <8 x half>, <8 x half>, i8)
7160 case Intrinsic::x86_avx512_rsqrt14_ps_512:
7161 case Intrinsic::x86_avx512_rsqrt14_ps_256:
7162 case Intrinsic::x86_avx512_rsqrt14_ps_128:
7163 case Intrinsic::x86_avx512_rsqrt14_pd_512:
7164 case Intrinsic::x86_avx512_rsqrt14_pd_256:
7165 case Intrinsic::x86_avx512_rsqrt14_pd_128:
7166 case Intrinsic::x86_avx10_mask_rsqrt_bf16_512:
7167 case Intrinsic::x86_avx10_mask_rsqrt_bf16_256:
7168 case Intrinsic::x86_avx10_mask_rsqrt_bf16_128:
7169 case Intrinsic::x86_avx512fp16_mask_rsqrt_ph_512:
7170 case Intrinsic::x86_avx512fp16_mask_rsqrt_ph_256:
7171 case Intrinsic::x86_avx512fp16_mask_rsqrt_ph_128:
7172 handleAVX512VectorGenericMaskedFP(I, /*DataIndices=*/{0},
7173 /*WriteThruIndex=*/1,
7174 /*MaskIndex=*/2);
7175 break;
7176
7177 // AVX512/AVX10 Reciprocal Square Root
7178 // <16 x float> @llvm.x86.avx512.rcp14.ps.512
7179 // (<16 x float>, <16 x float>, i16)
7180 // <8 x float> @llvm.x86.avx512.rcp14.ps.256
7181 // (<8 x float>, <8 x float>, i8)
7182 // <4 x float> @llvm.x86.avx512.rcp14.ps.128
7183 // (<4 x float>, <4 x float>, i8)
7184 //
7185 // <8 x double> @llvm.x86.avx512.rcp14.pd.512
7186 // (<8 x double>, <8 x double>, i8)
7187 // <4 x double> @llvm.x86.avx512.rcp14.pd.256
7188 // (<4 x double>, <4 x double>, i8)
7189 // <2 x double> @llvm.x86.avx512.rcp14.pd.128
7190 // (<2 x double>, <2 x double>, i8)
7191 //
7192 // <32 x bfloat> @llvm.x86.avx10.mask.rcp.bf16.512
7193 // (<32 x bfloat>, <32 x bfloat>, i32)
7194 // <16 x bfloat> @llvm.x86.avx10.mask.rcp.bf16.256
7195 // (<16 x bfloat>, <16 x bfloat>, i16)
7196 // <8 x bfloat> @llvm.x86.avx10.mask.rcp.bf16.128
7197 // (<8 x bfloat>, <8 x bfloat>, i8)
7198 //
7199 // <32 x half> @llvm.x86.avx512fp16.mask.rcp.ph.512
7200 // (<32 x half>, <32 x half>, i32)
7201 // <16 x half> @llvm.x86.avx512fp16.mask.rcp.ph.256
7202 // (<16 x half>, <16 x half>, i16)
7203 // <8 x half> @llvm.x86.avx512fp16.mask.rcp.ph.128
7204 // (<8 x half>, <8 x half>, i8)
7205 //
7206 // TODO: 3-operand variants are not handled:
7207 // <2 x double> @llvm.x86.avx512.rcp14.sd
7208 // (<2 x double>, <2 x double>, <2 x double>, i8)
7209 // <4 x float> @llvm.x86.avx512.rcp14.ss
7210 // (<4 x float>, <4 x float>, <4 x float>, i8)
7211 // <8 x half> @llvm.x86.avx512fp16.mask.rcp.sh
7212 // (<8 x half>, <8 x half>, <8 x half>, i8)
7213 case Intrinsic::x86_avx512_rcp14_ps_512:
7214 case Intrinsic::x86_avx512_rcp14_ps_256:
7215 case Intrinsic::x86_avx512_rcp14_ps_128:
7216 case Intrinsic::x86_avx512_rcp14_pd_512:
7217 case Intrinsic::x86_avx512_rcp14_pd_256:
7218 case Intrinsic::x86_avx512_rcp14_pd_128:
7219 case Intrinsic::x86_avx10_mask_rcp_bf16_512:
7220 case Intrinsic::x86_avx10_mask_rcp_bf16_256:
7221 case Intrinsic::x86_avx10_mask_rcp_bf16_128:
7222 case Intrinsic::x86_avx512fp16_mask_rcp_ph_512:
7223 case Intrinsic::x86_avx512fp16_mask_rcp_ph_256:
7224 case Intrinsic::x86_avx512fp16_mask_rcp_ph_128:
7225 handleAVX512VectorGenericMaskedFP(I, /*DataIndices=*/{0},
7226 /*WriteThruIndex=*/1,
7227 /*MaskIndex=*/2);
7228 break;
7229
7230 // <32 x half> @llvm.x86.avx512fp16.mask.rndscale.ph.512
7231 // (<32 x half>, i32, <32 x half>, i32, i32)
7232 // <16 x half> @llvm.x86.avx512fp16.mask.rndscale.ph.256
7233 // (<16 x half>, i32, <16 x half>, i32, i16)
7234 // <8 x half> @llvm.x86.avx512fp16.mask.rndscale.ph.128
7235 // (<8 x half>, i32, <8 x half>, i32, i8)
7236 //
7237 // <16 x float> @llvm.x86.avx512.mask.rndscale.ps.512
7238 // (<16 x float>, i32, <16 x float>, i16, i32)
7239 // <8 x float> @llvm.x86.avx512.mask.rndscale.ps.256
7240 // (<8 x float>, i32, <8 x float>, i8)
7241 // <4 x float> @llvm.x86.avx512.mask.rndscale.ps.128
7242 // (<4 x float>, i32, <4 x float>, i8)
7243 //
7244 // <8 x double> @llvm.x86.avx512.mask.rndscale.pd.512
7245 // (<8 x double>, i32, <8 x double>, i8, i32)
7246 // A Imm WriteThru Mask Rounding
7247 // <4 x double> @llvm.x86.avx512.mask.rndscale.pd.256
7248 // (<4 x double>, i32, <4 x double>, i8)
7249 // <2 x double> @llvm.x86.avx512.mask.rndscale.pd.128
7250 // (<2 x double>, i32, <2 x double>, i8)
7251 // A Imm WriteThru Mask
7252 //
7253 // <32 x bfloat> @llvm.x86.avx10.mask.rndscale.bf16.512
7254 // (<32 x bfloat>, i32, <32 x bfloat>, i32)
7255 // <16 x bfloat> @llvm.x86.avx10.mask.rndscale.bf16.256
7256 // (<16 x bfloat>, i32, <16 x bfloat>, i16)
7257 // <8 x bfloat> @llvm.x86.avx10.mask.rndscale.bf16.128
7258 // (<8 x bfloat>, i32, <8 x bfloat>, i8)
7259 //
7260 // Not supported: three vectors
7261 // - <8 x half> @llvm.x86.avx512fp16.mask.rndscale.sh
7262 // (<8 x half>, <8 x half>,<8 x half>, i8, i32, i32)
7263 // - <4 x float> @llvm.x86.avx512.mask.rndscale.ss
7264 // (<4 x float>, <4 x float>, <4 x float>, i8, i32, i32)
7265 // - <2 x double> @llvm.x86.avx512.mask.rndscale.sd
7266 // (<2 x double>, <2 x double>, <2 x double>, i8, i32,
7267 // i32)
7268 // A B WriteThru Mask Imm
7269 // Rounding
7270 case Intrinsic::x86_avx512fp16_mask_rndscale_ph_512:
7271 case Intrinsic::x86_avx512fp16_mask_rndscale_ph_256:
7272 case Intrinsic::x86_avx512fp16_mask_rndscale_ph_128:
7273 case Intrinsic::x86_avx512_mask_rndscale_ps_512:
7274 case Intrinsic::x86_avx512_mask_rndscale_ps_256:
7275 case Intrinsic::x86_avx512_mask_rndscale_ps_128:
7276 case Intrinsic::x86_avx512_mask_rndscale_pd_512:
7277 case Intrinsic::x86_avx512_mask_rndscale_pd_256:
7278 case Intrinsic::x86_avx512_mask_rndscale_pd_128:
7279 case Intrinsic::x86_avx10_mask_rndscale_bf16_512:
7280 case Intrinsic::x86_avx10_mask_rndscale_bf16_256:
7281 case Intrinsic::x86_avx10_mask_rndscale_bf16_128:
7282 handleAVX512VectorGenericMaskedFP(I, /*DataIndices=*/{0},
7283 /*WriteThruIndex=*/2,
7284 /*MaskIndex=*/3);
7285 break;
7286
7287 // AVX512 Vector Scale Float* Packed
7288 //
7289 // < 8 x double> @llvm.x86.avx512.mask.scalef.pd.512
7290 // (<8 x double>, <8 x double>, <8 x double>, i8, i32)
7291 // A B WriteThru Msk Round
7292 // < 4 x double> @llvm.x86.avx512.mask.scalef.pd.256
7293 // (<4 x double>, <4 x double>, <4 x double>, i8)
7294 // < 2 x double> @llvm.x86.avx512.mask.scalef.pd.128
7295 // (<2 x double>, <2 x double>, <2 x double>, i8)
7296 //
7297 // <16 x float> @llvm.x86.avx512.mask.scalef.ps.512
7298 // (<16 x float>, <16 x float>, <16 x float>, i16, i32)
7299 // < 8 x float> @llvm.x86.avx512.mask.scalef.ps.256
7300 // (<8 x float>, <8 x float>, <8 x float>, i8)
7301 // < 4 x float> @llvm.x86.avx512.mask.scalef.ps.128
7302 // (<4 x float>, <4 x float>, <4 x float>, i8)
7303 //
7304 // <32 x half> @llvm.x86.avx512fp16.mask.scalef.ph.512
7305 // (<32 x half>, <32 x half>, <32 x half>, i32, i32)
7306 // <16 x half> @llvm.x86.avx512fp16.mask.scalef.ph.256
7307 // (<16 x half>, <16 x half>, <16 x half>, i16)
7308 // < 8 x half> @llvm.x86.avx512fp16.mask.scalef.ph.128
7309 // (<8 x half>, <8 x half>, <8 x half>, i8)
7310 //
7311 // TODO: AVX10
7312 // <32 x bfloat> @llvm.x86.avx10.mask.scalef.bf16.512
7313 // (<32 x bfloat>, <32 x bfloat>, <32 x bfloat>, i32)
7314 // <16 x bfloat> @llvm.x86.avx10.mask.scalef.bf16.256
7315 // (<16 x bfloat>, <16 x bfloat>, <16 x bfloat>, i16)
7316 // < 8 x bfloat> @llvm.x86.avx10.mask.scalef.bf16.128
7317 // (<8 x bfloat>, <8 x bfloat>, <8 x bfloat>, i8)
7318 case Intrinsic::x86_avx512_mask_scalef_pd_512:
7319 case Intrinsic::x86_avx512_mask_scalef_pd_256:
7320 case Intrinsic::x86_avx512_mask_scalef_pd_128:
7321 case Intrinsic::x86_avx512_mask_scalef_ps_512:
7322 case Intrinsic::x86_avx512_mask_scalef_ps_256:
7323 case Intrinsic::x86_avx512_mask_scalef_ps_128:
7324 case Intrinsic::x86_avx512fp16_mask_scalef_ph_512:
7325 case Intrinsic::x86_avx512fp16_mask_scalef_ph_256:
7326 case Intrinsic::x86_avx512fp16_mask_scalef_ph_128:
7327 // The AVX512 512-bit operand variants have an extra operand (the
7328 // Rounding mode). The extra operand, if present, will be
7329 // automatically checked by the handler.
7330 handleAVX512VectorGenericMaskedFP(I, /*DataIndices=*/{0, 1},
7331 /*WriteThruIndex=*/2,
7332 /*MaskIndex=*/3);
7333 break;
7334
7335 // TODO: AVX512 Vector Scale Float* Scalar
7336 //
7337 // This is different from the Packed variant, because some bits are copied,
7338 // and some bits are zeroed.
7339 //
7340 // < 4 x float> @llvm.x86.avx512.mask.scalef.ss
7341 // (<4 x float>, <4 x float>, <4 x float>, i8, i32)
7342 //
7343 // < 2 x double> @llvm.x86.avx512.mask.scalef.sd
7344 // (<2 x double>, <2 x double>, <2 x double>, i8, i32)
7345 //
7346 // < 8 x half> @llvm.x86.avx512fp16.mask.scalef.sh
7347 // (<8 x half>, <8 x half>, <8 x half>, i8, i32)
7348
7349 // AVX512 FP16 Arithmetic
7350 case Intrinsic::x86_avx512fp16_mask_add_sh_round:
7351 case Intrinsic::x86_avx512fp16_mask_sub_sh_round:
7352 case Intrinsic::x86_avx512fp16_mask_mul_sh_round:
7353 case Intrinsic::x86_avx512fp16_mask_div_sh_round:
7354 case Intrinsic::x86_avx512fp16_mask_max_sh_round:
7355 case Intrinsic::x86_avx512fp16_mask_min_sh_round: {
7356 visitGenericScalarHalfwordInst(I);
7357 break;
7358 }
7359
7360 // AVX512 Floating-Point Classification
7361 // - <8 x i1> @llvm.x86.avx512.fpclass.pd.512(<8 x double>, i32)
7362 // - <16 x i1> @llvm.x86.avx512.fpclass.ps.512(<16 x float>, i32)
7363 case Intrinsic::x86_avx512_fpclass_pd_512:
7364 case Intrinsic::x86_avx512_fpclass_ps_512:
7365 handleAVX512FPClass(I);
7366 break;
7367
7368 // AVX Galois Field New Instructions
7369 case Intrinsic::x86_vgf2p8affineqb_128:
7370 case Intrinsic::x86_vgf2p8affineqb_256:
7371 case Intrinsic::x86_vgf2p8affineqb_512:
7372 handleAVXGF2P8Affine(I);
7373 break;
7374
7375 default:
7376 return false;
7377 }
7378
7379 return true;
7380 }
7381
7382 bool maybeHandleArmSIMDIntrinsic(IntrinsicInst &I) {
7383 switch (I.getIntrinsicID()) {
7384 // Two operands e.g.,
7385 // - <8 x i8> @llvm.aarch64.neon.rshrn.v8i8 (<8 x i16>, i32)
7386 // - <4 x i16> @llvm.aarch64.neon.uqrshl.v4i16(<4 x i16>, <4 x i16>)
7387 case Intrinsic::aarch64_neon_rshrn:
7388 case Intrinsic::aarch64_neon_sqrshl:
7389 case Intrinsic::aarch64_neon_sqrshrn:
7390 case Intrinsic::aarch64_neon_sqrshrun:
7391 case Intrinsic::aarch64_neon_sqshl:
7392 case Intrinsic::aarch64_neon_sqshlu:
7393 case Intrinsic::aarch64_neon_sqshrn:
7394 case Intrinsic::aarch64_neon_sqshrun:
7395 case Intrinsic::aarch64_neon_srshl:
7396 case Intrinsic::aarch64_neon_sshl:
7397 case Intrinsic::aarch64_neon_uqrshl:
7398 case Intrinsic::aarch64_neon_uqrshrn:
7399 case Intrinsic::aarch64_neon_uqshl:
7400 case Intrinsic::aarch64_neon_uqshrn:
7401 case Intrinsic::aarch64_neon_urshl:
7402 case Intrinsic::aarch64_neon_ushl:
7403 handleVectorShiftIntrinsic(I, /* Variable */ false);
7404 break;
7405
7406 // Vector Shift Left/Right and Insert
7407 //
7408 // Three operands e.g.,
7409 // - <4 x i16> @llvm.aarch64.neon.vsli.v4i16
7410 // (<4 x i16> %a, <4 x i16> %b, i32 %n)
7411 // - <16 x i8> @llvm.aarch64.neon.vsri.v16i8
7412 // (<16 x i8> %a, <16 x i8> %b, i32 %n)
7413 //
7414 // %b is shifted by %n bits, and the "missing" bits are filled in with %a
7415 // (instead of zero-extending/sign-extending).
7416 case Intrinsic::aarch64_neon_vsli:
7417 case Intrinsic::aarch64_neon_vsri:
7418 handleIntrinsicByApplyingToShadow(I, shadowIntrinsicID: I.getIntrinsicID(),
7419 /*trailingVerbatimArgs=*/1,
7420 /*forceIntegerIntrinsic=*/false);
7421 break;
7422
7423 // TODO: handling max/min similarly to AND/OR may be more precise
7424 // Floating-Point Maximum/Minimum Pairwise
7425 case Intrinsic::aarch64_neon_fmaxp:
7426 case Intrinsic::aarch64_neon_fminp:
7427 // Floating-Point Maximum/Minimum Number Pairwise
7428 case Intrinsic::aarch64_neon_fmaxnmp:
7429 case Intrinsic::aarch64_neon_fminnmp:
7430 // Signed/Unsigned Maximum/Minimum Pairwise
7431 case Intrinsic::aarch64_neon_smaxp:
7432 case Intrinsic::aarch64_neon_sminp:
7433 case Intrinsic::aarch64_neon_umaxp:
7434 case Intrinsic::aarch64_neon_uminp:
7435 // Add Pairwise
7436 case Intrinsic::aarch64_neon_addp:
7437 // Floating-point Add Pairwise
7438 case Intrinsic::aarch64_neon_faddp:
7439 // Add Long Pairwise
7440 case Intrinsic::aarch64_neon_saddlp:
7441 case Intrinsic::aarch64_neon_uaddlp: {
7442 handlePairwiseShadowOrIntrinsic(I, /*Shards=*/1);
7443 break;
7444 }
7445
7446 // Floating-point Convert to integer, rounding to nearest with ties to Away
7447 case Intrinsic::aarch64_neon_fcvtas:
7448 case Intrinsic::aarch64_neon_fcvtau:
7449 // Floating-point convert to integer, rounding toward minus infinity
7450 case Intrinsic::aarch64_neon_fcvtms:
7451 case Intrinsic::aarch64_neon_fcvtmu:
7452 // Floating-point convert to integer, rounding to nearest with ties to even
7453 case Intrinsic::aarch64_neon_fcvtns:
7454 case Intrinsic::aarch64_neon_fcvtnu:
7455 // Floating-point convert to integer, rounding toward plus infinity
7456 case Intrinsic::aarch64_neon_fcvtps:
7457 case Intrinsic::aarch64_neon_fcvtpu:
7458 // Floating-point Convert to integer, rounding toward Zero
7459 case Intrinsic::aarch64_neon_fcvtzs:
7460 case Intrinsic::aarch64_neon_fcvtzu:
7461 // Floating-point convert to lower precision narrow, rounding to odd
7462 case Intrinsic::aarch64_neon_fcvtxn:
7463 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/false);
7464 break;
7465
7466 // Vector Conversions Between Fixed-Point and Floating-Point
7467 case Intrinsic::aarch64_neon_vcvtfxs2fp:
7468 case Intrinsic::aarch64_neon_vcvtfp2fxs:
7469 case Intrinsic::aarch64_neon_vcvtfxu2fp:
7470 case Intrinsic::aarch64_neon_vcvtfp2fxu:
7471 handleGenericVectorConvertIntrinsic(I, /*FixedPoint=*/true);
7472 break;
7473
7474 // TODO: bfloat conversions
7475 // - bfloat @llvm.aarch64.neon.bfcvt(float)
7476 // - <8 x bfloat> @llvm.aarch64.neon.bfcvtn(<4 x float>)
7477 // - <8 x bfloat> @llvm.aarch64.neon.bfcvtn2(<8 x bfloat>, <4 x float>)
7478
7479 // Add reduction to scalar
7480 case Intrinsic::aarch64_neon_faddv:
7481 case Intrinsic::aarch64_neon_saddv:
7482 case Intrinsic::aarch64_neon_uaddv:
7483 // Signed/Unsigned min/max (Vector)
7484 // TODO: handling similarly to AND/OR may be more precise.
7485 case Intrinsic::aarch64_neon_smaxv:
7486 case Intrinsic::aarch64_neon_sminv:
7487 case Intrinsic::aarch64_neon_umaxv:
7488 case Intrinsic::aarch64_neon_uminv:
7489 // Floating-point min/max (vector)
7490 // The f{min,max}"nm"v variants handle NaN differently than f{min,max}v,
7491 // but our shadow propagation is the same.
7492 case Intrinsic::aarch64_neon_fmaxv:
7493 case Intrinsic::aarch64_neon_fminv:
7494 case Intrinsic::aarch64_neon_fmaxnmv:
7495 case Intrinsic::aarch64_neon_fminnmv:
7496 // Sum long across vector
7497 case Intrinsic::aarch64_neon_saddlv:
7498 case Intrinsic::aarch64_neon_uaddlv:
7499 handleVectorReduceIntrinsic(I, /*AllowShadowCast=*/true);
7500 break;
7501
7502 case Intrinsic::aarch64_neon_ld1x2:
7503 case Intrinsic::aarch64_neon_ld1x3:
7504 case Intrinsic::aarch64_neon_ld1x4:
7505 case Intrinsic::aarch64_neon_ld2:
7506 case Intrinsic::aarch64_neon_ld3:
7507 case Intrinsic::aarch64_neon_ld4:
7508 case Intrinsic::aarch64_neon_ld2r:
7509 case Intrinsic::aarch64_neon_ld3r:
7510 case Intrinsic::aarch64_neon_ld4r: {
7511 handleNEONVectorLoad(I, /*WithLane=*/false);
7512 break;
7513 }
7514
7515 case Intrinsic::aarch64_neon_ld2lane:
7516 case Intrinsic::aarch64_neon_ld3lane:
7517 case Intrinsic::aarch64_neon_ld4lane: {
7518 handleNEONVectorLoad(I, /*WithLane=*/true);
7519 break;
7520 }
7521
7522 // Saturating extract narrow
7523 case Intrinsic::aarch64_neon_sqxtn:
7524 case Intrinsic::aarch64_neon_sqxtun:
7525 case Intrinsic::aarch64_neon_uqxtn:
7526 // These only have one argument, but we (ab)use handleShadowOr because it
7527 // does work on single argument intrinsics and will typecast the shadow
7528 // (and update the origin).
7529 handleShadowOr(I);
7530 break;
7531
7532 case Intrinsic::aarch64_neon_st1x2:
7533 case Intrinsic::aarch64_neon_st1x3:
7534 case Intrinsic::aarch64_neon_st1x4:
7535 case Intrinsic::aarch64_neon_st2:
7536 case Intrinsic::aarch64_neon_st3:
7537 case Intrinsic::aarch64_neon_st4: {
7538 handleNEONVectorStoreIntrinsic(I, useLane: false);
7539 break;
7540 }
7541
7542 case Intrinsic::aarch64_neon_st2lane:
7543 case Intrinsic::aarch64_neon_st3lane:
7544 case Intrinsic::aarch64_neon_st4lane: {
7545 handleNEONVectorStoreIntrinsic(I, useLane: true);
7546 break;
7547 }
7548
7549 // Arm NEON vector table intrinsics have the source/table register(s) as
7550 // arguments, followed by the index register. They return the output.
7551 //
7552 // 'TBL writes a zero if an index is out-of-range, while TBX leaves the
7553 // original value unchanged in the destination register.'
7554 // Conveniently, zero denotes a clean shadow, which means out-of-range
7555 // indices for TBL will initialize the user data with zero and also clean
7556 // the shadow. (For TBX, neither the user data nor the shadow will be
7557 // updated, which is also correct.)
7558 case Intrinsic::aarch64_neon_tbl1:
7559 case Intrinsic::aarch64_neon_tbl2:
7560 case Intrinsic::aarch64_neon_tbl3:
7561 case Intrinsic::aarch64_neon_tbl4:
7562 case Intrinsic::aarch64_neon_tbx1:
7563 case Intrinsic::aarch64_neon_tbx2:
7564 case Intrinsic::aarch64_neon_tbx3:
7565 case Intrinsic::aarch64_neon_tbx4: {
7566 // The last trailing argument (index register) should be handled verbatim
7567 handleIntrinsicByApplyingToShadow(
7568 I, /*shadowIntrinsicID=*/I.getIntrinsicID(),
7569 /*trailingVerbatimArgs=*/1, /*forceIntegerIntrinsic=*/false);
7570 break;
7571 }
7572
7573 case Intrinsic::aarch64_neon_fmulx:
7574 case Intrinsic::aarch64_neon_pmul:
7575 case Intrinsic::aarch64_neon_pmull:
7576 case Intrinsic::aarch64_neon_smull:
7577 case Intrinsic::aarch64_neon_pmull64:
7578 case Intrinsic::aarch64_neon_umull: {
7579 handleNEONVectorMultiplyIntrinsic(I);
7580 break;
7581 }
7582
7583 case Intrinsic::aarch64_neon_smmla:
7584 case Intrinsic::aarch64_neon_ummla:
7585 case Intrinsic::aarch64_neon_usmmla:
7586 case Intrinsic::aarch64_neon_bfmmla:
7587 handleNEONMatrixMultiply(I);
7588 break;
7589
7590 // <2 x i32> @llvm.aarch64.neon.{u,s,us}dot.v2i32.v8i8
7591 // (<2 x i32> %acc, <8 x i8> %a, <8 x i8> %b)
7592 // <4 x i32> @llvm.aarch64.neon.{u,s,us}dot.v4i32.v16i8
7593 // (<4 x i32> %acc, <16 x i8> %a, <16 x i8> %b)
7594 case Intrinsic::aarch64_neon_sdot:
7595 case Intrinsic::aarch64_neon_udot:
7596 case Intrinsic::aarch64_neon_usdot:
7597 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/4,
7598 /*ZeroPurifies=*/true,
7599 /*EltSizeInBits=*/0,
7600 /*Lanes=*/kBothLanes);
7601 break;
7602
7603 // <2 x float> @llvm.aarch64.neon.bfdot.v2f32.v4bf16
7604 // (<2 x float> %acc, <4 x bfloat> %a, <4 x bfloat> %b)
7605 // <4 x float> @llvm.aarch64.neon.bfdot.v4f32.v8bf16
7606 // (<4 x float> %acc, <8 x bfloat> %a, <8 x bfloat> %b)
7607 case Intrinsic::aarch64_neon_bfdot:
7608 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
7609 /*ZeroPurifies=*/false,
7610 /*EltSizeInBits=*/0,
7611 /*Lanes=*/kBothLanes);
7612 break;
7613
7614 // <4 x half > @llvm.aarch64.neon.fp8.fdot2
7615 // (<4 x half >, < 8 x i8>, < 8 x i8>)
7616 // <8 x half > @llvm.aarch64.neon.fp8.fdot2
7617 // (<8 x half >, <16 x i8>, <16 x i8>)
7618 //
7619 // N.B. although the multiplicands are i8, they are actually fp8, thus
7620 // ZeroPurifies is not applicable.
7621 case Intrinsic::aarch64_neon_fp8_fdot2:
7622 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/2,
7623 /*ZeroPurifies=*/false,
7624 /*EltSizeInBits=*/0,
7625 /*Lanes=*/kBothLanes);
7626 break;
7627
7628 // <2 x float> @llvm.aarch64.neon.fp8.fdot4
7629 // (<2 x float>, < 8 x i8>, < 8 x i8>)
7630 // <4 x float> @llvm.aarch64.neon.fp8.fdot4
7631 // (<4 x float>, <16 x i8>, <16 x i8>)
7632 //
7633 // N.B. although the multiplicands are i8, they are actually fp8, thus
7634 // ZeroPurifies is not applicable.
7635 case Intrinsic::aarch64_neon_fp8_fdot4:
7636 handleVectorDotProductIntrinsic(I, /*ReductionFactor=*/4,
7637 /*ZeroPurifies=*/false,
7638 /*EltSizeInBits=*/0,
7639 /*Lanes=*/kBothLanes);
7640 break;
7641
7642 // <4 x half> @llvm.aarch64.neon.fp8.fdot2.lane
7643 // (<4 x half>, <8 x i8>, <16 x i8>, i32)
7644 // <8 x half> @llvm.aarch64.neon.fp8.fdot2.lane
7645 // (<8 x half>, <16 x i8>, <16 x i8>, i32)
7646 case Intrinsic::aarch64_neon_fp8_fdot2_lane:
7647 handleNEONDotProductLaneIntrinsic(I, /*ReductionFactor=*/2);
7648 break;
7649
7650 // <2 x float> @llvm.aarch64.neon.fp8.fdot4.lane
7651 // (<2 x float>, <8 x i8>, <16 x i8>, i32)
7652 // <4 x float> @llvm.aarch64.neon.fp8.fdot4.lane
7653 // (<4 x float>, <16 x i8>, <16 x i8>, i32)
7654 case Intrinsic::aarch64_neon_fp8_fdot4_lane:
7655 handleNEONDotProductLaneIntrinsic(I, /*ReductionFactor=*/4);
7656 break;
7657
7658 // Floating-Point Absolute Compare Greater Than/Equal
7659 case Intrinsic::aarch64_neon_facge:
7660 case Intrinsic::aarch64_neon_facgt:
7661 handleVectorComparePackedIntrinsic(I, /*PredicateAsOperand=*/false);
7662 break;
7663
7664 default:
7665 return false;
7666 }
7667
7668 return true;
7669 }
7670
7671 void visitIntrinsicInst(IntrinsicInst &I) {
7672 if (maybeHandleCrossPlatformIntrinsic(I))
7673 return;
7674
7675 if (maybeHandleX86SIMDIntrinsic(I))
7676 return;
7677
7678 if (maybeHandleArmSIMDIntrinsic(I))
7679 return;
7680
7681 if (maybeHandleUnknownIntrinsic(I))
7682 return;
7683
7684 visitInstruction(I);
7685 }
7686
7687 void visitLibAtomicLoad(CallBase &CB) {
7688 // Since we use getNextNode here, we can't have CB terminate the BB.
7689 assert(isa<CallInst>(CB));
7690
7691 IRBuilder<> IRB(&CB);
7692 Value *Size = CB.getArgOperand(i: 0);
7693 Value *SrcPtr = CB.getArgOperand(i: 1);
7694 Value *DstPtr = CB.getArgOperand(i: 2);
7695 Value *Ordering = CB.getArgOperand(i: 3);
7696 // Convert the call to have at least Acquire ordering to make sure
7697 // the shadow operations aren't reordered before it.
7698 Value *NewOrdering =
7699 IRB.CreateExtractElement(Vec: makeAddAcquireOrderingTable(IRB), Idx: Ordering);
7700 CB.setArgOperand(i: 3, v: NewOrdering);
7701
7702 NextNodeIRBuilder NextIRB(&CB);
7703 Value *SrcShadowPtr, *SrcOriginPtr;
7704 std::tie(args&: SrcShadowPtr, args&: SrcOriginPtr) =
7705 getShadowOriginPtr(Addr: SrcPtr, IRB&: NextIRB, ShadowTy: NextIRB.getInt8Ty(), Alignment: Align(1),
7706 /*isStore*/ false);
7707 Value *DstShadowPtr =
7708 getShadowOriginPtr(Addr: DstPtr, IRB&: NextIRB, ShadowTy: NextIRB.getInt8Ty(), Alignment: Align(1),
7709 /*isStore*/ true)
7710 .first;
7711
7712 NextIRB.CreateMemCpy(Dst: DstShadowPtr, DstAlign: Align(1), Src: SrcShadowPtr, SrcAlign: Align(1), Size);
7713 if (MS.TrackOrigins) {
7714 Value *SrcOrigin = NextIRB.CreateAlignedLoad(Ty: MS.OriginTy, Ptr: SrcOriginPtr,
7715 Align: kMinOriginAlignment);
7716 Value *NewOrigin = updateOrigin(V: SrcOrigin, IRB&: NextIRB);
7717 NextIRB.CreateCall(Callee: MS.MsanSetOriginFn, Args: {DstPtr, Size, NewOrigin});
7718 }
7719 }
7720
7721 void visitLibAtomicStore(CallBase &CB) {
7722 IRBuilder<> IRB(&CB);
7723 Value *Size = CB.getArgOperand(i: 0);
7724 Value *DstPtr = CB.getArgOperand(i: 2);
7725 Value *Ordering = CB.getArgOperand(i: 3);
7726 // Convert the call to have at least Release ordering to make sure
7727 // the shadow operations aren't reordered after it.
7728 Value *NewOrdering =
7729 IRB.CreateExtractElement(Vec: makeAddReleaseOrderingTable(IRB), Idx: Ordering);
7730 CB.setArgOperand(i: 3, v: NewOrdering);
7731
7732 Value *DstShadowPtr =
7733 getShadowOriginPtr(Addr: DstPtr, IRB, ShadowTy: IRB.getInt8Ty(), Alignment: Align(1),
7734 /*isStore*/ true)
7735 .first;
7736
7737 // Atomic store always paints clean shadow/origin. See file header.
7738 IRB.CreateMemSet(Ptr: DstShadowPtr, Val: getCleanShadow(OrigTy: IRB.getInt8Ty()), Size,
7739 Align: Align(1));
7740 }
7741
7742 void visitCallBase(CallBase &CB) {
7743 assert(!CB.getMetadata(LLVMContext::MD_nosanitize));
7744 if (CB.isInlineAsm()) {
7745 // For inline asm (either a call to asm function, or callbr instruction),
7746 // do the usual thing: check argument shadow and mark all outputs as
7747 // clean. Note that any side effects of the inline asm that are not
7748 // immediately visible in its constraints are not handled.
7749 if (ClHandleAsmConservative)
7750 visitAsmInstruction(I&: CB);
7751 else
7752 visitInstruction(I&: CB);
7753 return;
7754 }
7755 LibFunc LF = TLI->getLibFunc(CB);
7756 if (LF != NotLibFunc) {
7757 // libatomic.a functions need to have special handling because there isn't
7758 // a good way to intercept them or compile the library with
7759 // instrumentation.
7760 switch (LF) {
7761 case LibFunc_atomic_load:
7762 if (!isa<CallInst>(Val: CB)) {
7763 llvm::errs() << "MSAN -- cannot instrument invoke of libatomic load."
7764 "Ignoring!\n";
7765 break;
7766 }
7767 visitLibAtomicLoad(CB);
7768 return;
7769 case LibFunc_atomic_store:
7770 visitLibAtomicStore(CB);
7771 return;
7772 default:
7773 break;
7774 }
7775 }
7776
7777 if (auto *Call = dyn_cast<CallInst>(Val: &CB)) {
7778 assert(!isa<IntrinsicInst>(Call) && "intrinsics are handled elsewhere");
7779
7780 // We are going to insert code that relies on the fact that the callee
7781 // will become a non-readonly function after it is instrumented by us. To
7782 // prevent this code from being optimized out, mark that function
7783 // non-readonly in advance.
7784 // TODO: We can likely do better than dropping memory() completely here.
7785 AttributeMask B;
7786 B.addAttribute(Val: Attribute::Memory).addAttribute(Val: Attribute::Speculatable);
7787
7788 Call->removeFnAttrs(AttrsToRemove: B);
7789 if (Function *Func = Call->getCalledFunction()) {
7790 Func->removeFnAttrs(Attrs: B);
7791 }
7792
7793 maybeMarkSanitizerLibraryCallNoBuiltin(CI: Call, TLI);
7794 }
7795 IRBuilder<> IRB(&CB);
7796 bool MayCheckCall = MS.EagerChecks;
7797 if (Function *Func = CB.getCalledFunction()) {
7798 // __sanitizer_unaligned_{load,store} functions may be called by users
7799 // and always expects shadows in the TLS. So don't check them.
7800 MayCheckCall &= !Func->getName().starts_with(Prefix: "__sanitizer_unaligned_");
7801 }
7802
7803 unsigned ArgOffset = 0;
7804 LLVM_DEBUG(dbgs() << " CallSite: " << CB << "\n");
7805 for (const auto &[i, A] : llvm::enumerate(First: CB.args())) {
7806 if (!A->getType()->isSized()) {
7807 LLVM_DEBUG(dbgs() << "Arg " << i << " is not sized: " << CB << "\n");
7808 continue;
7809 }
7810
7811 if (A->getType()->isScalableTy()) {
7812 LLVM_DEBUG(dbgs() << "Arg " << i << " is vscale: " << CB << "\n");
7813 // Handle as noundef, but don't reserve tls slots.
7814 insertCheckShadowOf(Val: A, OrigIns: &CB);
7815 continue;
7816 }
7817
7818 unsigned Size = 0;
7819 const DataLayout &DL = F.getDataLayout();
7820
7821 bool ByVal = CB.isByValArgument(ArgNo: i);
7822 bool NoUndef = CB.paramHasAttr(ArgNo: i, Kind: Attribute::NoUndef);
7823 bool EagerCheck = MayCheckCall && !ByVal && NoUndef;
7824
7825 if (EagerCheck) {
7826 insertCheckShadowOf(Val: A, OrigIns: &CB);
7827 Size = DL.getTypeAllocSize(Ty: A->getType());
7828 } else {
7829 [[maybe_unused]] Value *Store = nullptr;
7830 // Compute the Shadow for arg even if it is ByVal, because
7831 // in that case getShadow() will copy the actual arg shadow to
7832 // __msan_param_tls.
7833 Value *ArgShadow = getShadow(V: A);
7834 Value *ArgShadowBase = getShadowPtrForArgument(IRB, ArgOffset);
7835 LLVM_DEBUG(dbgs() << " Arg#" << i << ": " << *A
7836 << " Shadow: " << *ArgShadow << "\n");
7837 if (ByVal) {
7838 // ByVal requires some special handling as it's too big for a single
7839 // load
7840 assert(A->getType()->isPointerTy() &&
7841 "ByVal argument is not a pointer!");
7842 Size = DL.getTypeAllocSize(Ty: CB.getParamByValType(ArgNo: i));
7843 if (ArgOffset + Size > kParamTLSSize)
7844 break;
7845 const MaybeAlign ParamAlignment(CB.getParamAlign(ArgNo: i));
7846 MaybeAlign Alignment = std::nullopt;
7847 if (ParamAlignment)
7848 Alignment = std::min(a: *ParamAlignment, b: kShadowTLSAlignment);
7849 Value *AShadowPtr, *AOriginPtr;
7850 std::tie(args&: AShadowPtr, args&: AOriginPtr) =
7851 getShadowOriginPtr(Addr: A, IRB, ShadowTy: IRB.getInt8Ty(), Alignment,
7852 /*isStore*/ false);
7853 if (!PropagateShadow) {
7854 Store = IRB.CreateMemSet(Ptr: ArgShadowBase,
7855 Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
7856 Size, Align: Alignment);
7857 } else {
7858 Store = IRB.CreateMemCpy(Dst: ArgShadowBase, DstAlign: Alignment, Src: AShadowPtr,
7859 SrcAlign: Alignment, Size);
7860 if (MS.TrackOrigins) {
7861 Value *ArgOriginBase = getOriginPtrForArgument(IRB, ArgOffset);
7862 // FIXME: OriginSize should be:
7863 // alignTo(A % kMinOriginAlignment + Size, kMinOriginAlignment)
7864 unsigned OriginSize = alignTo(Size, A: kMinOriginAlignment);
7865 IRB.CreateMemCpy(
7866 Dst: ArgOriginBase,
7867 /* by origin_tls[ArgOffset] */ DstAlign: kMinOriginAlignment,
7868 Src: AOriginPtr,
7869 /* by getShadowOriginPtr */ SrcAlign: kMinOriginAlignment, Size: OriginSize);
7870 }
7871 }
7872 } else {
7873 // Any other parameters mean we need bit-grained tracking of uninit
7874 // data
7875 Size = DL.getTypeAllocSize(Ty: A->getType());
7876 if (ArgOffset + Size > kParamTLSSize)
7877 break;
7878 Store = IRB.CreateAlignedStore(Val: ArgShadow, Ptr: ArgShadowBase,
7879 Align: kShadowTLSAlignment);
7880 Constant *Cst = dyn_cast<Constant>(Val: ArgShadow);
7881 if (MS.TrackOrigins && !(Cst && Cst->isNullValue())) {
7882 IRB.CreateStore(Val: getOrigin(V: A),
7883 Ptr: getOriginPtrForArgument(IRB, ArgOffset));
7884 }
7885 }
7886 assert(Store != nullptr);
7887 LLVM_DEBUG(dbgs() << " Param:" << *Store << "\n");
7888 }
7889 assert(Size != 0);
7890 ArgOffset += alignTo(Size, A: kShadowTLSAlignment);
7891 }
7892 LLVM_DEBUG(dbgs() << " done with call args\n");
7893
7894 FunctionType *FT = CB.getFunctionType();
7895 if (FT->isVarArg()) {
7896 VAHelper->visitCallBase(CB, IRB);
7897 }
7898
7899 // Now, get the shadow for the RetVal.
7900 if (!CB.getType()->isSized())
7901 return;
7902 // Don't emit the epilogue for musttail call returns.
7903 if (isa<CallInst>(Val: CB) && cast<CallInst>(Val&: CB).isMustTailCall())
7904 return;
7905
7906 if (MayCheckCall && CB.hasRetAttr(Kind: Attribute::NoUndef)) {
7907 setShadow(V: &CB, SV: getCleanShadow(V: &CB));
7908 setOrigin(V: &CB, Origin: getCleanOrigin());
7909 return;
7910 }
7911
7912 IRBuilder<> IRBBefore(&CB);
7913 // Until we have full dynamic coverage, make sure the retval shadow is 0.
7914 Value *Base = getShadowPtrForRetval(IRB&: IRBBefore);
7915 IRBBefore.CreateAlignedStore(Val: getCleanShadow(V: &CB), Ptr: Base,
7916 Align: kShadowTLSAlignment);
7917 BasicBlock::iterator NextInsn;
7918 if (isa<CallInst>(Val: CB)) {
7919 NextInsn = ++CB.getIterator();
7920 assert(NextInsn != CB.getParent()->end());
7921 } else {
7922 BasicBlock *NormalDest = cast<InvokeInst>(Val&: CB).getNormalDest();
7923 if (!NormalDest->getSinglePredecessor()) {
7924 // FIXME: this case is tricky, so we are just conservative here.
7925 // Perhaps we need to split the edge between this BB and NormalDest,
7926 // but a naive attempt to use SplitEdge leads to a crash.
7927 setShadow(V: &CB, SV: getCleanShadow(V: &CB));
7928 setOrigin(V: &CB, Origin: getCleanOrigin());
7929 return;
7930 }
7931 // FIXME: NextInsn is likely in a basic block that has not been visited
7932 // yet. Anything inserted there will be instrumented by MSan later!
7933 NextInsn = NormalDest->getFirstInsertionPt();
7934 assert(NextInsn != NormalDest->end() &&
7935 "Could not find insertion point for retval shadow load");
7936 }
7937 IRBuilder<> IRBAfter(&*NextInsn);
7938 Value *RetvalShadow = IRBAfter.CreateAlignedLoad(
7939 Ty: getShadowTy(V: &CB), Ptr: getShadowPtrForRetval(IRB&: IRBAfter), Align: kShadowTLSAlignment,
7940 Name: "_msret");
7941 setShadow(V: &CB, SV: RetvalShadow);
7942 if (MS.TrackOrigins)
7943 setOrigin(V: &CB, Origin: IRBAfter.CreateLoad(Ty: MS.OriginTy, Ptr: getOriginPtrForRetval()));
7944 }
7945
7946 bool isAMustTailRetVal(Value *RetVal) {
7947 if (auto *I = dyn_cast<BitCastInst>(Val: RetVal)) {
7948 RetVal = I->getOperand(i_nocapture: 0);
7949 }
7950 if (auto *I = dyn_cast<CallInst>(Val: RetVal)) {
7951 return I->isMustTailCall();
7952 }
7953 return false;
7954 }
7955
7956 void visitReturnInst(ReturnInst &I) {
7957 IRBuilder<> IRB(&I);
7958 Value *RetVal = I.getReturnValue();
7959 if (!RetVal)
7960 return;
7961 // Don't emit the epilogue for musttail call returns.
7962 if (isAMustTailRetVal(RetVal))
7963 return;
7964 Value *ShadowPtr = getShadowPtrForRetval(IRB);
7965 bool HasNoUndef = F.hasRetAttribute(Kind: Attribute::NoUndef);
7966 bool StoreShadow = !(MS.EagerChecks && HasNoUndef);
7967 // FIXME: Consider using SpecialCaseList to specify a list of functions that
7968 // must always return fully initialized values. For now, we hardcode "main".
7969 bool EagerCheck = (MS.EagerChecks && HasNoUndef) || (F.getName() == "main");
7970
7971 Value *Shadow = getShadow(V: RetVal);
7972 bool StoreOrigin = true;
7973 if (EagerCheck) {
7974 insertCheckShadowOf(Val: RetVal, OrigIns: &I);
7975 Shadow = getCleanShadow(V: RetVal);
7976 StoreOrigin = false;
7977 }
7978
7979 // The caller may still expect information passed over TLS if we pass our
7980 // check
7981 if (StoreShadow) {
7982 IRB.CreateAlignedStore(Val: Shadow, Ptr: ShadowPtr, Align: kShadowTLSAlignment);
7983 if (MS.TrackOrigins && StoreOrigin)
7984 IRB.CreateStore(Val: getOrigin(V: RetVal), Ptr: getOriginPtrForRetval());
7985 }
7986 }
7987
7988 void visitPHINode(PHINode &I) {
7989 IRBuilder<> IRB(&I);
7990 if (!PropagateShadow) {
7991 setShadow(V: &I, SV: getCleanShadow(V: &I));
7992 setOrigin(V: &I, Origin: getCleanOrigin());
7993 return;
7994 }
7995
7996 ShadowPHINodes.push_back(Elt: &I);
7997 setShadow(V: &I, SV: IRB.CreatePHI(Ty: getShadowTy(V: &I), NumReservedValues: I.getNumIncomingValues(),
7998 Name: "_msphi_s"));
7999 if (MS.TrackOrigins)
8000 setOrigin(
8001 V: &I, Origin: IRB.CreatePHI(Ty: MS.OriginTy, NumReservedValues: I.getNumIncomingValues(), Name: "_msphi_o"));
8002 }
8003
8004 Value *getLocalVarIdptr(AllocaInst &I) {
8005 ConstantInt *IntConst =
8006 ConstantInt::get(Ty: Type::getInt32Ty(C&: (*F.getParent()).getContext()), V: 0);
8007 return new GlobalVariable(*F.getParent(), IntConst->getType(),
8008 /*isConstant=*/false, GlobalValue::PrivateLinkage,
8009 IntConst);
8010 }
8011
8012 Value *getLocalVarDescription(AllocaInst &I) {
8013 return createPrivateConstGlobalForString(M&: *F.getParent(), Str: I.getName());
8014 }
8015
8016 void poisonAllocaUserspace(AllocaInst &I, IRBuilder<> &IRB, Value *Len) {
8017 if (PoisonStack && ClPoisonStackWithCall) {
8018 IRB.CreateCall(Callee: MS.MsanPoisonStackFn, Args: {&I, Len});
8019 } else {
8020 Value *ShadowBase, *OriginBase;
8021 std::tie(args&: ShadowBase, args&: OriginBase) = getShadowOriginPtr(
8022 Addr: &I, IRB, ShadowTy: IRB.getInt8Ty(), Alignment: Align(1), /*isStore*/ true);
8023
8024 Value *PoisonValue = IRB.getInt8(C: PoisonStack ? ClPoisonStackPattern : 0);
8025 IRB.CreateMemSet(Ptr: ShadowBase, Val: PoisonValue, Size: Len, Align: I.getAlign());
8026 }
8027
8028 if (PoisonStack && MS.TrackOrigins) {
8029 Value *Idptr = getLocalVarIdptr(I);
8030 if (ClPrintStackNames) {
8031 Value *Descr = getLocalVarDescription(I);
8032 IRB.CreateCall(Callee: MS.MsanSetAllocaOriginWithDescriptionFn,
8033 Args: {&I, Len, Idptr, Descr});
8034 } else {
8035 IRB.CreateCall(Callee: MS.MsanSetAllocaOriginNoDescriptionFn, Args: {&I, Len, Idptr});
8036 }
8037 }
8038 }
8039
8040 void poisonAllocaKmsan(AllocaInst &I, IRBuilder<> &IRB, Value *Len) {
8041 Value *Descr = getLocalVarDescription(I);
8042 if (PoisonStack) {
8043 IRB.CreateCall(Callee: MS.MsanPoisonAllocaFn, Args: {&I, Len, Descr});
8044 } else {
8045 IRB.CreateCall(Callee: MS.MsanUnpoisonAllocaFn, Args: {&I, Len});
8046 }
8047 }
8048
8049 void instrumentAlloca(AllocaInst &I, Instruction *InsPoint = nullptr) {
8050 if (!InsPoint)
8051 InsPoint = &I;
8052 NextNodeIRBuilder IRB(InsPoint);
8053 Value *Len = IRB.CreateAllocationSize(DestTy: MS.IntptrTy, AI: &I);
8054
8055 if (MS.CompileKernel)
8056 poisonAllocaKmsan(I, IRB, Len);
8057 else
8058 poisonAllocaUserspace(I, IRB, Len);
8059 }
8060
8061 void visitAllocaInst(AllocaInst &I) {
8062 setShadow(V: &I, SV: getCleanShadow(V: &I));
8063 setOrigin(V: &I, Origin: getCleanOrigin());
8064 // We'll get to this alloca later unless it's poisoned at the corresponding
8065 // llvm.lifetime.start.
8066 AllocaSet.insert(X: &I);
8067 }
8068
8069 void visitSelectInst(SelectInst &I) {
8070 // a = select b, c, d
8071 Value *B = I.getCondition();
8072 Value *C = I.getTrueValue();
8073 Value *D = I.getFalseValue();
8074
8075 handleSelectLikeInst(I, B, C, D);
8076 }
8077
8078 void handleSelectLikeInst(Instruction &I, Value *B, Value *C, Value *D) {
8079 IRBuilder<> IRB(&I);
8080
8081 Value *Sb = getShadow(V: B);
8082 Value *Sc = getShadow(V: C);
8083 Value *Sd = getShadow(V: D);
8084
8085 Value *Ob = MS.TrackOrigins ? getOrigin(V: B) : nullptr;
8086 Value *Oc = MS.TrackOrigins ? getOrigin(V: C) : nullptr;
8087 Value *Od = MS.TrackOrigins ? getOrigin(V: D) : nullptr;
8088
8089 // Result shadow if condition shadow is 0.
8090 Value *Sa0 = IRB.CreateSelect(C: B, True: Sc, False: Sd);
8091 Value *Sa1;
8092 if (I.getType()->isAggregateType()) {
8093 // To avoid "sign extending" i1 to an arbitrary aggregate type, we just do
8094 // an extra "select". This results in much more compact IR.
8095 // Sa = select Sb, poisoned, (select b, Sc, Sd)
8096 Sa1 = getPoisonedShadow(ShadowTy: getShadowTy(OrigTy: I.getType()));
8097 } else if (isScalableNonVectorType(Ty: I.getType())) {
8098 // This is intended to handle target("aarch64.svcount"), which can't be
8099 // handled in the else branch because of incompatibility with CreateXor
8100 // ("The supported LLVM operations on this type are limited to load,
8101 // store, phi, select and alloca instructions").
8102
8103 // TODO: this currently underapproximates. Use Arm SVE EOR in the else
8104 // branch as needed instead.
8105 Sa1 = getCleanShadow(OrigTy: getShadowTy(OrigTy: I.getType()));
8106 } else {
8107 // Sa = select Sb, [ (c^d) | Sc | Sd ], [ b ? Sc : Sd ]
8108 // If Sb (condition is poisoned), look for bits in c and d that are equal
8109 // and both unpoisoned.
8110 // If !Sb (condition is unpoisoned), simply pick one of Sc and Sd.
8111
8112 // Cast arguments to shadow-compatible type.
8113 C = CreateAppToShadowCast(IRB, V: C);
8114 D = CreateAppToShadowCast(IRB, V: D);
8115
8116 // Result shadow if condition shadow is 1.
8117 Sa1 = IRB.CreateOr(Ops: {IRB.CreateXor(LHS: C, RHS: D), Sc, Sd});
8118 }
8119 Value *Sa = IRB.CreateSelect(C: Sb, True: Sa1, False: Sa0, Name: "_msprop_select");
8120 setShadow(V: &I, SV: Sa);
8121 if (MS.TrackOrigins) {
8122 // Origins are always i32, so any vector conditions must be flattened.
8123 // FIXME: consider tracking vector origins for app vectors?
8124 if (B->getType()->isVectorTy()) {
8125 B = convertToBool(V: B, IRB);
8126 Sb = convertToBool(V: Sb, IRB);
8127 }
8128 // a = select b, c, d
8129 // Oa = Sb ? Ob : (b ? Oc : Od)
8130 setOrigin(V: &I, Origin: IRB.CreateSelect(C: Sb, True: Ob, False: IRB.CreateSelect(C: B, True: Oc, False: Od)));
8131 }
8132 }
8133
8134 void visitLandingPadInst(LandingPadInst &I) {
8135 // Do nothing.
8136 // See https://github.com/google/sanitizers/issues/504
8137 setShadow(V: &I, SV: getCleanShadow(V: &I));
8138 setOrigin(V: &I, Origin: getCleanOrigin());
8139 }
8140
8141 void visitCatchSwitchInst(CatchSwitchInst &I) {
8142 setShadow(V: &I, SV: getCleanShadow(V: &I));
8143 setOrigin(V: &I, Origin: getCleanOrigin());
8144 }
8145
8146 void visitFuncletPadInst(FuncletPadInst &I) {
8147 setShadow(V: &I, SV: getCleanShadow(V: &I));
8148 setOrigin(V: &I, Origin: getCleanOrigin());
8149 }
8150
8151 void visitGetElementPtrInst(GetElementPtrInst &I) { handleShadowOr(I); }
8152
8153 void visitExtractValueInst(ExtractValueInst &I) {
8154 IRBuilder<> IRB(&I);
8155 Value *Agg = I.getAggregateOperand();
8156 LLVM_DEBUG(dbgs() << "ExtractValue: " << I << "\n");
8157 Value *AggShadow = getShadow(V: Agg);
8158 LLVM_DEBUG(dbgs() << " AggShadow: " << *AggShadow << "\n");
8159 Value *ResShadow = IRB.CreateExtractValue(Agg: AggShadow, Idxs: I.getIndices());
8160 LLVM_DEBUG(dbgs() << " ResShadow: " << *ResShadow << "\n");
8161 setShadow(V: &I, SV: ResShadow);
8162 setOriginForNaryOp(I);
8163 }
8164
8165 void visitInsertValueInst(InsertValueInst &I) {
8166 IRBuilder<> IRB(&I);
8167 LLVM_DEBUG(dbgs() << "InsertValue: " << I << "\n");
8168 Value *AggShadow = getShadow(V: I.getAggregateOperand());
8169 Value *InsShadow = getShadow(V: I.getInsertedValueOperand());
8170 LLVM_DEBUG(dbgs() << " AggShadow: " << *AggShadow << "\n");
8171 LLVM_DEBUG(dbgs() << " InsShadow: " << *InsShadow << "\n");
8172 Value *Res = IRB.CreateInsertValue(Agg: AggShadow, Val: InsShadow, Idxs: I.getIndices());
8173 LLVM_DEBUG(dbgs() << " Res: " << *Res << "\n");
8174 setShadow(V: &I, SV: Res);
8175 setOriginForNaryOp(I);
8176 }
8177
8178 void dumpInst(Instruction &I, const Twine &Prefix) {
8179 // Instruction name only
8180 // For intrinsics, the full/overloaded name is used
8181 //
8182 // e.g., "call llvm.aarch64.neon.uqsub.v16i8"
8183 if (CallInst *CI = dyn_cast<CallInst>(Val: &I)) {
8184 errs() << "ZZZ:" << Prefix << " call "
8185 << CI->getCalledFunction()->getName() << "\n";
8186 } else {
8187 errs() << "ZZZ:" << Prefix << " " << I.getOpcodeName() << "\n";
8188 }
8189
8190 // Instruction prototype (including return type and parameter types)
8191 // For intrinsics, we use the base/non-overloaded name
8192 //
8193 // e.g., "call <16 x i8> @llvm.aarch64.neon.uqsub(<16 x i8>, <16 x i8>)"
8194 unsigned NumOperands = I.getNumOperands();
8195 if (CallInst *CI = dyn_cast<CallInst>(Val: &I)) {
8196 errs() << "YYY:" << Prefix << " call " << *I.getType() << " @";
8197
8198 if (IntrinsicInst *II = dyn_cast<IntrinsicInst>(Val: CI))
8199 errs() << Intrinsic::getBaseName(id: II->getIntrinsicID());
8200 else
8201 errs() << CI->getCalledFunction()->getName();
8202
8203 errs() << "(";
8204
8205 // The last operand of a CallInst is the function itself.
8206 NumOperands--;
8207 } else
8208 errs() << "YYY:" << Prefix << " " << *I.getType() << " "
8209 << I.getOpcodeName() << "(";
8210
8211 for (size_t i = 0; i < NumOperands; i++) {
8212 if (i > 0)
8213 errs() << ", ";
8214
8215 errs() << *(I.getOperand(i)->getType());
8216 }
8217
8218 errs() << ")\n";
8219
8220 // Full instruction, including types and operand values
8221 // For intrinsics, the full/overloaded name is used
8222 //
8223 // e.g., "%vqsubq_v.i15 = call noundef <16 x i8>
8224 // @llvm.aarch64.neon.uqsub.v16i8(<16 x i8> %vext21.i,
8225 // <16 x i8> splat (i8 1)), !dbg !66"
8226 errs() << "QQQ:" << Prefix << " " << I << "\n";
8227 }
8228
8229 void visitResumeInst(ResumeInst &I) {
8230 LLVM_DEBUG(dbgs() << "Resume: " << I << "\n");
8231 // Nothing to do here.
8232 }
8233
8234 void visitCleanupReturnInst(CleanupReturnInst &CRI) {
8235 LLVM_DEBUG(dbgs() << "CleanupReturn: " << CRI << "\n");
8236 // Nothing to do here.
8237 }
8238
8239 void visitCatchReturnInst(CatchReturnInst &CRI) {
8240 LLVM_DEBUG(dbgs() << "CatchReturn: " << CRI << "\n");
8241 // Nothing to do here.
8242 }
8243
8244 void instrumentAsmArgument(Value *Operand, Type *ElemTy, Instruction &I,
8245 IRBuilder<> &IRB, const DataLayout &DL,
8246 bool isOutput) {
8247 // For each assembly argument, we check its value for being initialized.
8248 // If the argument is a pointer, we assume it points to a single element
8249 // of the corresponding type (or to a 8-byte word, if the type is unsized).
8250 // Each such pointer is instrumented with a call to the runtime library.
8251 Type *OpType = Operand->getType();
8252 // Check the operand value itself.
8253 insertCheckShadowOf(Val: Operand, OrigIns: &I);
8254 if (!OpType->isPointerTy() || !isOutput) {
8255 assert(!isOutput);
8256 return;
8257 }
8258 if (!ElemTy->isSized())
8259 return;
8260 auto Size = DL.getTypeStoreSize(Ty: ElemTy);
8261 Value *SizeVal = IRB.CreateTypeSize(Ty: MS.IntptrTy, Size);
8262 if (MS.CompileKernel) {
8263 IRB.CreateCall(Callee: MS.MsanInstrumentAsmStoreFn, Args: {Operand, SizeVal});
8264 } else {
8265 // ElemTy, derived from elementtype(), does not encode the alignment of
8266 // the pointer. Conservatively assume that the shadow memory is unaligned.
8267 // When Size is large, avoid StoreInst as it would expand to many
8268 // instructions.
8269 auto [ShadowPtr, _] =
8270 getShadowOriginPtrUserspace(Addr: Operand, IRB, ShadowTy: IRB.getInt8Ty(), Alignment: Align(1));
8271 if (Size <= 32)
8272 IRB.CreateAlignedStore(Val: getCleanShadow(OrigTy: ElemTy), Ptr: ShadowPtr, Align: Align(1));
8273 else
8274 IRB.CreateMemSet(Ptr: ShadowPtr, Val: ConstantInt::getNullValue(Ty: IRB.getInt8Ty()),
8275 Size: SizeVal, Align: Align(1));
8276 }
8277 }
8278
8279 /// Get the number of output arguments returned by pointers.
8280 int getNumOutputArgs(InlineAsm *IA, CallBase *CB) {
8281 int NumRetOutputs = 0;
8282 int NumOutputs = 0;
8283 Type *RetTy = cast<Value>(Val: CB)->getType();
8284 if (!RetTy->isVoidTy()) {
8285 // Register outputs are returned via the CallInst return value.
8286 auto *ST = dyn_cast<StructType>(Val: RetTy);
8287 if (ST)
8288 NumRetOutputs = ST->getNumElements();
8289 else
8290 NumRetOutputs = 1;
8291 }
8292 InlineAsm::ConstraintInfoVector Constraints = IA->ParseConstraints();
8293 for (const InlineAsm::ConstraintInfo &Info : Constraints) {
8294 switch (Info.Type) {
8295 case InlineAsm::isOutput:
8296 NumOutputs++;
8297 break;
8298 default:
8299 break;
8300 }
8301 }
8302 return NumOutputs - NumRetOutputs;
8303 }
8304
8305 void visitAsmInstruction(Instruction &I) {
8306 // Conservative inline assembly handling: check for poisoned shadow of
8307 // asm() arguments, then unpoison the result and all the memory locations
8308 // pointed to by those arguments.
8309 // An inline asm() statement in C++ contains lists of input and output
8310 // arguments used by the assembly code. These are mapped to operands of the
8311 // CallInst as follows:
8312 // - nR register outputs ("=r) are returned by value in a single structure
8313 // (SSA value of the CallInst);
8314 // - nO other outputs ("=m" and others) are returned by pointer as first
8315 // nO operands of the CallInst;
8316 // - nI inputs ("r", "m" and others) are passed to CallInst as the
8317 // remaining nI operands.
8318 // The total number of asm() arguments in the source is nR+nO+nI, and the
8319 // corresponding CallInst has nO+nI+1 operands (the last operand is the
8320 // function to be called).
8321 const DataLayout &DL = F.getDataLayout();
8322 CallBase *CB = cast<CallBase>(Val: &I);
8323 IRBuilder<> IRB(&I);
8324 InlineAsm *IA = cast<InlineAsm>(Val: CB->getCalledOperand());
8325 int OutputArgs = getNumOutputArgs(IA, CB);
8326 // The last operand of a CallInst is the function itself.
8327 int NumOperands = CB->getNumOperands() - 1;
8328
8329 // Check input arguments. Doing so before unpoisoning output arguments, so
8330 // that we won't overwrite uninit values before checking them.
8331 for (int i = OutputArgs; i < NumOperands; i++) {
8332 Value *Operand = CB->getOperand(i_nocapture: i);
8333 instrumentAsmArgument(Operand, ElemTy: CB->getParamElementType(ArgNo: i), I, IRB, DL,
8334 /*isOutput*/ false);
8335 }
8336 // Unpoison output arguments. This must happen before the actual InlineAsm
8337 // call, so that the shadow for memory published in the asm() statement
8338 // remains valid.
8339 for (int i = 0; i < OutputArgs; i++) {
8340 Value *Operand = CB->getOperand(i_nocapture: i);
8341 instrumentAsmArgument(Operand, ElemTy: CB->getParamElementType(ArgNo: i), I, IRB, DL,
8342 /*isOutput*/ true);
8343 }
8344
8345 setShadow(V: &I, SV: getCleanShadow(V: &I));
8346 setOrigin(V: &I, Origin: getCleanOrigin());
8347 }
8348
8349 void visitFreezeInst(FreezeInst &I) {
8350 // Freeze always returns a fully defined value.
8351 setShadow(V: &I, SV: getCleanShadow(V: &I));
8352 setOrigin(V: &I, Origin: getCleanOrigin());
8353 }
8354
8355 void visitInstruction(Instruction &I) {
8356 // Everything else: stop propagating and check for poisoned shadow.
8357 if (ClDumpStrictInstructions)
8358 dumpInst(I, Prefix: "Strict");
8359 LLVM_DEBUG(dbgs() << "DEFAULT: " << I << "\n");
8360 for (size_t i = 0, n = I.getNumOperands(); i < n; i++) {
8361 Value *Operand = I.getOperand(i);
8362 if (Operand->getType()->isSized())
8363 insertCheckShadowOf(Val: Operand, OrigIns: &I);
8364 }
8365 setShadow(V: &I, SV: getCleanShadow(V: &I));
8366 setOrigin(V: &I, Origin: getCleanOrigin());
8367 }
8368};
8369
8370struct VarArgHelperBase : public VarArgHelper {
8371 Function &F;
8372 MemorySanitizer &MS;
8373 MemorySanitizerVisitor &MSV;
8374 SmallVector<CallInst *, 16> VAStartInstrumentationList;
8375 const unsigned VAListTagSize;
8376
8377 VarArgHelperBase(Function &F, MemorySanitizer &MS,
8378 MemorySanitizerVisitor &MSV, unsigned VAListTagSize)
8379 : F(F), MS(MS), MSV(MSV), VAListTagSize(VAListTagSize) {}
8380
8381 Value *getShadowAddrForVAArgument(IRBuilder<> &IRB, unsigned ArgOffset) {
8382 Value *Base = IRB.CreatePointerCast(V: MS.VAArgTLS, DestTy: MS.IntptrTy);
8383 return IRB.CreateAdd(LHS: Base, RHS: ConstantInt::get(Ty: MS.IntptrTy, V: ArgOffset));
8384 }
8385
8386 /// Compute the shadow address for a given va_arg.
8387 Value *getShadowPtrForVAArgument(IRBuilder<> &IRB, unsigned ArgOffset) {
8388 return IRB.CreatePtrAdd(
8389 Ptr: MS.VAArgTLS, Offset: ConstantInt::get(Ty: MS.IntptrTy, V: ArgOffset), Name: "_msarg_va_s");
8390 }
8391
8392 /// Compute the shadow address for a given va_arg.
8393 Value *getShadowPtrForVAArgument(IRBuilder<> &IRB, unsigned ArgOffset,
8394 unsigned ArgSize) {
8395 // Make sure we don't overflow __msan_va_arg_tls.
8396 if (ArgOffset + ArgSize > kParamTLSSize)
8397 return nullptr;
8398 return getShadowPtrForVAArgument(IRB, ArgOffset);
8399 }
8400
8401 /// Compute the origin address for a given va_arg.
8402 Value *getOriginPtrForVAArgument(IRBuilder<> &IRB, int ArgOffset) {
8403 // getOriginPtrForVAArgument() is always called after
8404 // getShadowPtrForVAArgument(), so __msan_va_arg_origin_tls can never
8405 // overflow.
8406 return IRB.CreatePtrAdd(Ptr: MS.VAArgOriginTLS,
8407 Offset: ConstantInt::get(Ty: MS.IntptrTy, V: ArgOffset),
8408 Name: "_msarg_va_o");
8409 }
8410
8411 void CleanUnusedTLS(IRBuilder<> &IRB, Value *ShadowBase,
8412 unsigned BaseOffset) {
8413 // The tails of __msan_va_arg_tls is not large enough to fit full
8414 // value shadow, but it will be copied to backup anyway. Make it
8415 // clean.
8416 if (BaseOffset >= kParamTLSSize)
8417 return;
8418 Value *TailSize =
8419 ConstantInt::getSigned(Ty: IRB.getInt32Ty(), V: kParamTLSSize - BaseOffset);
8420 IRB.CreateMemSet(Ptr: ShadowBase, Val: ConstantInt::getNullValue(Ty: IRB.getInt8Ty()),
8421 Size: TailSize, Align: Align(8));
8422 }
8423
8424 void unpoisonVAListTagForInst(IntrinsicInst &I) {
8425 IRBuilder<> IRB(&I);
8426 Value *VAListTag = I.getArgOperand(i: 0);
8427 const Align Alignment = Align(8);
8428 auto [ShadowPtr, OriginPtr] = MSV.getShadowOriginPtr(
8429 Addr: VAListTag, IRB, ShadowTy: IRB.getInt8Ty(), Alignment, /*isStore*/ true);
8430 // Unpoison the whole __va_list_tag.
8431 IRB.CreateMemSet(Ptr: ShadowPtr, Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
8432 Size: VAListTagSize, Align: Alignment, isVolatile: false);
8433 }
8434
8435 void visitVAStartInst(VAStartInst &I) override {
8436 if (F.getCallingConv() == CallingConv::Win64)
8437 return;
8438 VAStartInstrumentationList.push_back(Elt: &I);
8439 unpoisonVAListTagForInst(I);
8440 }
8441
8442 void visitVACopyInst(VACopyInst &I) override {
8443 if (F.getCallingConv() == CallingConv::Win64)
8444 return;
8445 unpoisonVAListTagForInst(I);
8446 }
8447};
8448
8449/// AMD64-specific implementation of VarArgHelper.
8450struct VarArgAMD64Helper : public VarArgHelperBase {
8451 // An unfortunate workaround for asymmetric lowering of va_arg stuff.
8452 // See a comment in visitCallBase for more details.
8453 static const unsigned AMD64GpEndOffset = 48; // AMD64 ABI Draft 0.99.6 p3.5.7
8454 static const unsigned AMD64FpEndOffsetSSE = 176;
8455 // If SSE is disabled, fp_offset in va_list is zero.
8456 static const unsigned AMD64FpEndOffsetNoSSE = AMD64GpEndOffset;
8457
8458 unsigned AMD64FpEndOffset;
8459 AllocaInst *VAArgTLSCopy = nullptr;
8460 AllocaInst *VAArgTLSOriginCopy = nullptr;
8461 Value *VAArgOverflowSize = nullptr;
8462
8463 enum ArgKind { AK_GeneralPurpose, AK_FloatingPoint, AK_Memory };
8464
8465 VarArgAMD64Helper(Function &F, MemorySanitizer &MS,
8466 MemorySanitizerVisitor &MSV)
8467 : VarArgHelperBase(F, MS, MSV, /*VAListTagSize=*/24) {
8468 AMD64FpEndOffset = AMD64FpEndOffsetSSE;
8469 for (const auto &Attr : F.getAttributes().getFnAttrs()) {
8470 if (Attr.isStringAttribute() &&
8471 (Attr.getKindAsString() == "target-features")) {
8472 if (Attr.getValueAsString().contains(Other: "-sse"))
8473 AMD64FpEndOffset = AMD64FpEndOffsetNoSSE;
8474 break;
8475 }
8476 }
8477 }
8478
8479 ArgKind classifyArgument(Value *arg) {
8480 // A very rough approximation of X86_64 argument classification rules.
8481 Type *T = arg->getType();
8482 if (T->isX86_FP80Ty())
8483 return AK_Memory;
8484 if (T->isFPOrFPVectorTy())
8485 return AK_FloatingPoint;
8486 if (T->isIntegerTy() && T->getPrimitiveSizeInBits() <= 64)
8487 return AK_GeneralPurpose;
8488 if (T->isPointerTy())
8489 return AK_GeneralPurpose;
8490 return AK_Memory;
8491 }
8492
8493 // For VarArg functions, store the argument shadow in an ABI-specific format
8494 // that corresponds to va_list layout.
8495 // We do this because Clang lowers va_arg in the frontend, and this pass
8496 // only sees the low level code that deals with va_list internals.
8497 // A much easier alternative (provided that Clang emits va_arg instructions)
8498 // would have been to associate each live instance of va_list with a copy of
8499 // MSanParamTLS, and extract shadow on va_arg() call in the argument list
8500 // order.
8501 void visitCallBase(CallBase &CB, IRBuilder<> &IRB) override {
8502 unsigned GpOffset = 0;
8503 unsigned FpOffset = AMD64GpEndOffset;
8504 unsigned OverflowOffset = AMD64FpEndOffset;
8505 const DataLayout &DL = F.getDataLayout();
8506
8507 for (const auto &[ArgNo, A] : llvm::enumerate(First: CB.args())) {
8508 bool IsFixed = ArgNo < CB.getFunctionType()->getNumParams();
8509 bool IsByVal = CB.isByValArgument(ArgNo);
8510 if (IsByVal) {
8511 // ByVal arguments always go to the overflow area.
8512 // Fixed arguments passed through the overflow area will be stepped
8513 // over by va_start, so don't count them towards the offset.
8514 if (IsFixed)
8515 continue;
8516 assert(A->getType()->isPointerTy());
8517 Type *RealTy = CB.getParamByValType(ArgNo);
8518 uint64_t ArgSize = DL.getTypeAllocSize(Ty: RealTy);
8519 uint64_t AlignedSize = alignTo(Value: ArgSize, Align: 8);
8520 unsigned BaseOffset = OverflowOffset;
8521 Value *ShadowBase = getShadowPtrForVAArgument(IRB, ArgOffset: OverflowOffset);
8522 Value *OriginBase = nullptr;
8523 if (MS.TrackOrigins)
8524 OriginBase = getOriginPtrForVAArgument(IRB, ArgOffset: OverflowOffset);
8525 OverflowOffset += AlignedSize;
8526
8527 if (OverflowOffset > kParamTLSSize) {
8528 CleanUnusedTLS(IRB, ShadowBase, BaseOffset);
8529 continue; // We have no space to copy shadow there.
8530 }
8531
8532 Value *ShadowPtr, *OriginPtr;
8533 std::tie(args&: ShadowPtr, args&: OriginPtr) =
8534 MSV.getShadowOriginPtr(Addr: A, IRB, ShadowTy: IRB.getInt8Ty(), Alignment: kShadowTLSAlignment,
8535 /*isStore*/ false);
8536 IRB.CreateMemCpy(Dst: ShadowBase, DstAlign: kShadowTLSAlignment, Src: ShadowPtr,
8537 SrcAlign: kShadowTLSAlignment, Size: ArgSize);
8538 if (MS.TrackOrigins)
8539 IRB.CreateMemCpy(Dst: OriginBase, DstAlign: kShadowTLSAlignment, Src: OriginPtr,
8540 SrcAlign: kShadowTLSAlignment, Size: ArgSize);
8541 } else {
8542 ArgKind AK = classifyArgument(arg: A);
8543 if (AK == AK_GeneralPurpose && GpOffset >= AMD64GpEndOffset)
8544 AK = AK_Memory;
8545 if (AK == AK_FloatingPoint && FpOffset >= AMD64FpEndOffset)
8546 AK = AK_Memory;
8547 Value *ShadowBase, *OriginBase = nullptr;
8548 switch (AK) {
8549 case AK_GeneralPurpose:
8550 ShadowBase = getShadowPtrForVAArgument(IRB, ArgOffset: GpOffset);
8551 if (MS.TrackOrigins)
8552 OriginBase = getOriginPtrForVAArgument(IRB, ArgOffset: GpOffset);
8553 GpOffset += 8;
8554 assert(GpOffset <= kParamTLSSize);
8555 break;
8556 case AK_FloatingPoint:
8557 ShadowBase = getShadowPtrForVAArgument(IRB, ArgOffset: FpOffset);
8558 if (MS.TrackOrigins)
8559 OriginBase = getOriginPtrForVAArgument(IRB, ArgOffset: FpOffset);
8560 FpOffset += 16;
8561 assert(FpOffset <= kParamTLSSize);
8562 break;
8563 case AK_Memory:
8564 if (IsFixed)
8565 continue;
8566 uint64_t ArgSize = DL.getTypeAllocSize(Ty: A->getType());
8567 uint64_t AlignedSize = alignTo(Value: ArgSize, Align: 8);
8568 unsigned BaseOffset = OverflowOffset;
8569 ShadowBase = getShadowPtrForVAArgument(IRB, ArgOffset: OverflowOffset);
8570 if (MS.TrackOrigins) {
8571 OriginBase = getOriginPtrForVAArgument(IRB, ArgOffset: OverflowOffset);
8572 }
8573 OverflowOffset += AlignedSize;
8574 if (OverflowOffset > kParamTLSSize) {
8575 // We have no space to copy shadow there.
8576 CleanUnusedTLS(IRB, ShadowBase, BaseOffset);
8577 continue;
8578 }
8579 }
8580 // Take fixed arguments into account for GpOffset and FpOffset,
8581 // but don't actually store shadows for them.
8582 // TODO(glider): don't call get*PtrForVAArgument() for them.
8583 if (IsFixed)
8584 continue;
8585 Value *Shadow = MSV.getShadow(V: A);
8586 IRB.CreateAlignedStore(Val: Shadow, Ptr: ShadowBase, Align: kShadowTLSAlignment);
8587 if (MS.TrackOrigins) {
8588 Value *Origin = MSV.getOrigin(V: A);
8589 TypeSize StoreSize = DL.getTypeStoreSize(Ty: Shadow->getType());
8590 MSV.paintOrigin(IRB, Origin, OriginPtr: OriginBase, TS: StoreSize,
8591 Alignment: std::max(a: kShadowTLSAlignment, b: kMinOriginAlignment));
8592 }
8593 }
8594 }
8595 Constant *OverflowSize =
8596 ConstantInt::get(Ty: IRB.getInt64Ty(), V: OverflowOffset - AMD64FpEndOffset);
8597 IRB.CreateStore(Val: OverflowSize, Ptr: MS.VAArgOverflowSizeTLS);
8598 }
8599
8600 void finalizeInstrumentation() override {
8601 assert(!VAArgOverflowSize && !VAArgTLSCopy &&
8602 "finalizeInstrumentation called twice");
8603 if (!VAStartInstrumentationList.empty()) {
8604 // If there is a va_start in this function, make a backup copy of
8605 // va_arg_tls somewhere in the function entry block.
8606 IRBuilder<> IRB(MSV.FnPrologueEnd);
8607 VAArgOverflowSize =
8608 IRB.CreateLoad(Ty: IRB.getInt64Ty(), Ptr: MS.VAArgOverflowSizeTLS);
8609 Value *CopySize = IRB.CreateAdd(
8610 LHS: ConstantInt::get(Ty: MS.IntptrTy, V: AMD64FpEndOffset), RHS: VAArgOverflowSize);
8611 VAArgTLSCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
8612 VAArgTLSCopy->setAlignment(kShadowTLSAlignment);
8613 IRB.CreateMemSet(Ptr: VAArgTLSCopy, Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
8614 Size: CopySize, Align: kShadowTLSAlignment, isVolatile: false);
8615
8616 Value *SrcSize = IRB.CreateBinaryIntrinsic(
8617 ID: Intrinsic::umin, LHS: CopySize,
8618 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: kParamTLSSize));
8619 IRB.CreateMemCpy(Dst: VAArgTLSCopy, DstAlign: kShadowTLSAlignment, Src: MS.VAArgTLS,
8620 SrcAlign: kShadowTLSAlignment, Size: SrcSize);
8621 if (MS.TrackOrigins) {
8622 VAArgTLSOriginCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
8623 VAArgTLSOriginCopy->setAlignment(kShadowTLSAlignment);
8624 IRB.CreateMemCpy(Dst: VAArgTLSOriginCopy, DstAlign: kShadowTLSAlignment,
8625 Src: MS.VAArgOriginTLS, SrcAlign: kShadowTLSAlignment, Size: SrcSize);
8626 }
8627 }
8628
8629 // Instrument va_start.
8630 // Copy va_list shadow from the backup copy of the TLS contents.
8631 for (CallInst *OrigInst : VAStartInstrumentationList) {
8632 NextNodeIRBuilder IRB(OrigInst);
8633 Value *VAListTag = OrigInst->getArgOperand(i: 0);
8634
8635 Value *RegSaveAreaPtrPtr =
8636 IRB.CreatePtrAdd(Ptr: VAListTag, Offset: ConstantInt::get(Ty: MS.IntptrTy, V: 16));
8637 Value *RegSaveAreaPtr = IRB.CreateLoad(Ty: MS.PtrTy, Ptr: RegSaveAreaPtrPtr);
8638 Value *RegSaveAreaShadowPtr, *RegSaveAreaOriginPtr;
8639 const Align Alignment = Align(16);
8640 std::tie(args&: RegSaveAreaShadowPtr, args&: RegSaveAreaOriginPtr) =
8641 MSV.getShadowOriginPtr(Addr: RegSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
8642 Alignment, /*isStore*/ true);
8643 IRB.CreateMemCpy(Dst: RegSaveAreaShadowPtr, DstAlign: Alignment, Src: VAArgTLSCopy, SrcAlign: Alignment,
8644 Size: AMD64FpEndOffset);
8645 if (MS.TrackOrigins)
8646 IRB.CreateMemCpy(Dst: RegSaveAreaOriginPtr, DstAlign: Alignment, Src: VAArgTLSOriginCopy,
8647 SrcAlign: Alignment, Size: AMD64FpEndOffset);
8648 Value *OverflowArgAreaPtrPtr =
8649 IRB.CreatePtrAdd(Ptr: VAListTag, Offset: ConstantInt::get(Ty: MS.IntptrTy, V: 8));
8650 Value *OverflowArgAreaPtr =
8651 IRB.CreateLoad(Ty: MS.PtrTy, Ptr: OverflowArgAreaPtrPtr);
8652 Value *OverflowArgAreaShadowPtr, *OverflowArgAreaOriginPtr;
8653 std::tie(args&: OverflowArgAreaShadowPtr, args&: OverflowArgAreaOriginPtr) =
8654 MSV.getShadowOriginPtr(Addr: OverflowArgAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
8655 Alignment, /*isStore*/ true);
8656 Value *SrcPtr = IRB.CreateConstGEP1_32(Ty: IRB.getInt8Ty(), Ptr: VAArgTLSCopy,
8657 Idx0: AMD64FpEndOffset);
8658 IRB.CreateMemCpy(Dst: OverflowArgAreaShadowPtr, DstAlign: Alignment, Src: SrcPtr, SrcAlign: Alignment,
8659 Size: VAArgOverflowSize);
8660 if (MS.TrackOrigins) {
8661 SrcPtr = IRB.CreateConstGEP1_32(Ty: IRB.getInt8Ty(), Ptr: VAArgTLSOriginCopy,
8662 Idx0: AMD64FpEndOffset);
8663 IRB.CreateMemCpy(Dst: OverflowArgAreaOriginPtr, DstAlign: Alignment, Src: SrcPtr, SrcAlign: Alignment,
8664 Size: VAArgOverflowSize);
8665 }
8666 }
8667 }
8668};
8669
8670/// AArch64-specific implementation of VarArgHelper.
8671struct VarArgAArch64Helper : public VarArgHelperBase {
8672 static const unsigned kAArch64GrArgSize = 64;
8673 static const unsigned kAArch64VrArgSize = 128;
8674
8675 static const unsigned AArch64GrBegOffset = 0;
8676 static const unsigned AArch64GrEndOffset = kAArch64GrArgSize;
8677 // Make VR space aligned to 16 bytes.
8678 static const unsigned AArch64VrBegOffset = AArch64GrEndOffset;
8679 static const unsigned AArch64VrEndOffset =
8680 AArch64VrBegOffset + kAArch64VrArgSize;
8681 static const unsigned AArch64VAEndOffset = AArch64VrEndOffset;
8682
8683 AllocaInst *VAArgTLSCopy = nullptr;
8684 Value *VAArgOverflowSize = nullptr;
8685
8686 enum ArgKind { AK_GeneralPurpose, AK_FloatingPoint, AK_Memory };
8687
8688 VarArgAArch64Helper(Function &F, MemorySanitizer &MS,
8689 MemorySanitizerVisitor &MSV)
8690 : VarArgHelperBase(F, MS, MSV, /*VAListTagSize=*/32) {}
8691
8692 // A very rough approximation of aarch64 argument classification rules.
8693 std::pair<ArgKind, uint64_t> classifyArgument(Type *T) {
8694 if (T->isIntOrPtrTy() && T->getPrimitiveSizeInBits() <= 64)
8695 return {AK_GeneralPurpose, 1};
8696 if (T->isFloatingPointTy() && T->getPrimitiveSizeInBits() <= 128)
8697 return {AK_FloatingPoint, 1};
8698
8699 if (T->isArrayTy()) {
8700 auto R = classifyArgument(T: T->getArrayElementType());
8701 R.second *= T->getScalarType()->getArrayNumElements();
8702 return R;
8703 }
8704
8705 if (const FixedVectorType *FV = dyn_cast<FixedVectorType>(Val: T)) {
8706 auto R = classifyArgument(T: FV->getScalarType());
8707 R.second *= FV->getNumElements();
8708 return R;
8709 }
8710
8711 LLVM_DEBUG(errs() << "Unknown vararg type: " << *T << "\n");
8712 return {AK_Memory, 0};
8713 }
8714
8715 // The instrumentation stores the argument shadow in a non ABI-specific
8716 // format because it does not know which argument is named (since Clang,
8717 // like x86_64 case, lowers the va_args in the frontend and this pass only
8718 // sees the low level code that deals with va_list internals).
8719 // The first seven GR registers are saved in the first 56 bytes of the
8720 // va_arg tls arra, followed by the first 8 FP/SIMD registers, and then
8721 // the remaining arguments.
8722 // Using constant offset within the va_arg TLS array allows fast copy
8723 // in the finalize instrumentation.
8724 void visitCallBase(CallBase &CB, IRBuilder<> &IRB) override {
8725 unsigned GrOffset = AArch64GrBegOffset;
8726 unsigned VrOffset = AArch64VrBegOffset;
8727 unsigned OverflowOffset = AArch64VAEndOffset;
8728
8729 const DataLayout &DL = F.getDataLayout();
8730 for (const auto &[ArgNo, A] : llvm::enumerate(First: CB.args())) {
8731 bool IsFixed = ArgNo < CB.getFunctionType()->getNumParams();
8732 auto [AK, RegNum] = classifyArgument(T: A->getType());
8733 if (AK == AK_GeneralPurpose &&
8734 (GrOffset + RegNum * 8) > AArch64GrEndOffset)
8735 AK = AK_Memory;
8736 if (AK == AK_FloatingPoint &&
8737 (VrOffset + RegNum * 16) > AArch64VrEndOffset)
8738 AK = AK_Memory;
8739 Value *Base;
8740 switch (AK) {
8741 case AK_GeneralPurpose:
8742 Base = getShadowPtrForVAArgument(IRB, ArgOffset: GrOffset);
8743 GrOffset += 8 * RegNum;
8744 break;
8745 case AK_FloatingPoint:
8746 Base = getShadowPtrForVAArgument(IRB, ArgOffset: VrOffset);
8747 VrOffset += 16 * RegNum;
8748 break;
8749 case AK_Memory:
8750 // Don't count fixed arguments in the overflow area - va_start will
8751 // skip right over them.
8752 if (IsFixed)
8753 continue;
8754 uint64_t ArgSize = DL.getTypeAllocSize(Ty: A->getType());
8755 uint64_t AlignedSize = alignTo(Value: ArgSize, Align: 8);
8756 unsigned BaseOffset = OverflowOffset;
8757 Base = getShadowPtrForVAArgument(IRB, ArgOffset: BaseOffset);
8758 OverflowOffset += AlignedSize;
8759 if (OverflowOffset > kParamTLSSize) {
8760 // We have no space to copy shadow there.
8761 CleanUnusedTLS(IRB, ShadowBase: Base, BaseOffset);
8762 continue;
8763 }
8764 break;
8765 }
8766 // Count Gp/Vr fixed arguments to their respective offsets, but don't
8767 // bother to actually store a shadow.
8768 if (IsFixed)
8769 continue;
8770 IRB.CreateAlignedStore(Val: MSV.getShadow(V: A), Ptr: Base, Align: kShadowTLSAlignment);
8771 }
8772 Constant *OverflowSize =
8773 ConstantInt::get(Ty: IRB.getInt64Ty(), V: OverflowOffset - AArch64VAEndOffset);
8774 IRB.CreateStore(Val: OverflowSize, Ptr: MS.VAArgOverflowSizeTLS);
8775 }
8776
8777 // Retrieve a va_list field of 'void*' size.
8778 Value *getVAField64(IRBuilder<> &IRB, Value *VAListTag, int offset) {
8779 Value *SaveAreaPtrPtr =
8780 IRB.CreatePtrAdd(Ptr: VAListTag, Offset: ConstantInt::get(Ty: MS.IntptrTy, V: offset));
8781 return IRB.CreateLoad(Ty: Type::getInt64Ty(C&: *MS.C), Ptr: SaveAreaPtrPtr);
8782 }
8783
8784 // Retrieve a va_list field of 'int' size.
8785 Value *getVAField32(IRBuilder<> &IRB, Value *VAListTag, int offset) {
8786 Value *SaveAreaPtr =
8787 IRB.CreatePtrAdd(Ptr: VAListTag, Offset: ConstantInt::get(Ty: MS.IntptrTy, V: offset));
8788 Value *SaveArea32 = IRB.CreateLoad(Ty: IRB.getInt32Ty(), Ptr: SaveAreaPtr);
8789 return IRB.CreateSExt(V: SaveArea32, DestTy: MS.IntptrTy);
8790 }
8791
8792 void finalizeInstrumentation() override {
8793 assert(!VAArgOverflowSize && !VAArgTLSCopy &&
8794 "finalizeInstrumentation called twice");
8795 if (!VAStartInstrumentationList.empty()) {
8796 // If there is a va_start in this function, make a backup copy of
8797 // va_arg_tls somewhere in the function entry block.
8798 IRBuilder<> IRB(MSV.FnPrologueEnd);
8799 VAArgOverflowSize =
8800 IRB.CreateLoad(Ty: IRB.getInt64Ty(), Ptr: MS.VAArgOverflowSizeTLS);
8801 Value *CopySize = IRB.CreateAdd(
8802 LHS: ConstantInt::get(Ty: MS.IntptrTy, V: AArch64VAEndOffset), RHS: VAArgOverflowSize);
8803 VAArgTLSCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
8804 VAArgTLSCopy->setAlignment(kShadowTLSAlignment);
8805 IRB.CreateMemSet(Ptr: VAArgTLSCopy, Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
8806 Size: CopySize, Align: kShadowTLSAlignment, isVolatile: false);
8807
8808 Value *SrcSize = IRB.CreateBinaryIntrinsic(
8809 ID: Intrinsic::umin, LHS: CopySize,
8810 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: kParamTLSSize));
8811 IRB.CreateMemCpy(Dst: VAArgTLSCopy, DstAlign: kShadowTLSAlignment, Src: MS.VAArgTLS,
8812 SrcAlign: kShadowTLSAlignment, Size: SrcSize);
8813 }
8814
8815 Value *GrArgSize = ConstantInt::get(Ty: MS.IntptrTy, V: kAArch64GrArgSize);
8816 Value *VrArgSize = ConstantInt::get(Ty: MS.IntptrTy, V: kAArch64VrArgSize);
8817
8818 // Instrument va_start, copy va_list shadow from the backup copy of
8819 // the TLS contents.
8820 for (CallInst *OrigInst : VAStartInstrumentationList) {
8821 NextNodeIRBuilder IRB(OrigInst);
8822
8823 Value *VAListTag = OrigInst->getArgOperand(i: 0);
8824
8825 // The variadic ABI for AArch64 creates two areas to save the incoming
8826 // argument registers (one for 64-bit general register xn-x7 and another
8827 // for 128-bit FP/SIMD vn-v7).
8828 // We need then to propagate the shadow arguments on both regions
8829 // 'va::__gr_top + va::__gr_offs' and 'va::__vr_top + va::__vr_offs'.
8830 // The remaining arguments are saved on shadow for 'va::stack'.
8831 // One caveat is it requires only to propagate the non-named arguments,
8832 // however on the call site instrumentation 'all' the arguments are
8833 // saved. So to copy the shadow values from the va_arg TLS array
8834 // we need to adjust the offset for both GR and VR fields based on
8835 // the __{gr,vr}_offs value (since they are stores based on incoming
8836 // named arguments).
8837 Type *RegSaveAreaPtrTy = IRB.getPtrTy();
8838
8839 // Read the stack pointer from the va_list.
8840 Value *StackSaveAreaPtr =
8841 IRB.CreateIntToPtr(V: getVAField64(IRB, VAListTag, offset: 0), DestTy: RegSaveAreaPtrTy);
8842
8843 // Read both the __gr_top and __gr_off and add them up.
8844 Value *GrTopSaveAreaPtr = getVAField64(IRB, VAListTag, offset: 8);
8845 Value *GrOffSaveArea = getVAField32(IRB, VAListTag, offset: 24);
8846
8847 Value *GrRegSaveAreaPtr = IRB.CreateIntToPtr(
8848 V: IRB.CreateAdd(LHS: GrTopSaveAreaPtr, RHS: GrOffSaveArea), DestTy: RegSaveAreaPtrTy);
8849
8850 // Read both the __vr_top and __vr_off and add them up.
8851 Value *VrTopSaveAreaPtr = getVAField64(IRB, VAListTag, offset: 16);
8852 Value *VrOffSaveArea = getVAField32(IRB, VAListTag, offset: 28);
8853
8854 Value *VrRegSaveAreaPtr = IRB.CreateIntToPtr(
8855 V: IRB.CreateAdd(LHS: VrTopSaveAreaPtr, RHS: VrOffSaveArea), DestTy: RegSaveAreaPtrTy);
8856
8857 // It does not know how many named arguments is being used and, on the
8858 // callsite all the arguments were saved. Since __gr_off is defined as
8859 // '0 - ((8 - named_gr) * 8)', the idea is to just propagate the variadic
8860 // argument by ignoring the bytes of shadow from named arguments.
8861 Value *GrRegSaveAreaShadowPtrOff =
8862 IRB.CreateAdd(LHS: GrArgSize, RHS: GrOffSaveArea);
8863
8864 Value *GrRegSaveAreaShadowPtr =
8865 MSV.getShadowOriginPtr(Addr: GrRegSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
8866 Alignment: Align(8), /*isStore*/ true)
8867 .first;
8868
8869 Value *GrSrcPtr =
8870 IRB.CreateInBoundsPtrAdd(Ptr: VAArgTLSCopy, Offset: GrRegSaveAreaShadowPtrOff);
8871 Value *GrCopySize = IRB.CreateSub(LHS: GrArgSize, RHS: GrRegSaveAreaShadowPtrOff);
8872
8873 IRB.CreateMemCpy(Dst: GrRegSaveAreaShadowPtr, DstAlign: Align(8), Src: GrSrcPtr, SrcAlign: Align(8),
8874 Size: GrCopySize);
8875
8876 // Again, but for FP/SIMD values.
8877 Value *VrRegSaveAreaShadowPtrOff =
8878 IRB.CreateAdd(LHS: VrArgSize, RHS: VrOffSaveArea);
8879
8880 Value *VrRegSaveAreaShadowPtr =
8881 MSV.getShadowOriginPtr(Addr: VrRegSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
8882 Alignment: Align(8), /*isStore*/ true)
8883 .first;
8884
8885 Value *VrSrcPtr = IRB.CreateInBoundsPtrAdd(
8886 Ptr: IRB.CreateInBoundsPtrAdd(Ptr: VAArgTLSCopy,
8887 Offset: IRB.getInt32(C: AArch64VrBegOffset)),
8888 Offset: VrRegSaveAreaShadowPtrOff);
8889 Value *VrCopySize = IRB.CreateSub(LHS: VrArgSize, RHS: VrRegSaveAreaShadowPtrOff);
8890
8891 IRB.CreateMemCpy(Dst: VrRegSaveAreaShadowPtr, DstAlign: Align(8), Src: VrSrcPtr, SrcAlign: Align(8),
8892 Size: VrCopySize);
8893
8894 // And finally for remaining arguments.
8895 Value *StackSaveAreaShadowPtr =
8896 MSV.getShadowOriginPtr(Addr: StackSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
8897 Alignment: Align(16), /*isStore*/ true)
8898 .first;
8899
8900 Value *StackSrcPtr = IRB.CreateInBoundsPtrAdd(
8901 Ptr: VAArgTLSCopy, Offset: IRB.getInt32(C: AArch64VAEndOffset));
8902
8903 IRB.CreateMemCpy(Dst: StackSaveAreaShadowPtr, DstAlign: Align(16), Src: StackSrcPtr,
8904 SrcAlign: Align(16), Size: VAArgOverflowSize);
8905 }
8906 }
8907};
8908
8909/// PowerPC64-specific implementation of VarArgHelper.
8910struct VarArgPowerPC64Helper : public VarArgHelperBase {
8911 AllocaInst *VAArgTLSCopy = nullptr;
8912 Value *VAArgSize = nullptr;
8913
8914 VarArgPowerPC64Helper(Function &F, MemorySanitizer &MS,
8915 MemorySanitizerVisitor &MSV)
8916 : VarArgHelperBase(F, MS, MSV, /*VAListTagSize=*/8) {}
8917
8918 void visitCallBase(CallBase &CB, IRBuilder<> &IRB) override {
8919 // For PowerPC, we need to deal with alignment of stack arguments -
8920 // they are mostly aligned to 8 bytes, but vectors and i128 arrays
8921 // are aligned to 16 bytes, byvals can be aligned to 8 or 16 bytes,
8922 // For that reason, we compute current offset from stack pointer (which is
8923 // always properly aligned), and offset for the first vararg, then subtract
8924 // them.
8925 unsigned VAArgBase;
8926 Triple TargetTriple(F.getParent()->getTargetTriple());
8927 // Parameter save area starts at 48 bytes from frame pointer for ABIv1,
8928 // and 32 bytes for ABIv2. This is usually determined by target
8929 // endianness, but in theory could be overridden by function attribute.
8930 if (TargetTriple.isPPC64ELFv2ABI())
8931 VAArgBase = 32;
8932 else
8933 VAArgBase = 48;
8934 unsigned VAArgOffset = VAArgBase;
8935 const DataLayout &DL = F.getDataLayout();
8936 for (const auto &[ArgNo, A] : llvm::enumerate(First: CB.args())) {
8937 bool IsFixed = ArgNo < CB.getFunctionType()->getNumParams();
8938 bool IsByVal = CB.isByValArgument(ArgNo);
8939 if (IsByVal) {
8940 assert(A->getType()->isPointerTy());
8941 Type *RealTy = CB.getParamByValType(ArgNo);
8942 uint64_t ArgSize = DL.getTypeAllocSize(Ty: RealTy);
8943 Align ArgAlign = CB.getParamAlign(ArgNo).value_or(u: Align(8));
8944 if (ArgAlign < 8)
8945 ArgAlign = Align(8);
8946 VAArgOffset = alignTo(Size: VAArgOffset, A: ArgAlign);
8947 if (!IsFixed) {
8948 Value *Base =
8949 getShadowPtrForVAArgument(IRB, ArgOffset: VAArgOffset - VAArgBase, ArgSize);
8950 if (Base) {
8951 Value *AShadowPtr, *AOriginPtr;
8952 std::tie(args&: AShadowPtr, args&: AOriginPtr) =
8953 MSV.getShadowOriginPtr(Addr: A, IRB, ShadowTy: IRB.getInt8Ty(),
8954 Alignment: kShadowTLSAlignment, /*isStore*/ false);
8955
8956 IRB.CreateMemCpy(Dst: Base, DstAlign: kShadowTLSAlignment, Src: AShadowPtr,
8957 SrcAlign: kShadowTLSAlignment, Size: ArgSize);
8958 }
8959 }
8960 VAArgOffset += alignTo(Size: ArgSize, A: Align(8));
8961 } else {
8962 Value *Base;
8963 uint64_t ArgSize = DL.getTypeAllocSize(Ty: A->getType());
8964 Align ArgAlign = Align(8);
8965 if (A->getType()->isArrayTy()) {
8966 // Arrays are aligned to element size, except for long double
8967 // arrays, which are aligned to 8 bytes.
8968 Type *ElementTy = A->getType()->getArrayElementType();
8969 if (!ElementTy->isPPC_FP128Ty())
8970 ArgAlign = Align(DL.getTypeAllocSize(Ty: ElementTy));
8971 } else if (A->getType()->isVectorTy()) {
8972 // Vectors are naturally aligned.
8973 ArgAlign = Align(ArgSize);
8974 }
8975 if (ArgAlign < 8)
8976 ArgAlign = Align(8);
8977 VAArgOffset = alignTo(Size: VAArgOffset, A: ArgAlign);
8978 if (DL.isBigEndian()) {
8979 // Adjusting the shadow for argument with size < 8 to match the
8980 // placement of bits in big endian system
8981 if (ArgSize < 8)
8982 VAArgOffset += (8 - ArgSize);
8983 }
8984 if (!IsFixed) {
8985 Base =
8986 getShadowPtrForVAArgument(IRB, ArgOffset: VAArgOffset - VAArgBase, ArgSize);
8987 if (Base)
8988 IRB.CreateAlignedStore(Val: MSV.getShadow(V: A), Ptr: Base, Align: kShadowTLSAlignment);
8989 }
8990 VAArgOffset += ArgSize;
8991 VAArgOffset = alignTo(Size: VAArgOffset, A: Align(8));
8992 }
8993 if (IsFixed)
8994 VAArgBase = VAArgOffset;
8995 }
8996
8997 Constant *TotalVAArgSize =
8998 ConstantInt::get(Ty: MS.IntptrTy, V: VAArgOffset - VAArgBase);
8999 // Here using VAArgOverflowSizeTLS as VAArgSizeTLS to avoid creation of
9000 // a new class member i.e. it is the total size of all VarArgs.
9001 IRB.CreateStore(Val: TotalVAArgSize, Ptr: MS.VAArgOverflowSizeTLS);
9002 }
9003
9004 void finalizeInstrumentation() override {
9005 assert(!VAArgSize && !VAArgTLSCopy &&
9006 "finalizeInstrumentation called twice");
9007 IRBuilder<> IRB(MSV.FnPrologueEnd);
9008 VAArgSize = IRB.CreateLoad(Ty: IRB.getInt64Ty(), Ptr: MS.VAArgOverflowSizeTLS);
9009 Value *CopySize = VAArgSize;
9010
9011 if (!VAStartInstrumentationList.empty()) {
9012 // If there is a va_start in this function, make a backup copy of
9013 // va_arg_tls somewhere in the function entry block.
9014
9015 VAArgTLSCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
9016 VAArgTLSCopy->setAlignment(kShadowTLSAlignment);
9017 IRB.CreateMemSet(Ptr: VAArgTLSCopy, Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
9018 Size: CopySize, Align: kShadowTLSAlignment, isVolatile: false);
9019
9020 Value *SrcSize = IRB.CreateBinaryIntrinsic(
9021 ID: Intrinsic::umin, LHS: CopySize,
9022 RHS: ConstantInt::get(Ty: IRB.getInt64Ty(), V: kParamTLSSize));
9023 IRB.CreateMemCpy(Dst: VAArgTLSCopy, DstAlign: kShadowTLSAlignment, Src: MS.VAArgTLS,
9024 SrcAlign: kShadowTLSAlignment, Size: SrcSize);
9025 }
9026
9027 // Instrument va_start.
9028 // Copy va_list shadow from the backup copy of the TLS contents.
9029 for (CallInst *OrigInst : VAStartInstrumentationList) {
9030 NextNodeIRBuilder IRB(OrigInst);
9031 Value *VAListTag = OrigInst->getArgOperand(i: 0);
9032 Value *RegSaveAreaPtrPtr = IRB.CreatePtrToInt(V: VAListTag, DestTy: MS.IntptrTy);
9033
9034 RegSaveAreaPtrPtr = IRB.CreateIntToPtr(V: RegSaveAreaPtrPtr, DestTy: MS.PtrTy);
9035
9036 Value *RegSaveAreaPtr = IRB.CreateLoad(Ty: MS.PtrTy, Ptr: RegSaveAreaPtrPtr);
9037 Value *RegSaveAreaShadowPtr, *RegSaveAreaOriginPtr;
9038 const DataLayout &DL = F.getDataLayout();
9039 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
9040 const Align Alignment = Align(IntptrSize);
9041 std::tie(args&: RegSaveAreaShadowPtr, args&: RegSaveAreaOriginPtr) =
9042 MSV.getShadowOriginPtr(Addr: RegSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
9043 Alignment, /*isStore*/ true);
9044 IRB.CreateMemCpy(Dst: RegSaveAreaShadowPtr, DstAlign: Alignment, Src: VAArgTLSCopy, SrcAlign: Alignment,
9045 Size: CopySize);
9046 }
9047 }
9048};
9049
9050/// PowerPC32-specific implementation of VarArgHelper.
9051struct VarArgPowerPC32Helper : public VarArgHelperBase {
9052 AllocaInst *VAArgTLSCopy = nullptr;
9053 Value *VAArgSize = nullptr;
9054
9055 VarArgPowerPC32Helper(Function &F, MemorySanitizer &MS,
9056 MemorySanitizerVisitor &MSV)
9057 : VarArgHelperBase(F, MS, MSV, /*VAListTagSize=*/12) {}
9058
9059 void visitCallBase(CallBase &CB, IRBuilder<> &IRB) override {
9060 unsigned VAArgBase;
9061 // Parameter save area is 8 bytes from frame pointer in PPC32
9062 VAArgBase = 8;
9063 unsigned VAArgOffset = VAArgBase;
9064 const DataLayout &DL = F.getDataLayout();
9065 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
9066 for (const auto &[ArgNo, A] : llvm::enumerate(First: CB.args())) {
9067 bool IsFixed = ArgNo < CB.getFunctionType()->getNumParams();
9068 bool IsByVal = CB.isByValArgument(ArgNo);
9069 if (IsByVal) {
9070 assert(A->getType()->isPointerTy());
9071 Type *RealTy = CB.getParamByValType(ArgNo);
9072 uint64_t ArgSize = DL.getTypeAllocSize(Ty: RealTy);
9073 Align ArgAlign = CB.getParamAlign(ArgNo).value_or(u: Align(IntptrSize));
9074 if (ArgAlign < IntptrSize)
9075 ArgAlign = Align(IntptrSize);
9076 VAArgOffset = alignTo(Size: VAArgOffset, A: ArgAlign);
9077 if (!IsFixed) {
9078 Value *Base =
9079 getShadowPtrForVAArgument(IRB, ArgOffset: VAArgOffset - VAArgBase, ArgSize);
9080 if (Base) {
9081 Value *AShadowPtr, *AOriginPtr;
9082 std::tie(args&: AShadowPtr, args&: AOriginPtr) =
9083 MSV.getShadowOriginPtr(Addr: A, IRB, ShadowTy: IRB.getInt8Ty(),
9084 Alignment: kShadowTLSAlignment, /*isStore*/ false);
9085
9086 IRB.CreateMemCpy(Dst: Base, DstAlign: kShadowTLSAlignment, Src: AShadowPtr,
9087 SrcAlign: kShadowTLSAlignment, Size: ArgSize);
9088 }
9089 }
9090 VAArgOffset += alignTo(Size: ArgSize, A: Align(IntptrSize));
9091 } else {
9092 Value *Base;
9093 Type *ArgTy = A->getType();
9094
9095 // On PPC 32 floating point variable arguments are stored in separate
9096 // area: fp_save_area = reg_save_area + 4*8. We do not copy shaodow for
9097 // them as they will be found when checking call arguments.
9098 if (!ArgTy->isFloatingPointTy()) {
9099 uint64_t ArgSize = DL.getTypeAllocSize(Ty: ArgTy);
9100 Align ArgAlign = Align(IntptrSize);
9101 if (ArgTy->isArrayTy()) {
9102 // Arrays are aligned to element size, except for long double
9103 // arrays, which are aligned to 8 bytes.
9104 Type *ElementTy = ArgTy->getArrayElementType();
9105 if (!ElementTy->isPPC_FP128Ty())
9106 ArgAlign = Align(DL.getTypeAllocSize(Ty: ElementTy));
9107 } else if (ArgTy->isVectorTy()) {
9108 // Vectors are naturally aligned.
9109 ArgAlign = Align(ArgSize);
9110 }
9111 if (ArgAlign < IntptrSize)
9112 ArgAlign = Align(IntptrSize);
9113 VAArgOffset = alignTo(Size: VAArgOffset, A: ArgAlign);
9114 if (DL.isBigEndian()) {
9115 // Adjusting the shadow for argument with size < IntptrSize to match
9116 // the placement of bits in big endian system
9117 if (ArgSize < IntptrSize)
9118 VAArgOffset += (IntptrSize - ArgSize);
9119 }
9120 if (!IsFixed) {
9121 Base = getShadowPtrForVAArgument(IRB, ArgOffset: VAArgOffset - VAArgBase,
9122 ArgSize);
9123 if (Base)
9124 IRB.CreateAlignedStore(Val: MSV.getShadow(V: A), Ptr: Base,
9125 Align: kShadowTLSAlignment);
9126 }
9127 VAArgOffset += ArgSize;
9128 VAArgOffset = alignTo(Size: VAArgOffset, A: Align(IntptrSize));
9129 }
9130 }
9131 }
9132
9133 Constant *TotalVAArgSize =
9134 ConstantInt::get(Ty: MS.IntptrTy, V: VAArgOffset - VAArgBase);
9135 // Here using VAArgOverflowSizeTLS as VAArgSizeTLS to avoid creation of
9136 // a new class member i.e. it is the total size of all VarArgs.
9137 IRB.CreateStore(Val: TotalVAArgSize, Ptr: MS.VAArgOverflowSizeTLS);
9138 }
9139
9140 void finalizeInstrumentation() override {
9141 assert(!VAArgSize && !VAArgTLSCopy &&
9142 "finalizeInstrumentation called twice");
9143 IRBuilder<> IRB(MSV.FnPrologueEnd);
9144 VAArgSize = IRB.CreateLoad(Ty: MS.IntptrTy, Ptr: MS.VAArgOverflowSizeTLS);
9145 Value *CopySize = VAArgSize;
9146
9147 if (!VAStartInstrumentationList.empty()) {
9148 // If there is a va_start in this function, make a backup copy of
9149 // va_arg_tls somewhere in the function entry block.
9150
9151 VAArgTLSCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
9152 VAArgTLSCopy->setAlignment(kShadowTLSAlignment);
9153 IRB.CreateMemSet(Ptr: VAArgTLSCopy, Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
9154 Size: CopySize, Align: kShadowTLSAlignment, isVolatile: false);
9155
9156 Value *SrcSize = IRB.CreateBinaryIntrinsic(
9157 ID: Intrinsic::umin, LHS: CopySize,
9158 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: kParamTLSSize));
9159 IRB.CreateMemCpy(Dst: VAArgTLSCopy, DstAlign: kShadowTLSAlignment, Src: MS.VAArgTLS,
9160 SrcAlign: kShadowTLSAlignment, Size: SrcSize);
9161 }
9162
9163 // Instrument va_start.
9164 // Copy va_list shadow from the backup copy of the TLS contents.
9165 for (CallInst *OrigInst : VAStartInstrumentationList) {
9166 NextNodeIRBuilder IRB(OrigInst);
9167 Value *VAListTag = OrigInst->getArgOperand(i: 0);
9168 Value *RegSaveAreaPtrPtr = IRB.CreatePtrToInt(V: VAListTag, DestTy: MS.IntptrTy);
9169 Value *RegSaveAreaSize = CopySize;
9170
9171 // In PPC32 va_list_tag is a struct
9172 RegSaveAreaPtrPtr =
9173 IRB.CreateAdd(LHS: RegSaveAreaPtrPtr, RHS: ConstantInt::get(Ty: MS.IntptrTy, V: 8));
9174
9175 // On PPC 32 reg_save_area can only hold 32 bytes of data
9176 RegSaveAreaSize = IRB.CreateBinaryIntrinsic(
9177 ID: Intrinsic::umin, LHS: CopySize, RHS: ConstantInt::get(Ty: MS.IntptrTy, V: 32));
9178
9179 RegSaveAreaPtrPtr = IRB.CreateIntToPtr(V: RegSaveAreaPtrPtr, DestTy: MS.PtrTy);
9180 Value *RegSaveAreaPtr = IRB.CreateLoad(Ty: MS.PtrTy, Ptr: RegSaveAreaPtrPtr);
9181
9182 const DataLayout &DL = F.getDataLayout();
9183 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
9184 const Align Alignment = Align(IntptrSize);
9185
9186 { // Copy reg save area
9187 Value *RegSaveAreaShadowPtr, *RegSaveAreaOriginPtr;
9188 std::tie(args&: RegSaveAreaShadowPtr, args&: RegSaveAreaOriginPtr) =
9189 MSV.getShadowOriginPtr(Addr: RegSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
9190 Alignment, /*isStore*/ true);
9191 IRB.CreateMemCpy(Dst: RegSaveAreaShadowPtr, DstAlign: Alignment, Src: VAArgTLSCopy,
9192 SrcAlign: Alignment, Size: RegSaveAreaSize);
9193
9194 RegSaveAreaShadowPtr =
9195 IRB.CreatePtrToInt(V: RegSaveAreaShadowPtr, DestTy: MS.IntptrTy);
9196 Value *FPSaveArea = IRB.CreateAdd(LHS: RegSaveAreaShadowPtr,
9197 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: 32));
9198 FPSaveArea = IRB.CreateIntToPtr(V: FPSaveArea, DestTy: MS.PtrTy);
9199 // We fill fp shadow with zeroes as uninitialized fp args should have
9200 // been found during call base check
9201 IRB.CreateMemSet(Ptr: FPSaveArea, Val: ConstantInt::getNullValue(Ty: IRB.getInt8Ty()),
9202 Size: ConstantInt::get(Ty: MS.IntptrTy, V: 32), Align: Alignment);
9203 }
9204
9205 { // Copy overflow area
9206 // RegSaveAreaSize is min(CopySize, 32) -> no overflow can occur
9207 Value *OverflowAreaSize = IRB.CreateSub(LHS: CopySize, RHS: RegSaveAreaSize);
9208
9209 Value *OverflowAreaPtrPtr = IRB.CreatePtrToInt(V: VAListTag, DestTy: MS.IntptrTy);
9210 OverflowAreaPtrPtr =
9211 IRB.CreateAdd(LHS: OverflowAreaPtrPtr, RHS: ConstantInt::get(Ty: MS.IntptrTy, V: 4));
9212 OverflowAreaPtrPtr = IRB.CreateIntToPtr(V: OverflowAreaPtrPtr, DestTy: MS.PtrTy);
9213
9214 Value *OverflowAreaPtr = IRB.CreateLoad(Ty: MS.PtrTy, Ptr: OverflowAreaPtrPtr);
9215
9216 Value *OverflowAreaShadowPtr, *OverflowAreaOriginPtr;
9217 std::tie(args&: OverflowAreaShadowPtr, args&: OverflowAreaOriginPtr) =
9218 MSV.getShadowOriginPtr(Addr: OverflowAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
9219 Alignment, /*isStore*/ true);
9220
9221 Value *OverflowVAArgTLSCopyPtr =
9222 IRB.CreatePtrToInt(V: VAArgTLSCopy, DestTy: MS.IntptrTy);
9223 OverflowVAArgTLSCopyPtr =
9224 IRB.CreateAdd(LHS: OverflowVAArgTLSCopyPtr, RHS: RegSaveAreaSize);
9225
9226 OverflowVAArgTLSCopyPtr =
9227 IRB.CreateIntToPtr(V: OverflowVAArgTLSCopyPtr, DestTy: MS.PtrTy);
9228 IRB.CreateMemCpy(Dst: OverflowAreaShadowPtr, DstAlign: Alignment,
9229 Src: OverflowVAArgTLSCopyPtr, SrcAlign: Alignment, Size: OverflowAreaSize);
9230 }
9231 }
9232 }
9233};
9234
9235/// SystemZ-specific implementation of VarArgHelper.
9236struct VarArgSystemZHelper : public VarArgHelperBase {
9237 static const unsigned SystemZGpOffset = 16;
9238 static const unsigned SystemZGpEndOffset = 56;
9239 static const unsigned SystemZFpOffset = 128;
9240 static const unsigned SystemZFpEndOffset = 160;
9241 static const unsigned SystemZMaxVrArgs = 8;
9242 static const unsigned SystemZRegSaveAreaSize = 160;
9243 static const unsigned SystemZOverflowOffset = 160;
9244 static const unsigned SystemZVAListTagSize = 32;
9245 static const unsigned SystemZOverflowArgAreaPtrOffset = 16;
9246 static const unsigned SystemZRegSaveAreaPtrOffset = 24;
9247
9248 bool IsSoftFloatABI;
9249 AllocaInst *VAArgTLSCopy = nullptr;
9250 AllocaInst *VAArgTLSOriginCopy = nullptr;
9251 Value *VAArgOverflowSize = nullptr;
9252
9253 enum class ArgKind {
9254 GeneralPurpose,
9255 FloatingPoint,
9256 Vector,
9257 Memory,
9258 Indirect,
9259 };
9260
9261 enum class ShadowExtension { None, Zero, Sign };
9262
9263 VarArgSystemZHelper(Function &F, MemorySanitizer &MS,
9264 MemorySanitizerVisitor &MSV)
9265 : VarArgHelperBase(F, MS, MSV, SystemZVAListTagSize),
9266 IsSoftFloatABI(F.getFnAttribute(Kind: "use-soft-float").getValueAsBool()) {}
9267
9268 ArgKind classifyArgument(Type *T) {
9269 // T is a SystemZABIInfo::classifyArgumentType() output, and there are
9270 // only a few possibilities of what it can be. In particular, enums, single
9271 // element structs and large types have already been taken care of.
9272
9273 // Some i128 and fp128 arguments are converted to pointers only in the
9274 // back end.
9275 if (T->isIntegerTy(BitWidth: 128) || T->isFP128Ty())
9276 return ArgKind::Indirect;
9277 if (T->isFloatingPointTy())
9278 return IsSoftFloatABI ? ArgKind::GeneralPurpose : ArgKind::FloatingPoint;
9279 if (T->isIntegerTy() || T->isPointerTy())
9280 return ArgKind::GeneralPurpose;
9281 if (T->isVectorTy())
9282 return ArgKind::Vector;
9283 return ArgKind::Memory;
9284 }
9285
9286 ShadowExtension getShadowExtension(const CallBase &CB, unsigned ArgNo) {
9287 // ABI says: "One of the simple integer types no more than 64 bits wide.
9288 // ... If such an argument is shorter than 64 bits, replace it by a full
9289 // 64-bit integer representing the same number, using sign or zero
9290 // extension". Shadow for an integer argument has the same type as the
9291 // argument itself, so it can be sign or zero extended as well.
9292 bool ZExt = CB.paramHasAttr(ArgNo, Kind: Attribute::ZExt);
9293 bool SExt = CB.paramHasAttr(ArgNo, Kind: Attribute::SExt);
9294 if (ZExt) {
9295 assert(!SExt);
9296 return ShadowExtension::Zero;
9297 }
9298 if (SExt) {
9299 assert(!ZExt);
9300 return ShadowExtension::Sign;
9301 }
9302 return ShadowExtension::None;
9303 }
9304
9305 void visitCallBase(CallBase &CB, IRBuilder<> &IRB) override {
9306 unsigned GpOffset = SystemZGpOffset;
9307 unsigned FpOffset = SystemZFpOffset;
9308 unsigned VrIndex = 0;
9309 unsigned OverflowOffset = SystemZOverflowOffset;
9310 const DataLayout &DL = F.getDataLayout();
9311 for (const auto &[ArgNo, A] : llvm::enumerate(First: CB.args())) {
9312 bool IsFixed = ArgNo < CB.getFunctionType()->getNumParams();
9313 // SystemZABIInfo does not produce ByVal parameters.
9314 assert(!CB.isByValArgument(ArgNo));
9315 Type *T = A->getType();
9316 ArgKind AK = classifyArgument(T);
9317 if (AK == ArgKind::Indirect) {
9318 T = MS.PtrTy;
9319 AK = ArgKind::GeneralPurpose;
9320 }
9321 if (AK == ArgKind::GeneralPurpose && GpOffset >= SystemZGpEndOffset)
9322 AK = ArgKind::Memory;
9323 if (AK == ArgKind::FloatingPoint && FpOffset >= SystemZFpEndOffset)
9324 AK = ArgKind::Memory;
9325 if (AK == ArgKind::Vector && (VrIndex >= SystemZMaxVrArgs || !IsFixed))
9326 AK = ArgKind::Memory;
9327 Value *ShadowBase = nullptr;
9328 Value *OriginBase = nullptr;
9329 ShadowExtension SE = ShadowExtension::None;
9330 switch (AK) {
9331 case ArgKind::GeneralPurpose: {
9332 // Always keep track of GpOffset, but store shadow only for varargs.
9333 uint64_t ArgSize = 8;
9334 if (GpOffset + ArgSize <= kParamTLSSize) {
9335 if (!IsFixed) {
9336 SE = getShadowExtension(CB, ArgNo);
9337 uint64_t GapSize = 0;
9338 if (SE == ShadowExtension::None) {
9339 uint64_t ArgAllocSize = DL.getTypeAllocSize(Ty: T);
9340 assert(ArgAllocSize <= ArgSize);
9341 GapSize = ArgSize - ArgAllocSize;
9342 }
9343 ShadowBase = getShadowAddrForVAArgument(IRB, ArgOffset: GpOffset + GapSize);
9344 if (MS.TrackOrigins)
9345 OriginBase = getOriginPtrForVAArgument(IRB, ArgOffset: GpOffset + GapSize);
9346 }
9347 GpOffset += ArgSize;
9348 } else {
9349 GpOffset = kParamTLSSize;
9350 }
9351 break;
9352 }
9353 case ArgKind::FloatingPoint: {
9354 // Always keep track of FpOffset, but store shadow only for varargs.
9355 uint64_t ArgSize = 8;
9356 if (FpOffset + ArgSize <= kParamTLSSize) {
9357 if (!IsFixed) {
9358 // PoP says: "A short floating-point datum requires only the
9359 // left-most 32 bit positions of a floating-point register".
9360 // Therefore, in contrast to AK_GeneralPurpose and AK_Memory,
9361 // don't extend shadow and don't mind the gap.
9362 ShadowBase = getShadowAddrForVAArgument(IRB, ArgOffset: FpOffset);
9363 if (MS.TrackOrigins)
9364 OriginBase = getOriginPtrForVAArgument(IRB, ArgOffset: FpOffset);
9365 }
9366 FpOffset += ArgSize;
9367 } else {
9368 FpOffset = kParamTLSSize;
9369 }
9370 break;
9371 }
9372 case ArgKind::Vector: {
9373 // Keep track of VrIndex. No need to store shadow, since vector varargs
9374 // go through AK_Memory.
9375 assert(IsFixed);
9376 VrIndex++;
9377 break;
9378 }
9379 case ArgKind::Memory: {
9380 // Keep track of OverflowOffset and store shadow only for varargs.
9381 // Ignore fixed args, since we need to copy only the vararg portion of
9382 // the overflow area shadow.
9383 if (!IsFixed) {
9384 uint64_t ArgAllocSize = DL.getTypeAllocSize(Ty: T);
9385 uint64_t ArgSize = alignTo(Value: ArgAllocSize, Align: 8);
9386 if (OverflowOffset + ArgSize <= kParamTLSSize) {
9387 SE = getShadowExtension(CB, ArgNo);
9388 uint64_t GapSize =
9389 SE == ShadowExtension::None ? ArgSize - ArgAllocSize : 0;
9390 ShadowBase =
9391 getShadowAddrForVAArgument(IRB, ArgOffset: OverflowOffset + GapSize);
9392 if (MS.TrackOrigins)
9393 OriginBase =
9394 getOriginPtrForVAArgument(IRB, ArgOffset: OverflowOffset + GapSize);
9395 OverflowOffset += ArgSize;
9396 } else {
9397 OverflowOffset = kParamTLSSize;
9398 }
9399 }
9400 break;
9401 }
9402 case ArgKind::Indirect:
9403 llvm_unreachable("Indirect must be converted to GeneralPurpose");
9404 }
9405 if (ShadowBase == nullptr)
9406 continue;
9407 Value *Shadow = MSV.getShadow(V: A);
9408 if (SE != ShadowExtension::None)
9409 Shadow = MSV.CreateShadowCast(IRB, V: Shadow, dstTy: IRB.getInt64Ty(),
9410 /*Signed*/ SE == ShadowExtension::Sign);
9411 ShadowBase = IRB.CreateIntToPtr(V: ShadowBase, DestTy: MS.PtrTy, Name: "_msarg_va_s");
9412 IRB.CreateStore(Val: Shadow, Ptr: ShadowBase);
9413 if (MS.TrackOrigins) {
9414 Value *Origin = MSV.getOrigin(V: A);
9415 TypeSize StoreSize = DL.getTypeStoreSize(Ty: Shadow->getType());
9416 MSV.paintOrigin(IRB, Origin, OriginPtr: OriginBase, TS: StoreSize,
9417 Alignment: kMinOriginAlignment);
9418 }
9419 }
9420 Constant *OverflowSize = ConstantInt::get(
9421 Ty: IRB.getInt64Ty(), V: OverflowOffset - SystemZOverflowOffset);
9422 IRB.CreateStore(Val: OverflowSize, Ptr: MS.VAArgOverflowSizeTLS);
9423 }
9424
9425 void copyRegSaveArea(IRBuilder<> &IRB, Value *VAListTag) {
9426 Value *RegSaveAreaPtrPtr = IRB.CreateIntToPtr(
9427 V: IRB.CreateAdd(
9428 LHS: IRB.CreatePtrToInt(V: VAListTag, DestTy: MS.IntptrTy),
9429 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: SystemZRegSaveAreaPtrOffset)),
9430 DestTy: MS.PtrTy);
9431 Value *RegSaveAreaPtr = IRB.CreateLoad(Ty: MS.PtrTy, Ptr: RegSaveAreaPtrPtr);
9432 Value *RegSaveAreaShadowPtr, *RegSaveAreaOriginPtr;
9433 const Align Alignment = Align(8);
9434 std::tie(args&: RegSaveAreaShadowPtr, args&: RegSaveAreaOriginPtr) =
9435 MSV.getShadowOriginPtr(Addr: RegSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(), Alignment,
9436 /*isStore*/ true);
9437 // TODO(iii): copy only fragments filled by visitCallBase()
9438 // TODO(iii): support packed-stack && !use-soft-float
9439 // For use-soft-float functions, it is enough to copy just the GPRs.
9440 unsigned RegSaveAreaSize =
9441 IsSoftFloatABI ? SystemZGpEndOffset : SystemZRegSaveAreaSize;
9442 IRB.CreateMemCpy(Dst: RegSaveAreaShadowPtr, DstAlign: Alignment, Src: VAArgTLSCopy, SrcAlign: Alignment,
9443 Size: RegSaveAreaSize);
9444 if (MS.TrackOrigins)
9445 IRB.CreateMemCpy(Dst: RegSaveAreaOriginPtr, DstAlign: Alignment, Src: VAArgTLSOriginCopy,
9446 SrcAlign: Alignment, Size: RegSaveAreaSize);
9447 }
9448
9449 // FIXME: This implementation limits OverflowOffset to kParamTLSSize, so we
9450 // don't know real overflow size and can't clear shadow beyond kParamTLSSize.
9451 void copyOverflowArea(IRBuilder<> &IRB, Value *VAListTag) {
9452 Value *OverflowArgAreaPtrPtr = IRB.CreateIntToPtr(
9453 V: IRB.CreateAdd(
9454 LHS: IRB.CreatePtrToInt(V: VAListTag, DestTy: MS.IntptrTy),
9455 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: SystemZOverflowArgAreaPtrOffset)),
9456 DestTy: MS.PtrTy);
9457 Value *OverflowArgAreaPtr = IRB.CreateLoad(Ty: MS.PtrTy, Ptr: OverflowArgAreaPtrPtr);
9458 Value *OverflowArgAreaShadowPtr, *OverflowArgAreaOriginPtr;
9459 const Align Alignment = Align(8);
9460 std::tie(args&: OverflowArgAreaShadowPtr, args&: OverflowArgAreaOriginPtr) =
9461 MSV.getShadowOriginPtr(Addr: OverflowArgAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
9462 Alignment, /*isStore*/ true);
9463 Value *SrcPtr = IRB.CreateConstGEP1_32(Ty: IRB.getInt8Ty(), Ptr: VAArgTLSCopy,
9464 Idx0: SystemZOverflowOffset);
9465 IRB.CreateMemCpy(Dst: OverflowArgAreaShadowPtr, DstAlign: Alignment, Src: SrcPtr, SrcAlign: Alignment,
9466 Size: VAArgOverflowSize);
9467 if (MS.TrackOrigins) {
9468 SrcPtr = IRB.CreateConstGEP1_32(Ty: IRB.getInt8Ty(), Ptr: VAArgTLSOriginCopy,
9469 Idx0: SystemZOverflowOffset);
9470 IRB.CreateMemCpy(Dst: OverflowArgAreaOriginPtr, DstAlign: Alignment, Src: SrcPtr, SrcAlign: Alignment,
9471 Size: VAArgOverflowSize);
9472 }
9473 }
9474
9475 void finalizeInstrumentation() override {
9476 assert(!VAArgOverflowSize && !VAArgTLSCopy &&
9477 "finalizeInstrumentation called twice");
9478 if (!VAStartInstrumentationList.empty()) {
9479 // If there is a va_start in this function, make a backup copy of
9480 // va_arg_tls somewhere in the function entry block.
9481 IRBuilder<> IRB(MSV.FnPrologueEnd);
9482 VAArgOverflowSize =
9483 IRB.CreateLoad(Ty: IRB.getInt64Ty(), Ptr: MS.VAArgOverflowSizeTLS);
9484 Value *CopySize =
9485 IRB.CreateAdd(LHS: ConstantInt::get(Ty: MS.IntptrTy, V: SystemZOverflowOffset),
9486 RHS: VAArgOverflowSize);
9487 VAArgTLSCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
9488 VAArgTLSCopy->setAlignment(kShadowTLSAlignment);
9489 IRB.CreateMemSet(Ptr: VAArgTLSCopy, Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
9490 Size: CopySize, Align: kShadowTLSAlignment, isVolatile: false);
9491
9492 Value *SrcSize = IRB.CreateBinaryIntrinsic(
9493 ID: Intrinsic::umin, LHS: CopySize,
9494 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: kParamTLSSize));
9495 IRB.CreateMemCpy(Dst: VAArgTLSCopy, DstAlign: kShadowTLSAlignment, Src: MS.VAArgTLS,
9496 SrcAlign: kShadowTLSAlignment, Size: SrcSize);
9497 if (MS.TrackOrigins) {
9498 VAArgTLSOriginCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
9499 VAArgTLSOriginCopy->setAlignment(kShadowTLSAlignment);
9500 IRB.CreateMemCpy(Dst: VAArgTLSOriginCopy, DstAlign: kShadowTLSAlignment,
9501 Src: MS.VAArgOriginTLS, SrcAlign: kShadowTLSAlignment, Size: SrcSize);
9502 }
9503 }
9504
9505 // Instrument va_start.
9506 // Copy va_list shadow from the backup copy of the TLS contents.
9507 for (CallInst *OrigInst : VAStartInstrumentationList) {
9508 NextNodeIRBuilder IRB(OrigInst);
9509 Value *VAListTag = OrigInst->getArgOperand(i: 0);
9510 copyRegSaveArea(IRB, VAListTag);
9511 copyOverflowArea(IRB, VAListTag);
9512 }
9513 }
9514};
9515
9516/// i386-specific implementation of VarArgHelper.
9517struct VarArgI386Helper : public VarArgHelperBase {
9518 AllocaInst *VAArgTLSCopy = nullptr;
9519 Value *VAArgSize = nullptr;
9520
9521 VarArgI386Helper(Function &F, MemorySanitizer &MS,
9522 MemorySanitizerVisitor &MSV)
9523 : VarArgHelperBase(F, MS, MSV, /*VAListTagSize=*/4) {}
9524
9525 void visitCallBase(CallBase &CB, IRBuilder<> &IRB) override {
9526 const DataLayout &DL = F.getDataLayout();
9527 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
9528 unsigned VAArgOffset = 0;
9529 for (const auto &[ArgNo, A] : llvm::enumerate(First: CB.args())) {
9530 bool IsFixed = ArgNo < CB.getFunctionType()->getNumParams();
9531 bool IsByVal = CB.isByValArgument(ArgNo);
9532 if (IsByVal) {
9533 assert(A->getType()->isPointerTy());
9534 Type *RealTy = CB.getParamByValType(ArgNo);
9535 uint64_t ArgSize = DL.getTypeAllocSize(Ty: RealTy);
9536 Align ArgAlign = CB.getParamAlign(ArgNo).value_or(u: Align(IntptrSize));
9537 if (ArgAlign < IntptrSize)
9538 ArgAlign = Align(IntptrSize);
9539 VAArgOffset = alignTo(Size: VAArgOffset, A: ArgAlign);
9540 if (!IsFixed) {
9541 Value *Base = getShadowPtrForVAArgument(IRB, ArgOffset: VAArgOffset, ArgSize);
9542 if (Base) {
9543 Value *AShadowPtr, *AOriginPtr;
9544 std::tie(args&: AShadowPtr, args&: AOriginPtr) =
9545 MSV.getShadowOriginPtr(Addr: A, IRB, ShadowTy: IRB.getInt8Ty(),
9546 Alignment: kShadowTLSAlignment, /*isStore*/ false);
9547
9548 IRB.CreateMemCpy(Dst: Base, DstAlign: kShadowTLSAlignment, Src: AShadowPtr,
9549 SrcAlign: kShadowTLSAlignment, Size: ArgSize);
9550 }
9551 VAArgOffset += alignTo(Size: ArgSize, A: Align(IntptrSize));
9552 }
9553 } else {
9554 Value *Base;
9555 uint64_t ArgSize = DL.getTypeAllocSize(Ty: A->getType());
9556 Align ArgAlign = Align(IntptrSize);
9557 VAArgOffset = alignTo(Size: VAArgOffset, A: ArgAlign);
9558 if (DL.isBigEndian()) {
9559 // Adjusting the shadow for argument with size < IntptrSize to match
9560 // the placement of bits in big endian system
9561 if (ArgSize < IntptrSize)
9562 VAArgOffset += (IntptrSize - ArgSize);
9563 }
9564 if (!IsFixed) {
9565 Base = getShadowPtrForVAArgument(IRB, ArgOffset: VAArgOffset, ArgSize);
9566 if (Base)
9567 IRB.CreateAlignedStore(Val: MSV.getShadow(V: A), Ptr: Base, Align: kShadowTLSAlignment);
9568 VAArgOffset += ArgSize;
9569 VAArgOffset = alignTo(Size: VAArgOffset, A: Align(IntptrSize));
9570 }
9571 }
9572 }
9573
9574 Constant *TotalVAArgSize = ConstantInt::get(Ty: MS.IntptrTy, V: VAArgOffset);
9575 // Here using VAArgOverflowSizeTLS as VAArgSizeTLS to avoid creation of
9576 // a new class member i.e. it is the total size of all VarArgs.
9577 IRB.CreateStore(Val: TotalVAArgSize, Ptr: MS.VAArgOverflowSizeTLS);
9578 }
9579
9580 void finalizeInstrumentation() override {
9581 assert(!VAArgSize && !VAArgTLSCopy &&
9582 "finalizeInstrumentation called twice");
9583 IRBuilder<> IRB(MSV.FnPrologueEnd);
9584 VAArgSize = IRB.CreateLoad(Ty: MS.IntptrTy, Ptr: MS.VAArgOverflowSizeTLS);
9585 Value *CopySize = VAArgSize;
9586
9587 if (!VAStartInstrumentationList.empty()) {
9588 // If there is a va_start in this function, make a backup copy of
9589 // va_arg_tls somewhere in the function entry block.
9590 VAArgTLSCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
9591 VAArgTLSCopy->setAlignment(kShadowTLSAlignment);
9592 IRB.CreateMemSet(Ptr: VAArgTLSCopy, Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
9593 Size: CopySize, Align: kShadowTLSAlignment, isVolatile: false);
9594
9595 Value *SrcSize = IRB.CreateBinaryIntrinsic(
9596 ID: Intrinsic::umin, LHS: CopySize,
9597 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: kParamTLSSize));
9598 IRB.CreateMemCpy(Dst: VAArgTLSCopy, DstAlign: kShadowTLSAlignment, Src: MS.VAArgTLS,
9599 SrcAlign: kShadowTLSAlignment, Size: SrcSize);
9600 }
9601
9602 // Instrument va_start.
9603 // Copy va_list shadow from the backup copy of the TLS contents.
9604 for (CallInst *OrigInst : VAStartInstrumentationList) {
9605 NextNodeIRBuilder IRB(OrigInst);
9606 Value *VAListTag = OrigInst->getArgOperand(i: 0);
9607 Type *RegSaveAreaPtrTy = PointerType::getUnqual(C&: *MS.C);
9608 Value *RegSaveAreaPtrPtr =
9609 IRB.CreateIntToPtr(V: IRB.CreatePtrToInt(V: VAListTag, DestTy: MS.IntptrTy),
9610 DestTy: PointerType::get(C&: *MS.C, AddressSpace: 0));
9611 Value *RegSaveAreaPtr =
9612 IRB.CreateLoad(Ty: RegSaveAreaPtrTy, Ptr: RegSaveAreaPtrPtr);
9613 Value *RegSaveAreaShadowPtr, *RegSaveAreaOriginPtr;
9614 const DataLayout &DL = F.getDataLayout();
9615 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
9616 const Align Alignment = Align(IntptrSize);
9617 std::tie(args&: RegSaveAreaShadowPtr, args&: RegSaveAreaOriginPtr) =
9618 MSV.getShadowOriginPtr(Addr: RegSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
9619 Alignment, /*isStore*/ true);
9620 IRB.CreateMemCpy(Dst: RegSaveAreaShadowPtr, DstAlign: Alignment, Src: VAArgTLSCopy, SrcAlign: Alignment,
9621 Size: CopySize);
9622 }
9623 }
9624};
9625
9626/// Implementation of VarArgHelper that is used for ARM32, MIPS, RISCV,
9627/// LoongArch64.
9628struct VarArgGenericHelper : public VarArgHelperBase {
9629 AllocaInst *VAArgTLSCopy = nullptr;
9630 Value *VAArgSize = nullptr;
9631
9632 VarArgGenericHelper(Function &F, MemorySanitizer &MS,
9633 MemorySanitizerVisitor &MSV, const unsigned VAListTagSize)
9634 : VarArgHelperBase(F, MS, MSV, VAListTagSize) {}
9635
9636 void visitCallBase(CallBase &CB, IRBuilder<> &IRB) override {
9637 unsigned VAArgOffset = 0;
9638 const DataLayout &DL = F.getDataLayout();
9639 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
9640 for (const auto &[ArgNo, A] : llvm::enumerate(First: CB.args())) {
9641 bool IsFixed = ArgNo < CB.getFunctionType()->getNumParams();
9642 if (IsFixed)
9643 continue;
9644 uint64_t ArgSize = DL.getTypeAllocSize(Ty: A->getType());
9645 if (DL.isBigEndian()) {
9646 // Adjusting the shadow for argument with size < IntptrSize to match the
9647 // placement of bits in big endian system
9648 if (ArgSize < IntptrSize)
9649 VAArgOffset += (IntptrSize - ArgSize);
9650 }
9651 Value *Base = getShadowPtrForVAArgument(IRB, ArgOffset: VAArgOffset, ArgSize);
9652 VAArgOffset += ArgSize;
9653 VAArgOffset = alignTo(Value: VAArgOffset, Align: IntptrSize);
9654 if (!Base)
9655 continue;
9656 IRB.CreateAlignedStore(Val: MSV.getShadow(V: A), Ptr: Base, Align: kShadowTLSAlignment);
9657 }
9658
9659 Constant *TotalVAArgSize = ConstantInt::get(Ty: MS.IntptrTy, V: VAArgOffset);
9660 // Here using VAArgOverflowSizeTLS as VAArgSizeTLS to avoid creation of
9661 // a new class member i.e. it is the total size of all VarArgs.
9662 IRB.CreateStore(Val: TotalVAArgSize, Ptr: MS.VAArgOverflowSizeTLS);
9663 }
9664
9665 void finalizeInstrumentation() override {
9666 assert(!VAArgSize && !VAArgTLSCopy &&
9667 "finalizeInstrumentation called twice");
9668 IRBuilder<> IRB(MSV.FnPrologueEnd);
9669 VAArgSize = IRB.CreateLoad(Ty: MS.IntptrTy, Ptr: MS.VAArgOverflowSizeTLS);
9670 Value *CopySize = VAArgSize;
9671
9672 if (!VAStartInstrumentationList.empty()) {
9673 // If there is a va_start in this function, make a backup copy of
9674 // va_arg_tls somewhere in the function entry block.
9675 VAArgTLSCopy = IRB.CreateAlloca(Ty: Type::getInt8Ty(C&: *MS.C), ArraySize: CopySize);
9676 VAArgTLSCopy->setAlignment(kShadowTLSAlignment);
9677 IRB.CreateMemSet(Ptr: VAArgTLSCopy, Val: Constant::getNullValue(Ty: IRB.getInt8Ty()),
9678 Size: CopySize, Align: kShadowTLSAlignment, isVolatile: false);
9679
9680 Value *SrcSize = IRB.CreateBinaryIntrinsic(
9681 ID: Intrinsic::umin, LHS: CopySize,
9682 RHS: ConstantInt::get(Ty: MS.IntptrTy, V: kParamTLSSize));
9683 IRB.CreateMemCpy(Dst: VAArgTLSCopy, DstAlign: kShadowTLSAlignment, Src: MS.VAArgTLS,
9684 SrcAlign: kShadowTLSAlignment, Size: SrcSize);
9685 }
9686
9687 // Instrument va_start.
9688 // Copy va_list shadow from the backup copy of the TLS contents.
9689 for (CallInst *OrigInst : VAStartInstrumentationList) {
9690 NextNodeIRBuilder IRB(OrigInst);
9691 Value *VAListTag = OrigInst->getArgOperand(i: 0);
9692 Type *RegSaveAreaPtrTy = PointerType::getUnqual(C&: *MS.C);
9693 Value *RegSaveAreaPtrPtr =
9694 IRB.CreateIntToPtr(V: IRB.CreatePtrToInt(V: VAListTag, DestTy: MS.IntptrTy),
9695 DestTy: PointerType::get(C&: *MS.C, AddressSpace: 0));
9696 Value *RegSaveAreaPtr =
9697 IRB.CreateLoad(Ty: RegSaveAreaPtrTy, Ptr: RegSaveAreaPtrPtr);
9698 Value *RegSaveAreaShadowPtr, *RegSaveAreaOriginPtr;
9699 const DataLayout &DL = F.getDataLayout();
9700 unsigned IntptrSize = DL.getTypeStoreSize(Ty: MS.IntptrTy);
9701 const Align Alignment = Align(IntptrSize);
9702 std::tie(args&: RegSaveAreaShadowPtr, args&: RegSaveAreaOriginPtr) =
9703 MSV.getShadowOriginPtr(Addr: RegSaveAreaPtr, IRB, ShadowTy: IRB.getInt8Ty(),
9704 Alignment, /*isStore*/ true);
9705 IRB.CreateMemCpy(Dst: RegSaveAreaShadowPtr, DstAlign: Alignment, Src: VAArgTLSCopy, SrcAlign: Alignment,
9706 Size: CopySize);
9707 }
9708 }
9709};
9710
9711// ARM32, Loongarch64, MIPS and RISCV share the same calling conventions
9712// regarding VAArgs.
9713using VarArgARM32Helper = VarArgGenericHelper;
9714using VarArgRISCVHelper = VarArgGenericHelper;
9715using VarArgMIPSHelper = VarArgGenericHelper;
9716using VarArgLoongArch64Helper = VarArgGenericHelper;
9717using VarArgHexagonHelper = VarArgGenericHelper;
9718
9719/// A no-op implementation of VarArgHelper.
9720struct VarArgNoOpHelper : public VarArgHelper {
9721 VarArgNoOpHelper(Function &F, MemorySanitizer &MS,
9722 MemorySanitizerVisitor &MSV) {}
9723
9724 void visitCallBase(CallBase &CB, IRBuilder<> &IRB) override {}
9725
9726 void visitVAStartInst(VAStartInst &I) override {}
9727
9728 void visitVACopyInst(VACopyInst &I) override {}
9729
9730 void finalizeInstrumentation() override {}
9731};
9732
9733} // end anonymous namespace
9734
9735static VarArgHelper *CreateVarArgHelper(Function &Func, MemorySanitizer &Msan,
9736 MemorySanitizerVisitor &Visitor) {
9737 // VarArg handling is only implemented on AMD64. False positives are possible
9738 // on other platforms.
9739 Triple TargetTriple(Func.getParent()->getTargetTriple());
9740
9741 if (TargetTriple.getArch() == Triple::x86)
9742 return new VarArgI386Helper(Func, Msan, Visitor);
9743
9744 if (TargetTriple.getArch() == Triple::x86_64)
9745 return new VarArgAMD64Helper(Func, Msan, Visitor);
9746
9747 if (TargetTriple.isARM())
9748 return new VarArgARM32Helper(Func, Msan, Visitor, /*VAListTagSize=*/4);
9749
9750 if (TargetTriple.isAArch64())
9751 return new VarArgAArch64Helper(Func, Msan, Visitor);
9752
9753 if (TargetTriple.isSystemZ())
9754 return new VarArgSystemZHelper(Func, Msan, Visitor);
9755
9756 // On PowerPC32 VAListTag is a struct
9757 // {char, char, i16 padding, char *, char *}
9758 if (TargetTriple.isPPC32())
9759 return new VarArgPowerPC32Helper(Func, Msan, Visitor);
9760
9761 if (TargetTriple.isPPC64())
9762 return new VarArgPowerPC64Helper(Func, Msan, Visitor);
9763
9764 if (TargetTriple.isRISCV32())
9765 return new VarArgRISCVHelper(Func, Msan, Visitor, /*VAListTagSize=*/4);
9766
9767 if (TargetTriple.isRISCV64())
9768 return new VarArgRISCVHelper(Func, Msan, Visitor, /*VAListTagSize=*/8);
9769
9770 if (TargetTriple.isMIPS32())
9771 return new VarArgMIPSHelper(Func, Msan, Visitor, /*VAListTagSize=*/4);
9772
9773 if (TargetTriple.isMIPS64())
9774 return new VarArgMIPSHelper(Func, Msan, Visitor, /*VAListTagSize=*/8);
9775
9776 if (TargetTriple.isLoongArch64())
9777 return new VarArgLoongArch64Helper(Func, Msan, Visitor,
9778 /*VAListTagSize=*/8);
9779
9780 if (TargetTriple.getArch() == Triple::hexagon)
9781 return new VarArgHexagonHelper(Func, Msan, Visitor, /*VAListTagSize=*/12);
9782
9783 return new VarArgNoOpHelper(Func, Msan, Visitor);
9784}
9785
9786bool MemorySanitizer::sanitizeFunction(Function &F, TargetLibraryInfo &TLI) {
9787 if (!CompileKernel && F.getName() == kMsanModuleCtorName)
9788 return false;
9789
9790 if (F.hasFnAttribute(Kind: Attribute::DisableSanitizerInstrumentation))
9791 return false;
9792
9793 MemorySanitizerVisitor Visitor(F, *this, TLI);
9794
9795 // Clear out memory attributes.
9796 AttributeMask B;
9797 B.addAttribute(Val: Attribute::Memory).addAttribute(Val: Attribute::Speculatable);
9798 F.removeFnAttrs(Attrs: B);
9799
9800 return Visitor.runOnFunction();
9801}
9802