1//===-- AutoUpgrade.cpp - Implement auto-upgrade helper functions ---------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the auto-upgrade helper functions.
10// This is where deprecated IR intrinsics and other IR features are updated to
11// current specifications.
12//
13//===----------------------------------------------------------------------===//
14
15#include "llvm/IR/AutoUpgrade.h"
16#include "llvm/ADT/ArrayRef.h"
17#include "llvm/ADT/StringExtras.h"
18#include "llvm/ADT/StringRef.h"
19#include "llvm/ADT/StringSwitch.h"
20#include "llvm/BinaryFormat/Dwarf.h"
21#include "llvm/IR/AttributeMask.h"
22#include "llvm/IR/Attributes.h"
23#include "llvm/IR/CallingConv.h"
24#include "llvm/IR/Constants.h"
25#include "llvm/IR/DebugInfo.h"
26#include "llvm/IR/DebugInfoMetadata.h"
27#include "llvm/IR/DiagnosticInfo.h"
28#include "llvm/IR/Function.h"
29#include "llvm/IR/GlobalValue.h"
30#include "llvm/IR/IRBuilder.h"
31#include "llvm/IR/InstVisitor.h"
32#include "llvm/IR/Instruction.h"
33#include "llvm/IR/IntrinsicInst.h"
34#include "llvm/IR/Intrinsics.h"
35#include "llvm/IR/IntrinsicsAArch64.h"
36#include "llvm/IR/IntrinsicsAMDGPU.h"
37#include "llvm/IR/IntrinsicsARM.h"
38#include "llvm/IR/IntrinsicsNVPTX.h"
39#include "llvm/IR/IntrinsicsRISCV.h"
40#include "llvm/IR/IntrinsicsWebAssembly.h"
41#include "llvm/IR/IntrinsicsX86.h"
42#include "llvm/IR/LLVMContext.h"
43#include "llvm/IR/MDBuilder.h"
44#include "llvm/IR/Metadata.h"
45#include "llvm/IR/Module.h"
46#include "llvm/IR/NVVMIntrinsicUtils.h"
47#include "llvm/IR/Value.h"
48#include "llvm/IR/Verifier.h"
49#include "llvm/Support/AMDGPUAddrSpace.h"
50#include "llvm/Support/CodeGen.h"
51#include "llvm/Support/CommandLine.h"
52#include "llvm/Support/Debug.h"
53#include "llvm/Support/ErrorHandling.h"
54#include "llvm/Support/NVPTXAddrSpace.h"
55#include "llvm/Support/NVVMAttributes.h"
56#include "llvm/Support/Regex.h"
57#include "llvm/Support/TimeProfiler.h"
58#include "llvm/TargetParser/Triple.h"
59#include <cstdint>
60#include <cstring>
61#include <numeric>
62
63using namespace llvm;
64
65#define DEBUG_TYPE "auto-upgrade"
66
67static cl::opt<bool>
68 DisableAutoUpgradeDebugInfo("disable-auto-upgrade-debug-info",
69 cl::desc("Disable autoupgrade of debug info"));
70
71static void rename(GlobalValue *GV) { GV->setName(GV->getName() + ".old"); }
72
73// Report a fatal error along with the
74// Call Instruction which caused the error
75[[noreturn]] static void reportFatalUsageErrorWithCI(StringRef reason,
76 CallBase *CI) {
77 CI->print(O&: llvm::errs());
78 llvm::errs() << "\n";
79 reportFatalUsageError(reason);
80}
81
82// Upgrade the declarations of the SSE4.1 ptest intrinsics whose arguments have
83// changed their type from v4f32 to v2i64.
84static bool upgradePTESTIntrinsic(Function *F, Intrinsic::ID IID,
85 Function *&NewFn) {
86 // Check whether this is an old version of the function, which received
87 // v4f32 arguments.
88 Type *Arg0Type = F->getFunctionType()->getParamType(i: 0);
89 if (Arg0Type != FixedVectorType::get(ElementType: Type::getFloatTy(C&: F->getContext()), NumElts: 4))
90 return false;
91
92 // Yes, it's old, replace it with new version.
93 rename(GV: F);
94 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
95 return true;
96}
97
98// Upgrade the declarations of intrinsic functions whose 8-bit immediate mask
99// arguments have changed their type from i32 to i8.
100static bool upgradeX86IntrinsicsWith8BitMask(Function *F, Intrinsic::ID IID,
101 Function *&NewFn) {
102 // Check that the last argument is an i32.
103 Type *LastArgType = F->getFunctionType()->getParamType(
104 i: F->getFunctionType()->getNumParams() - 1);
105 if (!LastArgType->isIntegerTy(BitWidth: 32))
106 return false;
107
108 // Move this function aside and map down.
109 rename(GV: F);
110 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
111 return true;
112}
113
114// Upgrade the declaration of fp compare intrinsics that change return type
115// from scalar to vXi1 mask.
116static bool upgradeX86MaskedFPCompare(Function *F, Intrinsic::ID IID,
117 Function *&NewFn) {
118 // Check if the return type is a vector.
119 if (F->getReturnType()->isVectorTy())
120 return false;
121
122 rename(GV: F);
123 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
124 return true;
125}
126
127// Upgrade the declaration of multiply and add bytes intrinsics whose input
128// arguments' types have changed from vectors of i32 to vectors of i8
129static bool upgradeX86MultiplyAddBytes(Function *F, Intrinsic::ID IID,
130 Function *&NewFn) {
131 // check if input argument type is a vector of i8
132 Type *Arg1Type = F->getFunctionType()->getParamType(i: 1);
133 Type *Arg2Type = F->getFunctionType()->getParamType(i: 2);
134 if (Arg1Type->isVectorTy() &&
135 cast<VectorType>(Val: Arg1Type)->getElementType()->isIntegerTy(BitWidth: 8) &&
136 Arg2Type->isVectorTy() &&
137 cast<VectorType>(Val: Arg2Type)->getElementType()->isIntegerTy(BitWidth: 8))
138 return false;
139
140 rename(GV: F);
141 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
142 return true;
143}
144
145// Upgrade the declaration of multipy and add words intrinsics whose input
146// arguments' types have changed to vectors of i32 to vectors of i16
147static bool upgradeX86MultiplyAddWords(Function *F, Intrinsic::ID IID,
148 Function *&NewFn) {
149 // check if input argument type is a vector of i16
150 Type *Arg1Type = F->getFunctionType()->getParamType(i: 1);
151 Type *Arg2Type = F->getFunctionType()->getParamType(i: 2);
152 if (Arg1Type->isVectorTy() &&
153 cast<VectorType>(Val: Arg1Type)->getElementType()->isIntegerTy(BitWidth: 16) &&
154 Arg2Type->isVectorTy() &&
155 cast<VectorType>(Val: Arg2Type)->getElementType()->isIntegerTy(BitWidth: 16))
156 return false;
157
158 rename(GV: F);
159 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
160 return true;
161}
162
163static bool upgradeX86BF16Intrinsic(Function *F, Intrinsic::ID IID,
164 Function *&NewFn) {
165 if (F->getReturnType()->getScalarType()->isBFloatTy())
166 return false;
167
168 rename(GV: F);
169 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
170 return true;
171}
172
173static bool upgradeX86BF16DPIntrinsic(Function *F, Intrinsic::ID IID,
174 Function *&NewFn) {
175 if (F->getFunctionType()->getParamType(i: 1)->getScalarType()->isBFloatTy())
176 return false;
177
178 rename(GV: F);
179 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
180 return true;
181}
182
183static bool shouldUpgradeX86Intrinsic(Function *F, StringRef Name) {
184 // All of the intrinsics matches below should be marked with which llvm
185 // version started autoupgrading them. At some point in the future we would
186 // like to use this information to remove upgrade code for some older
187 // intrinsics. It is currently undecided how we will determine that future
188 // point.
189 if (Name.consume_front(Prefix: "avx."))
190 return (Name.starts_with(Prefix: "blend.p") || // Added in 3.7
191 Name == "cvt.ps2.pd.256" || // Added in 3.9
192 Name == "cvtdq2.pd.256" || // Added in 3.9
193 Name == "cvtdq2.ps.256" || // Added in 7.0
194 Name.starts_with(Prefix: "movnt.") || // Added in 3.2
195 Name.starts_with(Prefix: "sqrt.p") || // Added in 7.0
196 Name.starts_with(Prefix: "storeu.") || // Added in 3.9
197 Name.starts_with(Prefix: "vbroadcast.s") || // Added in 3.5
198 Name.starts_with(Prefix: "vbroadcastf128") || // Added in 4.0
199 Name.starts_with(Prefix: "vextractf128.") || // Added in 3.7
200 Name.starts_with(Prefix: "vinsertf128.") || // Added in 3.7
201 Name.starts_with(Prefix: "vperm2f128.") || // Added in 6.0
202 Name.starts_with(Prefix: "vpermil.")); // Added in 3.1
203
204 if (Name.consume_front(Prefix: "avx2."))
205 return (Name == "movntdqa" || // Added in 5.0
206 Name.starts_with(Prefix: "pabs.") || // Added in 6.0
207 Name.starts_with(Prefix: "padds.") || // Added in 8.0
208 Name.starts_with(Prefix: "paddus.") || // Added in 8.0
209 Name.starts_with(Prefix: "pblendd.") || // Added in 3.7
210 Name == "pblendw" || // Added in 3.7
211 Name.starts_with(Prefix: "pbroadcast") || // Added in 3.8
212 Name.starts_with(Prefix: "pcmpeq.") || // Added in 3.1
213 Name.starts_with(Prefix: "pcmpgt.") || // Added in 3.1
214 Name.starts_with(Prefix: "pmax") || // Added in 3.9
215 Name.starts_with(Prefix: "pmin") || // Added in 3.9
216 Name.starts_with(Prefix: "pmovsx") || // Added in 3.9
217 Name.starts_with(Prefix: "pmovzx") || // Added in 3.9
218 Name.starts_with(Prefix: "pmulh.w") || // Added in 24.0
219 Name.starts_with(Prefix: "pmulhu.w") || // Added in 24.0
220 Name == "pmul.dq" || // Added in 7.0
221 Name == "pmulu.dq" || // Added in 7.0
222 Name.starts_with(Prefix: "psll.dq") || // Added in 3.7
223 Name.starts_with(Prefix: "psrl.dq") || // Added in 3.7
224 Name.starts_with(Prefix: "psubs.") || // Added in 8.0
225 Name.starts_with(Prefix: "psubus.") || // Added in 8.0
226 Name.starts_with(Prefix: "vbroadcast") || // Added in 3.8
227 Name == "vbroadcasti128" || // Added in 3.7
228 Name == "vextracti128" || // Added in 3.7
229 Name == "vinserti128" || // Added in 3.7
230 Name == "vperm2i128"); // Added in 6.0
231
232 if (Name.consume_front(Prefix: "avx512.")) {
233 if (Name.consume_front(Prefix: "mask."))
234 // 'avx512.mask.*'
235 return (Name.starts_with(Prefix: "add.p") || // Added in 7.0. 128/256 in 4.0
236 Name.starts_with(Prefix: "and.") || // Added in 3.9
237 Name.starts_with(Prefix: "andn.") || // Added in 3.9
238 Name.starts_with(Prefix: "broadcast.s") || // Added in 3.9
239 Name.starts_with(Prefix: "broadcastf32x4.") || // Added in 6.0
240 Name.starts_with(Prefix: "broadcastf32x8.") || // Added in 6.0
241 Name.starts_with(Prefix: "broadcastf64x2.") || // Added in 6.0
242 Name.starts_with(Prefix: "broadcastf64x4.") || // Added in 6.0
243 Name.starts_with(Prefix: "broadcasti32x4.") || // Added in 6.0
244 Name.starts_with(Prefix: "broadcasti32x8.") || // Added in 6.0
245 Name.starts_with(Prefix: "broadcasti64x2.") || // Added in 6.0
246 Name.starts_with(Prefix: "broadcasti64x4.") || // Added in 6.0
247 Name.starts_with(Prefix: "cmp.b") || // Added in 5.0
248 Name.starts_with(Prefix: "cmp.d") || // Added in 5.0
249 Name.starts_with(Prefix: "cmp.q") || // Added in 5.0
250 Name.starts_with(Prefix: "cmp.w") || // Added in 5.0
251 Name.starts_with(Prefix: "compress.b") || // Added in 9.0
252 Name.starts_with(Prefix: "compress.d") || // Added in 9.0
253 Name.starts_with(Prefix: "compress.p") || // Added in 9.0
254 Name.starts_with(Prefix: "compress.q") || // Added in 9.0
255 Name.starts_with(Prefix: "compress.store.") || // Added in 7.0
256 Name.starts_with(Prefix: "compress.w") || // Added in 9.0
257 Name.starts_with(Prefix: "conflict.") || // Added in 9.0
258 Name.starts_with(Prefix: "cvtdq2pd.") || // Added in 4.0
259 Name.starts_with(Prefix: "cvtdq2ps.") || // Added in 7.0 updated 9.0
260 Name == "cvtpd2dq.256" || // Added in 7.0
261 Name == "cvtpd2ps.256" || // Added in 7.0
262 Name == "cvtps2pd.128" || // Added in 7.0
263 Name == "cvtps2pd.256" || // Added in 7.0
264 Name.starts_with(Prefix: "cvtqq2pd.") || // Added in 7.0 updated 9.0
265 Name == "cvtqq2ps.256" || // Added in 9.0
266 Name == "cvtqq2ps.512" || // Added in 9.0
267 Name == "cvttpd2dq.256" || // Added in 7.0
268 Name == "cvttps2dq.128" || // Added in 7.0
269 Name == "cvttps2dq.256" || // Added in 7.0
270 Name.starts_with(Prefix: "cvtudq2pd.") || // Added in 4.0
271 Name.starts_with(Prefix: "cvtudq2ps.") || // Added in 7.0 updated 9.0
272 Name.starts_with(Prefix: "cvtuqq2pd.") || // Added in 7.0 updated 9.0
273 Name == "cvtuqq2ps.256" || // Added in 9.0
274 Name == "cvtuqq2ps.512" || // Added in 9.0
275 Name.starts_with(Prefix: "dbpsadbw.") || // Added in 7.0
276 Name.starts_with(Prefix: "div.p") || // Added in 7.0. 128/256 in 4.0
277 Name.starts_with(Prefix: "expand.b") || // Added in 9.0
278 Name.starts_with(Prefix: "expand.d") || // Added in 9.0
279 Name.starts_with(Prefix: "expand.load.") || // Added in 7.0
280 Name.starts_with(Prefix: "expand.p") || // Added in 9.0
281 Name.starts_with(Prefix: "expand.q") || // Added in 9.0
282 Name.starts_with(Prefix: "expand.w") || // Added in 9.0
283 Name.starts_with(Prefix: "fpclass.p") || // Added in 7.0
284 Name.starts_with(Prefix: "insert") || // Added in 4.0
285 Name.starts_with(Prefix: "load.") || // Added in 3.9
286 Name.starts_with(Prefix: "loadu.") || // Added in 3.9
287 Name.starts_with(Prefix: "lzcnt.") || // Added in 5.0
288 Name.starts_with(Prefix: "max.p") || // Added in 7.0. 128/256 in 5.0
289 Name.starts_with(Prefix: "min.p") || // Added in 7.0. 128/256 in 5.0
290 Name.starts_with(Prefix: "movddup") || // Added in 3.9
291 Name.starts_with(Prefix: "move.s") || // Added in 4.0
292 Name.starts_with(Prefix: "movshdup") || // Added in 3.9
293 Name.starts_with(Prefix: "movsldup") || // Added in 3.9
294 Name.starts_with(Prefix: "mul.p") || // Added in 7.0. 128/256 in 4.0
295 Name.starts_with(Prefix: "or.") || // Added in 3.9
296 Name.starts_with(Prefix: "pabs.") || // Added in 6.0
297 Name.starts_with(Prefix: "packssdw.") || // Added in 5.0
298 Name.starts_with(Prefix: "packsswb.") || // Added in 5.0
299 Name.starts_with(Prefix: "packusdw.") || // Added in 5.0
300 Name.starts_with(Prefix: "packuswb.") || // Added in 5.0
301 Name.starts_with(Prefix: "padd.") || // Added in 4.0
302 Name.starts_with(Prefix: "padds.") || // Added in 8.0
303 Name.starts_with(Prefix: "paddus.") || // Added in 8.0
304 Name.starts_with(Prefix: "palignr.") || // Added in 3.9
305 Name.starts_with(Prefix: "pand.") || // Added in 3.9
306 Name.starts_with(Prefix: "pandn.") || // Added in 3.9
307 Name.starts_with(Prefix: "pavg") || // Added in 6.0
308 Name.starts_with(Prefix: "pbroadcast") || // Added in 6.0
309 Name.starts_with(Prefix: "pcmpeq.") || // Added in 3.9
310 Name.starts_with(Prefix: "pcmpgt.") || // Added in 3.9
311 Name.starts_with(Prefix: "perm.df.") || // Added in 3.9
312 Name.starts_with(Prefix: "perm.di.") || // Added in 3.9
313 Name.starts_with(Prefix: "permvar.") || // Added in 7.0
314 Name.starts_with(Prefix: "pmaddubs.w.") || // Added in 7.0
315 Name.starts_with(Prefix: "pmaddw.d.") || // Added in 7.0
316 Name.starts_with(Prefix: "pmax") || // Added in 4.0
317 Name.starts_with(Prefix: "pmin") || // Added in 4.0
318 Name == "pmov.qd.256" || // Added in 9.0
319 Name == "pmov.qd.512" || // Added in 9.0
320 Name == "pmov.wb.256" || // Added in 9.0
321 Name == "pmov.wb.512" || // Added in 9.0
322 Name.starts_with(Prefix: "pmovsx") || // Added in 4.0
323 Name.starts_with(Prefix: "pmovzx") || // Added in 4.0
324 Name.starts_with(Prefix: "pmul.dq.") || // Added in 4.0
325 Name.starts_with(Prefix: "pmul.hr.sw.") || // Added in 7.0
326 Name.starts_with(Prefix: "pmulh.w.") || // Added in 7.0
327 Name.starts_with(Prefix: "pmulhu.w.") || // Added in 7.0
328 Name.starts_with(Prefix: "pmull.") || // Added in 4.0
329 Name.starts_with(Prefix: "pmultishift.qb.") || // Added in 8.0
330 Name.starts_with(Prefix: "pmulu.dq.") || // Added in 4.0
331 Name.starts_with(Prefix: "por.") || // Added in 3.9
332 Name.starts_with(Prefix: "prol.") || // Added in 8.0
333 Name.starts_with(Prefix: "prolv.") || // Added in 8.0
334 Name.starts_with(Prefix: "pror.") || // Added in 8.0
335 Name.starts_with(Prefix: "prorv.") || // Added in 8.0
336 Name.starts_with(Prefix: "pshuf.b.") || // Added in 4.0
337 Name.starts_with(Prefix: "pshuf.d.") || // Added in 3.9
338 Name.starts_with(Prefix: "pshufh.w.") || // Added in 3.9
339 Name.starts_with(Prefix: "pshufl.w.") || // Added in 3.9
340 Name.starts_with(Prefix: "psll.d") || // Added in 4.0
341 Name.starts_with(Prefix: "psll.q") || // Added in 4.0
342 Name.starts_with(Prefix: "psll.w") || // Added in 4.0
343 Name.starts_with(Prefix: "pslli") || // Added in 4.0
344 Name.starts_with(Prefix: "psllv") || // Added in 4.0
345 Name.starts_with(Prefix: "psra.d") || // Added in 4.0
346 Name.starts_with(Prefix: "psra.q") || // Added in 4.0
347 Name.starts_with(Prefix: "psra.w") || // Added in 4.0
348 Name.starts_with(Prefix: "psrai") || // Added in 4.0
349 Name.starts_with(Prefix: "psrav") || // Added in 4.0
350 Name.starts_with(Prefix: "psrl.d") || // Added in 4.0
351 Name.starts_with(Prefix: "psrl.q") || // Added in 4.0
352 Name.starts_with(Prefix: "psrl.w") || // Added in 4.0
353 Name.starts_with(Prefix: "psrli") || // Added in 4.0
354 Name.starts_with(Prefix: "psrlv") || // Added in 4.0
355 Name.starts_with(Prefix: "psub.") || // Added in 4.0
356 Name.starts_with(Prefix: "psubs.") || // Added in 8.0
357 Name.starts_with(Prefix: "psubus.") || // Added in 8.0
358 Name.starts_with(Prefix: "pternlog.") || // Added in 7.0
359 Name.starts_with(Prefix: "punpckh") || // Added in 3.9
360 Name.starts_with(Prefix: "punpckl") || // Added in 3.9
361 Name.starts_with(Prefix: "pxor.") || // Added in 3.9
362 Name.starts_with(Prefix: "shuf.f") || // Added in 6.0
363 Name.starts_with(Prefix: "shuf.i") || // Added in 6.0
364 Name.starts_with(Prefix: "shuf.p") || // Added in 4.0
365 Name.starts_with(Prefix: "sqrt.p") || // Added in 7.0
366 Name.starts_with(Prefix: "store.b.") || // Added in 3.9
367 Name.starts_with(Prefix: "store.d.") || // Added in 3.9
368 Name.starts_with(Prefix: "store.p") || // Added in 3.9
369 Name.starts_with(Prefix: "store.q.") || // Added in 3.9
370 Name.starts_with(Prefix: "store.w.") || // Added in 3.9
371 Name == "store.ss" || // Added in 7.0
372 Name.starts_with(Prefix: "storeu.") || // Added in 3.9
373 Name.starts_with(Prefix: "sub.p") || // Added in 7.0. 128/256 in 4.0
374 Name.starts_with(Prefix: "ucmp.") || // Added in 5.0
375 Name.starts_with(Prefix: "unpckh.") || // Added in 3.9
376 Name.starts_with(Prefix: "unpckl.") || // Added in 3.9
377 Name.starts_with(Prefix: "valign.") || // Added in 4.0
378 Name == "vcvtph2ps.128" || // Added in 11.0
379 Name == "vcvtph2ps.256" || // Added in 11.0
380 Name.starts_with(Prefix: "vextract") || // Added in 4.0
381 Name.starts_with(Prefix: "vfmadd.") || // Added in 7.0
382 Name.starts_with(Prefix: "vfmaddsub.") || // Added in 7.0
383 Name.starts_with(Prefix: "vfnmadd.") || // Added in 7.0
384 Name.starts_with(Prefix: "vfnmsub.") || // Added in 7.0
385 Name.starts_with(Prefix: "vpdpbusd.") || // Added in 7.0
386 Name.starts_with(Prefix: "vpdpbusds.") || // Added in 7.0
387 Name.starts_with(Prefix: "vpdpwssd.") || // Added in 7.0
388 Name.starts_with(Prefix: "vpdpwssds.") || // Added in 7.0
389 Name.starts_with(Prefix: "vpermi2var.") || // Added in 7.0
390 Name.starts_with(Prefix: "vpermil.p") || // Added in 3.9
391 Name.starts_with(Prefix: "vpermilvar.") || // Added in 4.0
392 Name.starts_with(Prefix: "vpermt2var.") || // Added in 7.0
393 Name.starts_with(Prefix: "vpmadd52") || // Added in 7.0
394 Name.starts_with(Prefix: "vpshld.") || // Added in 7.0
395 Name.starts_with(Prefix: "vpshldv.") || // Added in 8.0
396 Name.starts_with(Prefix: "vpshrd.") || // Added in 7.0
397 Name.starts_with(Prefix: "vpshrdv.") || // Added in 8.0
398 Name.starts_with(Prefix: "vpshufbitqmb.") || // Added in 8.0
399 Name.starts_with(Prefix: "xor.")); // Added in 3.9
400
401 if (Name.consume_front(Prefix: "mask3."))
402 // 'avx512.mask3.*'
403 return (Name.starts_with(Prefix: "vfmadd.") || // Added in 7.0
404 Name.starts_with(Prefix: "vfmaddsub.") || // Added in 7.0
405 Name.starts_with(Prefix: "vfmsub.") || // Added in 7.0
406 Name.starts_with(Prefix: "vfmsubadd.") || // Added in 7.0
407 Name.starts_with(Prefix: "vfnmsub.")); // Added in 7.0
408
409 if (Name.consume_front(Prefix: "maskz."))
410 // 'avx512.maskz.*'
411 return (Name.starts_with(Prefix: "pternlog.") || // Added in 7.0
412 Name.starts_with(Prefix: "vfmadd.") || // Added in 7.0
413 Name.starts_with(Prefix: "vfmaddsub.") || // Added in 7.0
414 Name.starts_with(Prefix: "vpdpbusd.") || // Added in 7.0
415 Name.starts_with(Prefix: "vpdpbusds.") || // Added in 7.0
416 Name.starts_with(Prefix: "vpdpwssd.") || // Added in 7.0
417 Name.starts_with(Prefix: "vpdpwssds.") || // Added in 7.0
418 Name.starts_with(Prefix: "vpermt2var.") || // Added in 7.0
419 Name.starts_with(Prefix: "vpmadd52") || // Added in 7.0
420 Name.starts_with(Prefix: "vpshldv.") || // Added in 8.0
421 Name.starts_with(Prefix: "vpshrdv.")); // Added in 8.0
422
423 // 'avx512.*'
424 return (Name == "movntdqa" || // Added in 5.0
425 Name == "pmul.dq.512" || // Added in 7.0
426 Name == "pmulu.dq.512" || // Added in 7.0
427 Name.starts_with(Prefix: "broadcastm") || // Added in 6.0
428 Name.starts_with(Prefix: "cmp.p") || // Added in 12.0
429 Name.starts_with(Prefix: "cvtb2mask.") || // Added in 7.0
430 Name.starts_with(Prefix: "cvtd2mask.") || // Added in 7.0
431 Name.starts_with(Prefix: "cvtmask2") || // Added in 5.0
432 Name.starts_with(Prefix: "cvtq2mask.") || // Added in 7.0
433 Name == "cvtusi2sd" || // Added in 7.0
434 Name.starts_with(Prefix: "cvtw2mask.") || // Added in 7.0
435 Name == "kand.w" || // Added in 7.0
436 Name == "kandn.w" || // Added in 7.0
437 Name == "knot.w" || // Added in 7.0
438 Name == "kor.w" || // Added in 7.0
439 Name == "kortestc.w" || // Added in 7.0
440 Name == "kortestz.w" || // Added in 7.0
441 Name.starts_with(Prefix: "kunpck") || // added in 6.0
442 Name == "kxnor.w" || // Added in 7.0
443 Name == "kxor.w" || // Added in 7.0
444 Name.starts_with(Prefix: "padds.") || // Added in 8.0
445 Name.starts_with(Prefix: "pbroadcast") || // Added in 3.9
446 Name.starts_with(Prefix: "pmulh.w") || // Added in 24.0
447 Name.starts_with(Prefix: "pmulhu.w") || // Added in 24.0
448 Name.starts_with(Prefix: "prol") || // Added in 8.0
449 Name.starts_with(Prefix: "pror") || // Added in 8.0
450 Name.starts_with(Prefix: "psll.dq") || // Added in 3.9
451 Name.starts_with(Prefix: "psrl.dq") || // Added in 3.9
452 Name.starts_with(Prefix: "psubs.") || // Added in 8.0
453 Name.starts_with(Prefix: "ptestm") || // Added in 6.0
454 Name.starts_with(Prefix: "ptestnm") || // Added in 6.0
455 Name.starts_with(Prefix: "storent.") || // Added in 3.9
456 Name.starts_with(Prefix: "vbroadcast.s") || // Added in 7.0
457 Name.starts_with(Prefix: "vpshld.") || // Added in 8.0
458 Name.starts_with(Prefix: "vpshrd.")); // Added in 8.0
459 }
460
461 if (Name.consume_front(Prefix: "fma."))
462 return (Name.starts_with(Prefix: "vfmadd.") || // Added in 7.0
463 Name.starts_with(Prefix: "vfmsub.") || // Added in 7.0
464 Name.starts_with(Prefix: "vfmsubadd.") || // Added in 7.0
465 Name.starts_with(Prefix: "vfnmadd.") || // Added in 7.0
466 Name.starts_with(Prefix: "vfnmsub.")); // Added in 7.0
467
468 if (Name.consume_front(Prefix: "fma4."))
469 return Name.starts_with(Prefix: "vfmadd.s"); // Added in 7.0
470
471 if (Name.consume_front(Prefix: "sse."))
472 return (Name == "add.ss" || // Added in 4.0
473 Name == "cvtsi2ss" || // Added in 7.0
474 Name == "cvtsi642ss" || // Added in 7.0
475 Name == "div.ss" || // Added in 4.0
476 Name == "mul.ss" || // Added in 4.0
477 Name.starts_with(Prefix: "sqrt.p") || // Added in 7.0
478 Name == "sqrt.ss" || // Added in 7.0
479 Name.starts_with(Prefix: "storeu.") || // Added in 3.9
480 Name == "sub.ss"); // Added in 4.0
481
482 if (Name.consume_front(Prefix: "sse2."))
483 return (Name == "add.sd" || // Added in 4.0
484 Name == "cvtdq2pd" || // Added in 3.9
485 Name == "cvtdq2ps" || // Added in 7.0
486 Name == "cvtps2pd" || // Added in 3.9
487 Name == "cvtsi2sd" || // Added in 7.0
488 Name == "cvtsi642sd" || // Added in 7.0
489 Name == "cvtss2sd" || // Added in 7.0
490 Name == "div.sd" || // Added in 4.0
491 Name == "mul.sd" || // Added in 4.0
492 Name.starts_with(Prefix: "padds.") || // Added in 8.0
493 Name.starts_with(Prefix: "paddus.") || // Added in 8.0
494 Name.starts_with(Prefix: "pcmpeq.") || // Added in 3.1
495 Name.starts_with(Prefix: "pcmpgt.") || // Added in 3.1
496 Name == "pmaxs.w" || // Added in 3.9
497 Name == "pmaxu.b" || // Added in 3.9
498 Name == "pmins.w" || // Added in 3.9
499 Name == "pminu.b" || // Added in 3.9
500 Name == "pmulh.w" || // Added in 24.0
501 Name == "pmulhu.w" || // Added in 24.0
502 Name == "pmulu.dq" || // Added in 7.0
503 Name.starts_with(Prefix: "pshuf") || // Added in 3.9
504 Name.starts_with(Prefix: "psll.dq") || // Added in 3.7
505 Name.starts_with(Prefix: "psrl.dq") || // Added in 3.7
506 Name.starts_with(Prefix: "psubs.") || // Added in 8.0
507 Name.starts_with(Prefix: "psubus.") || // Added in 8.0
508 Name.starts_with(Prefix: "sqrt.p") || // Added in 7.0
509 Name == "sqrt.sd" || // Added in 7.0
510 Name == "storel.dq" || // Added in 3.9
511 Name.starts_with(Prefix: "storeu.") || // Added in 3.9
512 Name == "sub.sd"); // Added in 4.0
513
514 if (Name.consume_front(Prefix: "sse41."))
515 return (Name.starts_with(Prefix: "blendp") || // Added in 3.7
516 Name == "movntdqa" || // Added in 5.0
517 Name == "pblendw" || // Added in 3.7
518 Name == "pmaxsb" || // Added in 3.9
519 Name == "pmaxsd" || // Added in 3.9
520 Name == "pmaxud" || // Added in 3.9
521 Name == "pmaxuw" || // Added in 3.9
522 Name == "pminsb" || // Added in 3.9
523 Name == "pminsd" || // Added in 3.9
524 Name == "pminud" || // Added in 3.9
525 Name == "pminuw" || // Added in 3.9
526 Name.starts_with(Prefix: "pmovsx") || // Added in 3.8
527 Name.starts_with(Prefix: "pmovzx") || // Added in 3.9
528 Name == "pmuldq"); // Added in 7.0
529
530 if (Name.consume_front(Prefix: "sse42."))
531 return Name == "crc32.64.8"; // Added in 3.4
532
533 if (Name.consume_front(Prefix: "sse4a."))
534 return Name.starts_with(Prefix: "movnt."); // Added in 3.9
535
536 if (Name.consume_front(Prefix: "ssse3."))
537 return (Name == "pabs.b.128" || // Added in 6.0
538 Name == "pabs.d.128" || // Added in 6.0
539 Name == "pabs.w.128"); // Added in 6.0
540
541 if (Name.consume_front(Prefix: "xop."))
542 return (Name == "vpcmov" || // Added in 3.8
543 Name == "vpcmov.256" || // Added in 5.0
544 Name.starts_with(Prefix: "vpcom") || // Added in 3.2, Updated in 9.0
545 Name.starts_with(Prefix: "vprot")); // Added in 8.0
546
547 if (Name.consume_front(Prefix: "bmi."))
548 return (Name.starts_with(Prefix: "pdep.") || // Added in 23.0
549 Name.starts_with(Prefix: "pext.")); // Added in 23.0
550
551 return (Name == "addcarry.u32" || // Added in 8.0
552 Name == "addcarry.u64" || // Added in 8.0
553 Name == "addcarryx.u32" || // Added in 8.0
554 Name == "addcarryx.u64" || // Added in 8.0
555 Name == "subborrow.u32" || // Added in 8.0
556 Name == "subborrow.u64" || // Added in 8.0
557 Name.starts_with(Prefix: "vcvtph2ps.")); // Added in 11.0
558}
559
560static bool upgradeX86IntrinsicFunction(Function *F, StringRef Name,
561 Function *&NewFn) {
562 // Only handle intrinsics that start with "x86.".
563 if (!Name.consume_front(Prefix: "x86."))
564 return false;
565
566 if (shouldUpgradeX86Intrinsic(F, Name)) {
567 NewFn = nullptr;
568 return true;
569 }
570
571 if (Name == "rdtscp") { // Added in 8.0
572 // If this intrinsic has 0 operands, it's the new version.
573 if (F->getFunctionType()->getNumParams() == 0)
574 return false;
575
576 rename(GV: F);
577 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(),
578 id: Intrinsic::x86_rdtscp);
579 return true;
580 }
581
582 Intrinsic::ID ID;
583
584 // SSE4.1 ptest functions may have an old signature.
585 if (Name.consume_front(Prefix: "sse41.ptest")) { // Added in 3.2
586 ID = StringSwitch<Intrinsic::ID>(Name)
587 .Case(S: "c", Value: Intrinsic::x86_sse41_ptestc)
588 .Case(S: "z", Value: Intrinsic::x86_sse41_ptestz)
589 .Case(S: "nzc", Value: Intrinsic::x86_sse41_ptestnzc)
590 .Default(Value: Intrinsic::not_intrinsic);
591 if (ID != Intrinsic::not_intrinsic)
592 return upgradePTESTIntrinsic(F, IID: ID, NewFn);
593
594 return false;
595 }
596
597 // Several blend and other instructions with masks used the wrong number of
598 // bits.
599
600 // Added in 3.6
601 ID = StringSwitch<Intrinsic::ID>(Name)
602 .Case(S: "sse41.insertps", Value: Intrinsic::x86_sse41_insertps)
603 .Case(S: "sse41.dppd", Value: Intrinsic::x86_sse41_dppd)
604 .Case(S: "sse41.dpps", Value: Intrinsic::x86_sse41_dpps)
605 .Case(S: "sse41.mpsadbw", Value: Intrinsic::x86_sse41_mpsadbw)
606 .Case(S: "avx.dp.ps.256", Value: Intrinsic::x86_avx_dp_ps_256)
607 .Case(S: "avx2.mpsadbw", Value: Intrinsic::x86_avx2_mpsadbw)
608 .Default(Value: Intrinsic::not_intrinsic);
609 if (ID != Intrinsic::not_intrinsic)
610 return upgradeX86IntrinsicsWith8BitMask(F, IID: ID, NewFn);
611
612 if (Name.consume_front(Prefix: "avx512.")) {
613 if (Name.consume_front(Prefix: "mask.cmp.")) {
614 // Added in 7.0
615 ID = StringSwitch<Intrinsic::ID>(Name)
616 .Case(S: "pd.128", Value: Intrinsic::x86_avx512_mask_cmp_pd_128)
617 .Case(S: "pd.256", Value: Intrinsic::x86_avx512_mask_cmp_pd_256)
618 .Case(S: "pd.512", Value: Intrinsic::x86_avx512_mask_cmp_pd_512)
619 .Case(S: "ps.128", Value: Intrinsic::x86_avx512_mask_cmp_ps_128)
620 .Case(S: "ps.256", Value: Intrinsic::x86_avx512_mask_cmp_ps_256)
621 .Case(S: "ps.512", Value: Intrinsic::x86_avx512_mask_cmp_ps_512)
622 .Default(Value: Intrinsic::not_intrinsic);
623 if (ID != Intrinsic::not_intrinsic)
624 return upgradeX86MaskedFPCompare(F, IID: ID, NewFn);
625 } else if (Name.starts_with(Prefix: "vpdpbusd.") ||
626 Name.starts_with(Prefix: "vpdpbusds.")) {
627 // Added in 21.1
628 ID = StringSwitch<Intrinsic::ID>(Name)
629 .Case(S: "vpdpbusd.128", Value: Intrinsic::x86_avx512_vpdpbusd_128)
630 .Case(S: "vpdpbusd.256", Value: Intrinsic::x86_avx512_vpdpbusd_256)
631 .Case(S: "vpdpbusd.512", Value: Intrinsic::x86_avx512_vpdpbusd_512)
632 .Case(S: "vpdpbusds.128", Value: Intrinsic::x86_avx512_vpdpbusds_128)
633 .Case(S: "vpdpbusds.256", Value: Intrinsic::x86_avx512_vpdpbusds_256)
634 .Case(S: "vpdpbusds.512", Value: Intrinsic::x86_avx512_vpdpbusds_512)
635 .Default(Value: Intrinsic::not_intrinsic);
636 if (ID != Intrinsic::not_intrinsic)
637 return upgradeX86MultiplyAddBytes(F, IID: ID, NewFn);
638 } else if (Name.starts_with(Prefix: "vpdpwssd.") ||
639 Name.starts_with(Prefix: "vpdpwssds.")) {
640 // Added in 21.1
641 ID = StringSwitch<Intrinsic::ID>(Name)
642 .Case(S: "vpdpwssd.128", Value: Intrinsic::x86_avx512_vpdpwssd_128)
643 .Case(S: "vpdpwssd.256", Value: Intrinsic::x86_avx512_vpdpwssd_256)
644 .Case(S: "vpdpwssd.512", Value: Intrinsic::x86_avx512_vpdpwssd_512)
645 .Case(S: "vpdpwssds.128", Value: Intrinsic::x86_avx512_vpdpwssds_128)
646 .Case(S: "vpdpwssds.256", Value: Intrinsic::x86_avx512_vpdpwssds_256)
647 .Case(S: "vpdpwssds.512", Value: Intrinsic::x86_avx512_vpdpwssds_512)
648 .Default(Value: Intrinsic::not_intrinsic);
649 if (ID != Intrinsic::not_intrinsic)
650 return upgradeX86MultiplyAddWords(F, IID: ID, NewFn);
651 }
652 return false; // No other 'x86.avx512.*'.
653 }
654
655 if (Name.consume_front(Prefix: "avx2.")) {
656 if (Name.consume_front(Prefix: "vpdpb")) {
657 // Added in 21.1
658 ID = StringSwitch<Intrinsic::ID>(Name)
659 .Case(S: "ssd.128", Value: Intrinsic::x86_avx2_vpdpbssd_128)
660 .Case(S: "ssd.256", Value: Intrinsic::x86_avx2_vpdpbssd_256)
661 .Case(S: "ssds.128", Value: Intrinsic::x86_avx2_vpdpbssds_128)
662 .Case(S: "ssds.256", Value: Intrinsic::x86_avx2_vpdpbssds_256)
663 .Case(S: "sud.128", Value: Intrinsic::x86_avx2_vpdpbsud_128)
664 .Case(S: "sud.256", Value: Intrinsic::x86_avx2_vpdpbsud_256)
665 .Case(S: "suds.128", Value: Intrinsic::x86_avx2_vpdpbsuds_128)
666 .Case(S: "suds.256", Value: Intrinsic::x86_avx2_vpdpbsuds_256)
667 .Case(S: "uud.128", Value: Intrinsic::x86_avx2_vpdpbuud_128)
668 .Case(S: "uud.256", Value: Intrinsic::x86_avx2_vpdpbuud_256)
669 .Case(S: "uuds.128", Value: Intrinsic::x86_avx2_vpdpbuuds_128)
670 .Case(S: "uuds.256", Value: Intrinsic::x86_avx2_vpdpbuuds_256)
671 .Default(Value: Intrinsic::not_intrinsic);
672 if (ID != Intrinsic::not_intrinsic)
673 return upgradeX86MultiplyAddBytes(F, IID: ID, NewFn);
674 } else if (Name.consume_front(Prefix: "vpdpw")) {
675 // Added in 21.1
676 ID = StringSwitch<Intrinsic::ID>(Name)
677 .Case(S: "sud.128", Value: Intrinsic::x86_avx2_vpdpwsud_128)
678 .Case(S: "sud.256", Value: Intrinsic::x86_avx2_vpdpwsud_256)
679 .Case(S: "suds.128", Value: Intrinsic::x86_avx2_vpdpwsuds_128)
680 .Case(S: "suds.256", Value: Intrinsic::x86_avx2_vpdpwsuds_256)
681 .Case(S: "usd.128", Value: Intrinsic::x86_avx2_vpdpwusd_128)
682 .Case(S: "usd.256", Value: Intrinsic::x86_avx2_vpdpwusd_256)
683 .Case(S: "usds.128", Value: Intrinsic::x86_avx2_vpdpwusds_128)
684 .Case(S: "usds.256", Value: Intrinsic::x86_avx2_vpdpwusds_256)
685 .Case(S: "uud.128", Value: Intrinsic::x86_avx2_vpdpwuud_128)
686 .Case(S: "uud.256", Value: Intrinsic::x86_avx2_vpdpwuud_256)
687 .Case(S: "uuds.128", Value: Intrinsic::x86_avx2_vpdpwuuds_128)
688 .Case(S: "uuds.256", Value: Intrinsic::x86_avx2_vpdpwuuds_256)
689 .Default(Value: Intrinsic::not_intrinsic);
690 if (ID != Intrinsic::not_intrinsic)
691 return upgradeX86MultiplyAddWords(F, IID: ID, NewFn);
692 }
693 return false; // No other 'x86.avx2.*'
694 }
695
696 if (Name.consume_front(Prefix: "avx10.")) {
697 if (Name.consume_front(Prefix: "vpdpb")) {
698 // Added in 21.1
699 ID = StringSwitch<Intrinsic::ID>(Name)
700 .Case(S: "ssd.512", Value: Intrinsic::x86_avx10_vpdpbssd_512)
701 .Case(S: "ssds.512", Value: Intrinsic::x86_avx10_vpdpbssds_512)
702 .Case(S: "sud.512", Value: Intrinsic::x86_avx10_vpdpbsud_512)
703 .Case(S: "suds.512", Value: Intrinsic::x86_avx10_vpdpbsuds_512)
704 .Case(S: "uud.512", Value: Intrinsic::x86_avx10_vpdpbuud_512)
705 .Case(S: "uuds.512", Value: Intrinsic::x86_avx10_vpdpbuuds_512)
706 .Default(Value: Intrinsic::not_intrinsic);
707 if (ID != Intrinsic::not_intrinsic)
708 return upgradeX86MultiplyAddBytes(F, IID: ID, NewFn);
709 } else if (Name.consume_front(Prefix: "vpdpw")) {
710 ID = StringSwitch<Intrinsic::ID>(Name)
711 .Case(S: "sud.512", Value: Intrinsic::x86_avx10_vpdpwsud_512)
712 .Case(S: "suds.512", Value: Intrinsic::x86_avx10_vpdpwsuds_512)
713 .Case(S: "usd.512", Value: Intrinsic::x86_avx10_vpdpwusd_512)
714 .Case(S: "usds.512", Value: Intrinsic::x86_avx10_vpdpwusds_512)
715 .Case(S: "uud.512", Value: Intrinsic::x86_avx10_vpdpwuud_512)
716 .Case(S: "uuds.512", Value: Intrinsic::x86_avx10_vpdpwuuds_512)
717 .Default(Value: Intrinsic::not_intrinsic);
718 if (ID != Intrinsic::not_intrinsic)
719 return upgradeX86MultiplyAddWords(F, IID: ID, NewFn);
720 }
721 return false; // No other 'x86.avx10.*'
722 }
723
724 if (Name.consume_front(Prefix: "avx512bf16.")) {
725 // Added in 9.0
726 ID = StringSwitch<Intrinsic::ID>(Name)
727 .Case(S: "cvtne2ps2bf16.128",
728 Value: Intrinsic::x86_avx512bf16_cvtne2ps2bf16_128)
729 .Case(S: "cvtne2ps2bf16.256",
730 Value: Intrinsic::x86_avx512bf16_cvtne2ps2bf16_256)
731 .Case(S: "cvtne2ps2bf16.512",
732 Value: Intrinsic::x86_avx512bf16_cvtne2ps2bf16_512)
733 .Case(S: "mask.cvtneps2bf16.128",
734 Value: Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128)
735 .Case(S: "cvtneps2bf16.256",
736 Value: Intrinsic::x86_avx512bf16_cvtneps2bf16_256)
737 .Case(S: "cvtneps2bf16.512",
738 Value: Intrinsic::x86_avx512bf16_cvtneps2bf16_512)
739 .Default(Value: Intrinsic::not_intrinsic);
740 if (ID != Intrinsic::not_intrinsic)
741 return upgradeX86BF16Intrinsic(F, IID: ID, NewFn);
742
743 // Added in 9.0
744 ID = StringSwitch<Intrinsic::ID>(Name)
745 .Case(S: "dpbf16ps.128", Value: Intrinsic::x86_avx512bf16_dpbf16ps_128)
746 .Case(S: "dpbf16ps.256", Value: Intrinsic::x86_avx512bf16_dpbf16ps_256)
747 .Case(S: "dpbf16ps.512", Value: Intrinsic::x86_avx512bf16_dpbf16ps_512)
748 .Default(Value: Intrinsic::not_intrinsic);
749 if (ID != Intrinsic::not_intrinsic)
750 return upgradeX86BF16DPIntrinsic(F, IID: ID, NewFn);
751 return false; // No other 'x86.avx512bf16.*'.
752 }
753
754 if (Name.consume_front(Prefix: "xop.")) {
755 Intrinsic::ID ID = Intrinsic::not_intrinsic;
756 if (Name.starts_with(Prefix: "vpermil2")) { // Added in 3.9
757 // Upgrade any XOP PERMIL2 index operand still using a float/double
758 // vector.
759 auto Idx = F->getFunctionType()->getParamType(i: 2);
760 if (Idx->isFPOrFPVectorTy()) {
761 unsigned IdxSize = Idx->getPrimitiveSizeInBits();
762 unsigned EltSize = Idx->getScalarSizeInBits();
763 if (EltSize == 64 && IdxSize == 128)
764 ID = Intrinsic::x86_xop_vpermil2pd;
765 else if (EltSize == 32 && IdxSize == 128)
766 ID = Intrinsic::x86_xop_vpermil2ps;
767 else if (EltSize == 64 && IdxSize == 256)
768 ID = Intrinsic::x86_xop_vpermil2pd_256;
769 else
770 ID = Intrinsic::x86_xop_vpermil2ps_256;
771 }
772 } else if (F->arg_size() == 2)
773 // frcz.ss/sd may need to have an argument dropped. Added in 3.2
774 ID = StringSwitch<Intrinsic::ID>(Name)
775 .Case(S: "vfrcz.ss", Value: Intrinsic::x86_xop_vfrcz_ss)
776 .Case(S: "vfrcz.sd", Value: Intrinsic::x86_xop_vfrcz_sd)
777 .Default(Value: Intrinsic::not_intrinsic);
778
779 if (ID != Intrinsic::not_intrinsic) {
780 rename(GV: F);
781 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
782 return true;
783 }
784 return false; // No other 'x86.xop.*'
785 }
786
787 if (Name == "seh.recoverfp") {
788 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(),
789 id: Intrinsic::eh_recoverfp);
790 return true;
791 }
792
793 return false;
794}
795
796// Upgrade ARM (IsArm) or Aarch64 (!IsArm) intrinsic fns. Return true iff so.
797// IsArm: 'arm.*', !IsArm: 'aarch64.*'.
798static bool upgradeArmOrAarch64IntrinsicFunction(bool IsArm, Function *F,
799 StringRef Name,
800 Function *&NewFn) {
801 if (Name.starts_with(Prefix: "rbit")) {
802 // '(arm|aarch64).rbit'.
803 NewFn = Intrinsic::getOrInsertDeclaration(
804 M: F->getParent(), id: Intrinsic::bitreverse, OverloadTys: F->arg_begin()->getType());
805 return true;
806 }
807
808 if (Name == "thread.pointer") {
809 // '(arm|aarch64).thread.pointer'.
810 NewFn = Intrinsic::getOrInsertDeclaration(
811 M: F->getParent(), id: Intrinsic::thread_pointer, OverloadTys: F->getReturnType());
812 return true;
813 }
814
815 bool Neon = Name.consume_front(Prefix: "neon.");
816 if (Neon) {
817 // '(arm|aarch64).neon.*'.
818 // Changed in 12.0: bfdot accept v4bf16 and v8bf16 instead of v8i8 and
819 // v16i8 respectively.
820 if (Name.consume_front(Prefix: "bfdot.")) {
821 // (arm|aarch64).neon.bfdot.*'.
822 Intrinsic::ID ID =
823 StringSwitch<Intrinsic::ID>(Name)
824 .Cases(CaseStrings: {"v2f32.v8i8", "v4f32.v16i8"},
825 Value: IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfdot
826 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfdot)
827 .Default(Value: Intrinsic::not_intrinsic);
828 if (ID != Intrinsic::not_intrinsic) {
829 size_t OperandWidth = F->getReturnType()->getPrimitiveSizeInBits();
830 assert((OperandWidth == 64 || OperandWidth == 128) &&
831 "Unexpected operand width");
832 LLVMContext &Ctx = F->getParent()->getContext();
833 std::array<Type *, 2> Tys{
834 ._M_elems: {F->getReturnType(),
835 FixedVectorType::get(ElementType: Type::getBFloatTy(C&: Ctx), NumElts: OperandWidth / 16)}};
836 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID, OverloadTys: Tys);
837 return true;
838 }
839 return false; // No other '(arm|aarch64).neon.bfdot.*'.
840 }
841
842 // Changed in 12.0: bfmmla, bfmlalb and bfmlalt are not polymorphic
843 // anymore and accept v8bf16 instead of v16i8.
844 if (Name.consume_front(Prefix: "bfm")) {
845 // (arm|aarch64).neon.bfm*'.
846 if (Name.consume_back(Suffix: ".v4f32.v16i8")) {
847 // (arm|aarch64).neon.bfm*.v4f32.v16i8'.
848 Intrinsic::ID ID =
849 StringSwitch<Intrinsic::ID>(Name)
850 .Case(S: "mla",
851 Value: IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmmla
852 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmmla)
853 .Case(S: "lalb",
854 Value: IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmlalb
855 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmlalb)
856 .Case(S: "lalt",
857 Value: IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmlalt
858 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmlalt)
859 .Default(Value: Intrinsic::not_intrinsic);
860 if (ID != Intrinsic::not_intrinsic) {
861 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
862 return true;
863 }
864 return false; // No other '(arm|aarch64).neon.bfm*.v16i8'.
865 }
866 return false; // No other '(arm|aarch64).neon.bfm*.
867 }
868 // Continue on to Aarch64 Neon or Arm Neon.
869 }
870 // Continue on to Arm or Aarch64.
871
872 if (IsArm) {
873 // 'arm.*'.
874 if (Neon) {
875 // 'arm.neon.*'.
876 Intrinsic::ID ID = StringSwitch<Intrinsic::ID>(Name)
877 .StartsWith(S: "vclz.", Value: Intrinsic::ctlz)
878 .StartsWith(S: "vcnt.", Value: Intrinsic::ctpop)
879 .StartsWith(S: "vqadds.", Value: Intrinsic::sadd_sat)
880 .StartsWith(S: "vqaddu.", Value: Intrinsic::uadd_sat)
881 .StartsWith(S: "vqsubs.", Value: Intrinsic::ssub_sat)
882 .StartsWith(S: "vqsubu.", Value: Intrinsic::usub_sat)
883 .StartsWith(S: "vrinta.", Value: Intrinsic::round)
884 .StartsWith(S: "vrintn.", Value: Intrinsic::roundeven)
885 .StartsWith(S: "vrintm.", Value: Intrinsic::floor)
886 .StartsWith(S: "vrintp.", Value: Intrinsic::ceil)
887 .StartsWith(S: "vrintx.", Value: Intrinsic::rint)
888 .StartsWith(S: "vrintz.", Value: Intrinsic::trunc)
889 .Default(Value: Intrinsic::not_intrinsic);
890 if (ID != Intrinsic::not_intrinsic) {
891 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID,
892 OverloadTys: F->arg_begin()->getType());
893 return true;
894 }
895
896 if (Name.consume_front(Prefix: "vst")) {
897 // 'arm.neon.vst*'.
898 static const Regex vstRegex("^([1234]|[234]lane)\\.v[a-z0-9]*$");
899 SmallVector<StringRef, 2> Groups;
900 if (vstRegex.match(String: Name, Matches: &Groups)) {
901 static const Intrinsic::ID StoreInts[] = {
902 Intrinsic::arm_neon_vst1, Intrinsic::arm_neon_vst2,
903 Intrinsic::arm_neon_vst3, Intrinsic::arm_neon_vst4};
904
905 static const Intrinsic::ID StoreLaneInts[] = {
906 Intrinsic::arm_neon_vst2lane, Intrinsic::arm_neon_vst3lane,
907 Intrinsic::arm_neon_vst4lane};
908
909 auto fArgs = F->getFunctionType()->params();
910 Type *Tys[] = {fArgs[0], fArgs[1]};
911 if (Groups[1].size() == 1)
912 NewFn = Intrinsic::getOrInsertDeclaration(
913 M: F->getParent(), id: StoreInts[fArgs.size() - 3], OverloadTys: Tys);
914 else
915 NewFn = Intrinsic::getOrInsertDeclaration(
916 M: F->getParent(), id: StoreLaneInts[fArgs.size() - 5], OverloadTys: Tys);
917 return true;
918 }
919 return false; // No other 'arm.neon.vst*'.
920 }
921
922 return false; // No other 'arm.neon.*'.
923 }
924
925 if (Name.consume_front(Prefix: "mve.")) {
926 // 'arm.mve.*'.
927 if (Name == "vctp64") {
928 if (cast<FixedVectorType>(Val: F->getReturnType())->getNumElements() == 4) {
929 // A vctp64 returning a v4i1 is converted to return a v2i1. Rename
930 // the function and deal with it below in UpgradeIntrinsicCall.
931 rename(GV: F);
932 return true;
933 }
934 return false; // Not 'arm.mve.vctp64'.
935 }
936
937 if (Name.starts_with(Prefix: "vrintn.v")) {
938 NewFn = Intrinsic::getOrInsertDeclaration(
939 M: F->getParent(), id: Intrinsic::roundeven, OverloadTys: F->arg_begin()->getType());
940 return true;
941 }
942
943 // These too are changed to accept a v2i1 instead of the old v4i1.
944 if (Name.consume_back(Suffix: ".v4i1")) {
945 // 'arm.mve.*.v4i1'.
946 if (Name.consume_back(Suffix: ".predicated.v2i64.v4i32"))
947 // 'arm.mve.*.predicated.v2i64.v4i32.v4i1'
948 return Name == "mull.int" || Name == "vqdmull";
949
950 if (Name.consume_back(Suffix: ".v2i64")) {
951 // 'arm.mve.*.v2i64.v4i1'
952 bool IsGather = Name.consume_front(Prefix: "vldr.gather.");
953 if (IsGather || Name.consume_front(Prefix: "vstr.scatter.")) {
954 if (Name.consume_front(Prefix: "base.")) {
955 // Optional 'wb.' prefix.
956 Name.consume_front(Prefix: "wb.");
957 // 'arm.mve.(vldr.gather|vstr.scatter).base.(wb.)?
958 // predicated.v2i64.v2i64.v4i1'.
959 return Name == "predicated.v2i64";
960 }
961
962 if (Name.consume_front(Prefix: "offset.predicated."))
963 return Name == (IsGather ? "v2i64.p0i64" : "p0i64.v2i64") ||
964 Name == (IsGather ? "v2i64.p0" : "p0.v2i64");
965
966 // No other 'arm.mve.(vldr.gather|vstr.scatter).*.v2i64.v4i1'.
967 return false;
968 }
969
970 return false; // No other 'arm.mve.*.v2i64.v4i1'.
971 }
972 return false; // No other 'arm.mve.*.v4i1'.
973 }
974 return false; // No other 'arm.mve.*'.
975 }
976
977 if (Name.consume_front(Prefix: "cde.vcx")) {
978 // 'arm.cde.vcx*'.
979 if (Name.consume_back(Suffix: ".predicated.v2i64.v4i1"))
980 // 'arm.cde.vcx*.predicated.v2i64.v4i1'.
981 return Name == "1q" || Name == "1qa" || Name == "2q" || Name == "2qa" ||
982 Name == "3q" || Name == "3qa";
983
984 return false; // No other 'arm.cde.vcx*'.
985 }
986 } else {
987 // 'aarch64.*'.
988 if (Neon) {
989 // 'aarch64.neon.*'.
990 Intrinsic::ID ID = StringSwitch<Intrinsic::ID>(Name)
991 .StartsWith(S: "frintn", Value: Intrinsic::roundeven)
992 .StartsWith(S: "rbit", Value: Intrinsic::bitreverse)
993 .Default(Value: Intrinsic::not_intrinsic);
994 if (ID != Intrinsic::not_intrinsic) {
995 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID,
996 OverloadTys: F->arg_begin()->getType());
997 return true;
998 }
999
1000 Intrinsic::ID MinMaxID =
1001 StringSwitch<Intrinsic::ID>(Name.split(Separator: '.').first)
1002 .Case(S: "smax", Value: Intrinsic::smax)
1003 .Case(S: "smin", Value: Intrinsic::smin)
1004 .Case(S: "umax", Value: Intrinsic::umax)
1005 .Case(S: "umin", Value: Intrinsic::umin)
1006 .Default(Value: Intrinsic::not_intrinsic);
1007 if (MinMaxID != Intrinsic::not_intrinsic) {
1008 if (F->arg_size() != 2 || !F->getReturnType()->isIntOrIntVectorTy())
1009 return false; // Invalid IR.
1010 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: MinMaxID,
1011 OverloadTys: F->getReturnType());
1012 return true;
1013 }
1014
1015 if (Name.starts_with(Prefix: "addp")) {
1016 // 'aarch64.neon.addp*'.
1017 if (F->arg_size() != 2)
1018 return false; // Invalid IR.
1019 VectorType *Ty = dyn_cast<VectorType>(Val: F->getReturnType());
1020 if (Ty && Ty->getElementType()->isFloatingPointTy()) {
1021 NewFn = Intrinsic::getOrInsertDeclaration(
1022 M: F->getParent(), id: Intrinsic::aarch64_neon_faddp, OverloadTys: Ty);
1023 return true;
1024 }
1025 }
1026
1027 // Changed in 20.0: bfcvt/bfcvtn/bcvtn2 have been replaced with fptrunc.
1028 if (Name.starts_with(Prefix: "bfcvt")) {
1029 NewFn = nullptr;
1030 return true;
1031 }
1032
1033 // vcvtfp2hf and vcvthf2fp -> fpext and fptrunc
1034 if (Name == "vcvtfp2hf" || Name == "vcvthf2fp") {
1035 NewFn = nullptr;
1036 return true;
1037 }
1038
1039 return false; // No other 'aarch64.neon.*'.
1040 }
1041 if (Name.consume_front(Prefix: "sve.")) {
1042 // 'aarch64.sve.*'.
1043 if (Name.consume_front(Prefix: "bf")) {
1044 if (Name == "mmla") {
1045 Type *Tys[] = {F->getReturnType(),
1046 std::next(x: F->arg_begin())->getType()};
1047 NewFn = Intrinsic::getOrInsertDeclaration(
1048 M: F->getParent(), id: Intrinsic::aarch64_sve_fmmla, OverloadTys: Tys);
1049 return true;
1050 }
1051 if (Name.consume_back(Suffix: ".lane")) {
1052 // 'aarch64.sve.bf*.lane'.
1053 Intrinsic::ID ID =
1054 StringSwitch<Intrinsic::ID>(Name)
1055 .Case(S: "dot", Value: Intrinsic::aarch64_sve_bfdot_lane_v2)
1056 .Case(S: "mlalb", Value: Intrinsic::aarch64_sve_bfmlalb_lane_v2)
1057 .Case(S: "mlalt", Value: Intrinsic::aarch64_sve_bfmlalt_lane_v2)
1058 .Default(Value: Intrinsic::not_intrinsic);
1059 if (ID != Intrinsic::not_intrinsic) {
1060 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
1061 return true;
1062 }
1063 return false; // No other 'aarch64.sve.bf*.lane'.
1064 }
1065 return false; // No other 'aarch64.sve.bf*'.
1066 }
1067
1068 // 'aarch64.sve.fcvt.bf16f32' || 'aarch64.sve.fcvtnt.bf16f32'
1069 if (Name == "fcvt.bf16f32" || Name == "fcvtnt.bf16f32") {
1070 NewFn = nullptr;
1071 return true;
1072 }
1073
1074 if (Name.consume_front(Prefix: "convert.from.svbool")) {
1075 // 'aarch64.sve.convert.from.svbool'
1076 auto *TTy = dyn_cast<TargetExtType>(Val: F->getReturnType());
1077 if (!TTy || TTy->getName() != "aarch64.svcount")
1078 return false;
1079
1080 Intrinsic::ID ID = Intrinsic::aarch64_sve_convert_to_svcount;
1081 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
1082 return true;
1083 }
1084
1085 if (Name.consume_front(Prefix: "convert.to.svbool")) {
1086 // 'aarch64.sve.convert.to.svbool'
1087 auto *TTy = dyn_cast<TargetExtType>(Val: F->arg_begin()->getType());
1088 if (!TTy || TTy->getName() != "aarch64.svcount")
1089 return false;
1090
1091 Intrinsic::ID ID = Intrinsic::aarch64_sve_convert_from_svcount;
1092 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
1093 return true;
1094 }
1095
1096 if (Name.consume_front(Prefix: "addqv")) {
1097 // 'aarch64.sve.addqv'.
1098 if (!F->getReturnType()->isFPOrFPVectorTy())
1099 return false;
1100
1101 auto Args = F->getFunctionType()->params();
1102 Type *Tys[] = {F->getReturnType(), Args[1]};
1103 NewFn = Intrinsic::getOrInsertDeclaration(
1104 M: F->getParent(), id: Intrinsic::aarch64_sve_faddqv, OverloadTys: Tys);
1105 return true;
1106 }
1107
1108 if (Name.consume_front(Prefix: "ld")) {
1109 // 'aarch64.sve.ld*'.
1110 static const Regex LdRegex("^[234](.nxv[a-z0-9]+|$)");
1111 if (LdRegex.match(String: Name)) {
1112 Type *ScalarTy =
1113 cast<VectorType>(Val: F->getReturnType())->getElementType();
1114 ElementCount EC =
1115 cast<VectorType>(Val: F->arg_begin()->getType())->getElementCount();
1116 assert(F->arg_size() == 2 &&
1117 "Expected 2 arguments for ld* intrinsic.");
1118 Type *PtrTy = F->getArg(i: 1)->getType();
1119 Type *Ty = VectorType::get(ElementType: ScalarTy, EC);
1120 static const Intrinsic::ID LoadIDs[] = {
1121 Intrinsic::aarch64_sve_ld2_sret,
1122 Intrinsic::aarch64_sve_ld3_sret,
1123 Intrinsic::aarch64_sve_ld4_sret,
1124 };
1125 NewFn = Intrinsic::getOrInsertDeclaration(
1126 M: F->getParent(), id: LoadIDs[Name[0] - '2'], OverloadTys: {Ty, PtrTy});
1127 return true;
1128 }
1129 return false; // No other 'aarch64.sve.ld*'.
1130 }
1131
1132 if (Name.consume_front(Prefix: "tuple.")) {
1133 // 'aarch64.sve.tuple.*'.
1134 if (Name.starts_with(Prefix: "get")) {
1135 // 'aarch64.sve.tuple.get*'.
1136 Type *Tys[] = {F->getReturnType(), F->arg_begin()->getType()};
1137 NewFn = Intrinsic::getOrInsertDeclaration(
1138 M: F->getParent(), id: Intrinsic::vector_extract, OverloadTys: Tys);
1139 return true;
1140 }
1141
1142 if (Name.starts_with(Prefix: "set")) {
1143 // 'aarch64.sve.tuple.set*'.
1144 auto Args = F->getFunctionType()->params();
1145 Type *Tys[] = {Args[0], Args[2], Args[1]};
1146 NewFn = Intrinsic::getOrInsertDeclaration(
1147 M: F->getParent(), id: Intrinsic::vector_insert, OverloadTys: Tys);
1148 return true;
1149 }
1150
1151 static const Regex CreateTupleRegex("^create[234](.nxv[a-z0-9]+|$)");
1152 if (CreateTupleRegex.match(String: Name)) {
1153 // 'aarch64.sve.tuple.create*'.
1154 auto Args = F->getFunctionType()->params();
1155 Type *Tys[] = {F->getReturnType(), Args[1]};
1156 NewFn = Intrinsic::getOrInsertDeclaration(
1157 M: F->getParent(), id: Intrinsic::vector_insert, OverloadTys: Tys);
1158 return true;
1159 }
1160 return false; // No other 'aarch64.sve.tuple.*'.
1161 }
1162
1163 if (Name.starts_with(Prefix: "rev.nxv")) {
1164 // 'aarch64.sve.rev.<Ty>'
1165 NewFn = Intrinsic::getOrInsertDeclaration(
1166 M: F->getParent(), id: Intrinsic::vector_reverse, OverloadTys: F->getReturnType());
1167 return true;
1168 }
1169
1170 return false; // No other 'aarch64.sve.*'.
1171 }
1172 if (Name.consume_front(Prefix: "sme.")) {
1173 // 'aarch64.sme.*'.
1174 if (Name.consume_front(Prefix: "ftmopa.")) {
1175 // The FP8 FTMOPA intrinsics were split out from the non-FP8 FTMOPA
1176 // intrinsics to model their FPMR dependency.
1177 Intrinsic::ID ID =
1178 StringSwitch<Intrinsic::ID>(Name)
1179 .Case(S: "za16.nxv16i8", Value: Intrinsic::aarch64_sme_fp8_ftmopa_za16)
1180 .Case(S: "za32.nxv16i8", Value: Intrinsic::aarch64_sme_fp8_ftmopa_za32)
1181 .Default(Value: Intrinsic::not_intrinsic);
1182 if (ID != Intrinsic::not_intrinsic) {
1183 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
1184 return true;
1185 }
1186 return false; // No other 'aarch64.sme.ftmopa.*'.
1187 }
1188
1189 return false; // No other 'aarch64.sme.*'.
1190 }
1191 }
1192 return false; // No other 'arm.*', 'aarch64.*'.
1193}
1194
1195// The TMA G2S (global-to-shared) tensor copy modes that have legacy
1196// declarations requiring an auto-upgrade. The same set applies to the
1197// cluster (g2s) and CTA (g2s_cta) variants.
1198#define NVVM_TMA_G2S_MODES(M) \
1199 M(tile_1d, "tile.1d") \
1200 M(tile_2d, "tile.2d") \
1201 M(tile_3d, "tile.3d") \
1202 M(tile_4d, "tile.4d") \
1203 M(tile_5d, "tile.5d") \
1204 M(tile_gather4_2d, "tile.gather4.2d") \
1205 M(im2col_3d, "im2col.3d") \
1206 M(im2col_4d, "im2col.4d") \
1207 M(im2col_5d, "im2col.5d") \
1208 M(im2col_w_3d, "im2col.w.3d") \
1209 M(im2col_w_4d, "im2col.w.4d") \
1210 M(im2col_w_5d, "im2col.w.5d") \
1211 M(im2col_w_128_3d, "im2col.w.128.3d") \
1212 M(im2col_w_128_4d, "im2col.w.128.4d") \
1213 M(im2col_w_128_5d, "im2col.w.128.5d")
1214
1215// Two legacy tails are:
1216//
1217// arg1, arg2, .. i64 %ch, i1 %flag_mc, i1 %flag_ch
1218// arg1, arg2, .. i64 %ch, i1 %flag_mc, i1 %flag_ch, i32 %cta_group
1219//
1220// The current tail appends a trailing i32 %flag_valid_pattern, so both
1221// legacy tails are recognized by an i1 at parameter N-2.
1222static Intrinsic::ID
1223shouldUpgradeNVPTXTMAG2SIntrinsics(Function *F, StringRef Name,
1224 SmallVectorImpl<Type *> &OvlTys) {
1225 if (!Name.consume_front(Prefix: "cp.async.bulk.tensor.g2s."))
1226 return Intrinsic::not_intrinsic;
1227
1228#define G2S_ID(ID_SUFFIX, NAME) \
1229 .Case(NAME, Intrinsic::nvvm_cp_async_bulk_tensor_g2s_##ID_SUFFIX)
1230 // clang-format off
1231 Intrinsic::ID ID = StringSwitch<Intrinsic::ID>(Name)
1232 NVVM_TMA_G2S_MODES(G2S_ID)
1233 .Default(Value: Intrinsic::not_intrinsic);
1234#undef G2S_ID
1235 // clang-format on
1236 if (ID == Intrinsic::not_intrinsic)
1237 return ID;
1238
1239 size_t NumParams = F->getFunctionType()->getNumParams();
1240
1241 // Parameter N-2 is i1 for both legacy tails; the current tail ends
1242 // with i32 %cta_group, i32 %flag_valid_pattern, for which N-2 is i32.
1243 if (!F->getFunctionType()->getParamType(i: NumParams - 2)->isIntegerTy(BitWidth: 1))
1244 return Intrinsic::not_intrinsic;
1245
1246 // The multicast mask is the parameter immediately before the i64
1247 // cache-hint: N-4 for the 2-flag tail, N-5 for the 3-flag tail.
1248 ArrayRef<Type *> Params = F->getFunctionType()->params();
1249 size_t MaskIdx =
1250 Params[NumParams - 1]->isIntegerTy(BitWidth: 1) ? NumParams - 4 : NumParams - 5;
1251 assert(Params[MaskIdx + 1]->isIntegerTy(64) &&
1252 "expected the i64 cache-hint after the multicast mask");
1253 Type *MaskTy = Params[MaskIdx];
1254 assert(MaskTy->isIntegerTy(16) && "unexpected multicast mask type");
1255 OvlTys.push_back(Elt: MaskTy);
1256
1257 return ID;
1258}
1259
1260// The legacy tail is:
1261//
1262// arg1, arg2, .. i64 %ch, i1 %flag_ch
1263//
1264// The current tail appends a trailing i32 %flag_valid_pattern, so the
1265// legacy tail is recognized by an i1 at parameter N-1.
1266static Intrinsic::ID shouldUpgradeNVPTXTMAG2SCTAIntrinsics(Function *F,
1267 StringRef Name) {
1268 if (!Name.consume_front(Prefix: "cp.async.bulk.tensor.g2s.cta."))
1269 return Intrinsic::not_intrinsic;
1270
1271#define G2S_CTA_ID(ID_SUFFIX, NAME) \
1272 .Case(NAME, Intrinsic::nvvm_cp_async_bulk_tensor_g2s_cta_##ID_SUFFIX)
1273 // clang-format off
1274 Intrinsic::ID ID = StringSwitch<Intrinsic::ID>(Name)
1275 NVVM_TMA_G2S_MODES(G2S_CTA_ID)
1276 .Default(Value: Intrinsic::not_intrinsic);
1277#undef G2S_CTA_ID
1278 // clang-format on
1279 if (ID == Intrinsic::not_intrinsic)
1280 return ID;
1281
1282 // Parameter N-1 is i1 for the legacy tail; the current tail ends
1283 // with i32 %flag_valid_pattern, for which N-1 is i32.
1284 if (!F->getFunctionType()
1285 ->getParamType(i: F->getFunctionType()->getNumParams() - 1)
1286 ->isIntegerTy(BitWidth: 1))
1287 return Intrinsic::not_intrinsic;
1288
1289 return ID;
1290}
1291// The legacy tail of llvm.nvvm.cp.async.bulk.global.to.shared.cluster is:
1292//
1293// ..., i16 %mc, i64 %ch, i1 %flag_mc, i1 %flag_ch
1294//
1295// The current intrinsic is overloaded on the multicast-mask type and takes a
1296// trailing i32 %flag_valid_pattern; the legacy tail is recognized by an i1 at
1297// parameter N-1.
1298static Intrinsic::ID
1299shouldUpgradeNVPTXBulkG2SClusterIntrinsic(Function *F, StringRef Name,
1300 SmallVectorImpl<Type *> &OvlTys) {
1301 if (!Name.consume_front(Prefix: "cp.async.bulk.global.to.shared.cluster"))
1302 return Intrinsic::not_intrinsic;
1303
1304 // Parameter N-1 is i1 for the legacy tail; the current tail ends with
1305 // i32 %flag_valid_pattern, for which N-1 is i32.
1306 size_t NumParams = F->getFunctionType()->getNumParams();
1307 if (!F->getFunctionType()->getParamType(i: NumParams - 1)->isIntegerTy(BitWidth: 1))
1308 return Intrinsic::not_intrinsic;
1309
1310 // The multicast mask is parameter 4; legacy IR only uses i16.
1311 Type *MaskTy = F->getFunctionType()->getParamType(i: NumParams - 4);
1312 if (!MaskTy->isIntegerTy(BitWidth: 16))
1313 return Intrinsic::not_intrinsic;
1314 OvlTys.push_back(Elt: MaskTy);
1315
1316 return Intrinsic::nvvm_cp_async_bulk_global_to_shared_cluster;
1317}
1318
1319// The legacy tail of llvm.nvvm.cp.async.bulk.global.to.shared.cta is:
1320//
1321// ..., i64 %ch, i1 %flag_ch
1322//
1323// The current intrinsic adds %ignore_bytes_left/%ignore_bytes_right before
1324// %ch and trailing %flag_oob/%flag_valid_pattern; the legacy tail is
1325// recognized by an i1 at parameter N-1, whereas the current tail ends
1326// with an i32.
1327static Intrinsic::ID shouldUpgradeNVPTXBulkG2SCTAIntrinsic(Function *F,
1328 StringRef Name) {
1329 if (!Name.consume_front(Prefix: "cp.async.bulk.global.to.shared.cta"))
1330 return Intrinsic::not_intrinsic;
1331
1332 // Parameter N-1 is i1 for the legacy tail; the current tail ends with
1333 // i32 %flag_valid_pattern, for which N-1 is i32.
1334 if (!F->getFunctionType()->getParamType(i: 5)->isIntegerTy(BitWidth: 1))
1335 return Intrinsic::not_intrinsic;
1336
1337 return Intrinsic::nvvm_cp_async_bulk_global_to_shared_cta;
1338}
1339
1340// The legacy TMA reduction intrinsics encode the reduction operator in their
1341// name, while the current ones take it as an immediate argument. Map the
1342// operator part of a legacy name to the corresponding immediate value.
1343static std::optional<unsigned> getNVPTXTMAReductionOp(StringRef Name) {
1344 return StringSwitch<std::optional<unsigned>>(Name)
1345 .Case(S: "add", Value: static_cast<unsigned>(nvvm::TMAReductionOp::ADD))
1346 .Case(S: "min", Value: static_cast<unsigned>(nvvm::TMAReductionOp::MIN))
1347 .Case(S: "max", Value: static_cast<unsigned>(nvvm::TMAReductionOp::MAX))
1348 .Case(S: "inc", Value: static_cast<unsigned>(nvvm::TMAReductionOp::INC))
1349 .Case(S: "dec", Value: static_cast<unsigned>(nvvm::TMAReductionOp::DEC))
1350 .Case(S: "and", Value: static_cast<unsigned>(nvvm::TMAReductionOp::AND))
1351 .Case(S: "or", Value: static_cast<unsigned>(nvvm::TMAReductionOp::OR))
1352 .Case(S: "xor", Value: static_cast<unsigned>(nvvm::TMAReductionOp::XOR))
1353 .Default(Value: std::nullopt);
1354}
1355
1356static Intrinsic::ID shouldUpgradeNVPTXTMAReductionIntrinsics(StringRef Name) {
1357 if (!Name.consume_front(Prefix: "cp.async.bulk.tensor.reduce."))
1358 return Intrinsic::not_intrinsic;
1359
1360 auto [RedOpName, ShapeName] = Name.split(Separator: '.');
1361 if (!getNVPTXTMAReductionOp(Name: RedOpName))
1362 return Intrinsic::not_intrinsic;
1363
1364 return StringSwitch<Intrinsic::ID>(ShapeName)
1365 .Case(S: "tile.1d", Value: Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_1d)
1366 .Case(S: "tile.2d", Value: Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_2d)
1367 .Case(S: "tile.3d", Value: Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_3d)
1368 .Case(S: "tile.4d", Value: Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_4d)
1369 .Case(S: "tile.5d", Value: Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_5d)
1370 .Case(S: "im2col.3d", Value: Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_3d)
1371 .Case(S: "im2col.4d", Value: Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_4d)
1372 .Case(S: "im2col.5d", Value: Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_5d)
1373 .Default(Value: Intrinsic::not_intrinsic);
1374}
1375
1376static Intrinsic::ID shouldUpgradeNVPTXSharedClusterIntrinsic(Function *F,
1377 StringRef Name) {
1378 if (Name.consume_front(Prefix: "mapa.shared.cluster"))
1379 if (F->getReturnType()->getPointerAddressSpace() ==
1380 NVPTXAS::ADDRESS_SPACE_SHARED)
1381 return Intrinsic::nvvm_mapa_shared_cluster;
1382
1383 if (Name.consume_front(Prefix: "cp.async.bulk.")) {
1384 Intrinsic::ID ID =
1385 StringSwitch<Intrinsic::ID>(Name)
1386 .Case(S: "shared.cta.to.cluster",
1387 Value: Intrinsic::nvvm_cp_async_bulk_shared_cta_to_cluster)
1388 .Default(Value: Intrinsic::not_intrinsic);
1389
1390 if (ID != Intrinsic::not_intrinsic)
1391 if (F->getArg(i: 0)->getType()->getPointerAddressSpace() ==
1392 NVPTXAS::ADDRESS_SPACE_SHARED)
1393 return ID;
1394 }
1395
1396 return Intrinsic::not_intrinsic;
1397}
1398
1399static Intrinsic::ID
1400shouldUpgradeNVPTXTcgen05CommitSharedIntrinsic(Function *F, StringRef Name) {
1401 if (!Name.consume_front(Prefix: "tcgen05.commit."))
1402 return Intrinsic::not_intrinsic;
1403
1404 if (Name.consume_front(Prefix: "shared."))
1405 return StringSwitch<Intrinsic::ID>(Name)
1406 .Case(S: "cg1", Value: Intrinsic::nvvm_tcgen05_commit_cg1)
1407 .Case(S: "cg2", Value: Intrinsic::nvvm_tcgen05_commit_cg2)
1408 .Default(Value: Intrinsic::not_intrinsic);
1409
1410 if (Name.consume_front(Prefix: "mc.shared.")) {
1411 // Only upgrade older i16 mc variants.
1412 if (!F->getArg(i: 1)->getType()->isIntegerTy(BitWidth: 16))
1413 return Intrinsic::not_intrinsic;
1414
1415 return StringSwitch<Intrinsic::ID>(Name)
1416 .Case(S: "cg1", Value: Intrinsic::nvvm_tcgen05_commit_mc_cg1)
1417 .Case(S: "cg2", Value: Intrinsic::nvvm_tcgen05_commit_mc_cg2)
1418 .Default(Value: Intrinsic::not_intrinsic);
1419 }
1420
1421 return Intrinsic::not_intrinsic;
1422}
1423
1424static Intrinsic::ID
1425shouldUpgradeNVPTXTcgen05AllocDeallocIntrinsic(Function *F, StringRef Name) {
1426 if (F->arg_size() != 2)
1427 return Intrinsic::not_intrinsic;
1428
1429 if (Name.consume_front(Prefix: "tcgen05.alloc.shared.") ||
1430 Name.consume_front(Prefix: "tcgen05.alloc."))
1431 return StringSwitch<Intrinsic::ID>(Name)
1432 .Case(S: "cg1", Value: Intrinsic::nvvm_tcgen05_alloc_cg1)
1433 .Case(S: "cg2", Value: Intrinsic::nvvm_tcgen05_alloc_cg2)
1434 .Default(Value: Intrinsic::not_intrinsic);
1435
1436 if (Name.consume_front(Prefix: "tcgen05.dealloc."))
1437 return StringSwitch<Intrinsic::ID>(Name)
1438 .Case(S: "cg1", Value: Intrinsic::nvvm_tcgen05_dealloc_cg1)
1439 .Case(S: "cg2", Value: Intrinsic::nvvm_tcgen05_dealloc_cg2)
1440 .Default(Value: Intrinsic::not_intrinsic);
1441
1442 return Intrinsic::not_intrinsic;
1443}
1444
1445static Intrinsic::ID shouldUpgradeNVPTXBF16Intrinsic(StringRef Name) {
1446 if (Name.consume_front(Prefix: "fma.rn."))
1447 return StringSwitch<Intrinsic::ID>(Name)
1448 .Case(S: "bf16", Value: Intrinsic::nvvm_fma_rn_bf16)
1449 .Case(S: "bf16x2", Value: Intrinsic::nvvm_fma_rn_bf16x2)
1450 .Case(S: "relu.bf16", Value: Intrinsic::nvvm_fma_rn_relu_bf16)
1451 .Case(S: "relu.bf16x2", Value: Intrinsic::nvvm_fma_rn_relu_bf16x2)
1452 .Default(Value: Intrinsic::not_intrinsic);
1453
1454 if (Name.consume_front(Prefix: "fmax."))
1455 return StringSwitch<Intrinsic::ID>(Name)
1456 .Case(S: "bf16", Value: Intrinsic::nvvm_fmax_bf16)
1457 .Case(S: "bf16x2", Value: Intrinsic::nvvm_fmax_bf16x2)
1458 .Case(S: "ftz.bf16", Value: Intrinsic::nvvm_fmax_ftz_bf16)
1459 .Case(S: "ftz.bf16x2", Value: Intrinsic::nvvm_fmax_ftz_bf16x2)
1460 .Case(S: "ftz.nan.bf16", Value: Intrinsic::nvvm_fmax_ftz_nan_bf16)
1461 .Case(S: "ftz.nan.bf16x2", Value: Intrinsic::nvvm_fmax_ftz_nan_bf16x2)
1462 .Case(S: "ftz.nan.xorsign.abs.bf16",
1463 Value: Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_bf16)
1464 .Case(S: "ftz.nan.xorsign.abs.bf16x2",
1465 Value: Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_bf16x2)
1466 .Case(S: "ftz.xorsign.abs.bf16", Value: Intrinsic::nvvm_fmax_ftz_xorsign_abs_bf16)
1467 .Case(S: "ftz.xorsign.abs.bf16x2",
1468 Value: Intrinsic::nvvm_fmax_ftz_xorsign_abs_bf16x2)
1469 .Case(S: "nan.bf16", Value: Intrinsic::nvvm_fmax_nan_bf16)
1470 .Case(S: "nan.bf16x2", Value: Intrinsic::nvvm_fmax_nan_bf16x2)
1471 .Case(S: "nan.xorsign.abs.bf16", Value: Intrinsic::nvvm_fmax_nan_xorsign_abs_bf16)
1472 .Case(S: "nan.xorsign.abs.bf16x2",
1473 Value: Intrinsic::nvvm_fmax_nan_xorsign_abs_bf16x2)
1474 .Case(S: "xorsign.abs.bf16", Value: Intrinsic::nvvm_fmax_xorsign_abs_bf16)
1475 .Case(S: "xorsign.abs.bf16x2", Value: Intrinsic::nvvm_fmax_xorsign_abs_bf16x2)
1476 .Default(Value: Intrinsic::not_intrinsic);
1477
1478 if (Name.consume_front(Prefix: "fmin."))
1479 return StringSwitch<Intrinsic::ID>(Name)
1480 .Case(S: "bf16", Value: Intrinsic::nvvm_fmin_bf16)
1481 .Case(S: "bf16x2", Value: Intrinsic::nvvm_fmin_bf16x2)
1482 .Case(S: "ftz.bf16", Value: Intrinsic::nvvm_fmin_ftz_bf16)
1483 .Case(S: "ftz.bf16x2", Value: Intrinsic::nvvm_fmin_ftz_bf16x2)
1484 .Case(S: "ftz.nan.bf16", Value: Intrinsic::nvvm_fmin_ftz_nan_bf16)
1485 .Case(S: "ftz.nan.bf16x2", Value: Intrinsic::nvvm_fmin_ftz_nan_bf16x2)
1486 .Case(S: "ftz.nan.xorsign.abs.bf16",
1487 Value: Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_bf16)
1488 .Case(S: "ftz.nan.xorsign.abs.bf16x2",
1489 Value: Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_bf16x2)
1490 .Case(S: "ftz.xorsign.abs.bf16", Value: Intrinsic::nvvm_fmin_ftz_xorsign_abs_bf16)
1491 .Case(S: "ftz.xorsign.abs.bf16x2",
1492 Value: Intrinsic::nvvm_fmin_ftz_xorsign_abs_bf16x2)
1493 .Case(S: "nan.bf16", Value: Intrinsic::nvvm_fmin_nan_bf16)
1494 .Case(S: "nan.bf16x2", Value: Intrinsic::nvvm_fmin_nan_bf16x2)
1495 .Case(S: "nan.xorsign.abs.bf16", Value: Intrinsic::nvvm_fmin_nan_xorsign_abs_bf16)
1496 .Case(S: "nan.xorsign.abs.bf16x2",
1497 Value: Intrinsic::nvvm_fmin_nan_xorsign_abs_bf16x2)
1498 .Case(S: "xorsign.abs.bf16", Value: Intrinsic::nvvm_fmin_xorsign_abs_bf16)
1499 .Case(S: "xorsign.abs.bf16x2", Value: Intrinsic::nvvm_fmin_xorsign_abs_bf16x2)
1500 .Default(Value: Intrinsic::not_intrinsic);
1501
1502 if (Name.consume_front(Prefix: "neg."))
1503 return StringSwitch<Intrinsic::ID>(Name)
1504 .Case(S: "bf16", Value: Intrinsic::nvvm_neg_bf16)
1505 .Case(S: "bf16x2", Value: Intrinsic::nvvm_neg_bf16x2)
1506 .Default(Value: Intrinsic::not_intrinsic);
1507
1508 return Intrinsic::not_intrinsic;
1509}
1510
1511static bool isLegacyNVPTXBF16IntSignature(Function *F, Intrinsic::ID IID) {
1512 FunctionType *NewFnTy = Intrinsic::getType(Context&: F->getContext(), id: IID);
1513 FunctionType *OldFnTy = F->getFunctionType();
1514 auto IsOldBF16StorageTy = [](Type *OldTy, Type *NewTy) {
1515 return OldTy->getScalarType()->isIntegerTy() &&
1516 OldTy->getPrimitiveSizeInBits() == NewTy->getPrimitiveSizeInBits();
1517 };
1518
1519 if (!IsOldBF16StorageTy(OldFnTy->getReturnType(), NewFnTy->getReturnType()))
1520 return false;
1521
1522 if (OldFnTy->getNumParams() != NewFnTy->getNumParams())
1523 return false;
1524
1525 for (unsigned I = 0, E = OldFnTy->getNumParams(); I != E; ++I)
1526 if (!IsOldBF16StorageTy(OldFnTy->getParamType(i: I), NewFnTy->getParamType(i: I)))
1527 return false;
1528
1529 return true;
1530}
1531
1532// Overloaded fadd/fmul intrinsic IDs, indexed by [`.ftz`][`.sat`].
1533static constexpr Intrinsic::ID NVVMFAddIIDs[2][2] = {
1534 {Intrinsic::nvvm_fadd, Intrinsic::nvvm_fadd_sat},
1535 {Intrinsic::nvvm_fadd_ftz, Intrinsic::nvvm_fadd_ftz_sat}};
1536static constexpr Intrinsic::ID NVVMFMulIIDs[2][2] = {
1537 {Intrinsic::nvvm_fmul, Intrinsic::nvvm_fmul_sat},
1538 {Intrinsic::nvvm_fmul_ftz, Intrinsic::nvvm_fmul_ftz_sat}};
1539
1540static std::optional<std::pair<Intrinsic::ID, RoundingMode>>
1541getNVVMFPArithUpgrade(StringRef Name, const Intrinsic::ID (&IIDs)[2][2]) {
1542 auto [Modifiers, Type] = Name.rsplit(Separator: '.');
1543 if (!is_contained(Set: {"f", "d", "f16", "v2f16"}, Element: Type))
1544 return std::nullopt;
1545
1546 std::optional<llvm::RoundingMode> RoundingMode =
1547 StringSwitch<std::optional<llvm::RoundingMode>>(Modifiers.take_front(N: 2))
1548 .Case(S: "rn", Value: llvm::RoundingMode::NearestTiesToEven)
1549 .Case(S: "rz", Value: llvm::RoundingMode::TowardZero)
1550 .Case(S: "rm", Value: llvm::RoundingMode::TowardNegative)
1551 .Case(S: "rp", Value: llvm::RoundingMode::TowardPositive)
1552 .Default(Value: std::nullopt);
1553 if (!RoundingMode)
1554 return std::nullopt;
1555
1556 StringRef Rest = Modifiers.drop_front(N: 2);
1557 const bool IsFTZ = Rest.consume_front(Prefix: ".ftz");
1558 const bool IsSat = Rest.consume_front(Prefix: ".sat");
1559 if (!Rest.empty())
1560 return std::nullopt;
1561
1562 return std::make_pair(x: IIDs[IsFTZ][IsSat], y&: *RoundingMode);
1563}
1564
1565static Intrinsic::ID shouldUpgradeNVPTXMBarrierInitIntrinsic(StringRef Name) {
1566 if (Name != "mbarrier.init" && Name != "mbarrier.init.shared")
1567 return Intrinsic::not_intrinsic;
1568
1569 return Intrinsic::nvvm_mbarrier_init;
1570}
1571
1572static bool consumeNVVMPtrAddrSpace(StringRef &Name) {
1573 return Name.consume_front(Prefix: "local") || Name.consume_front(Prefix: "shared") ||
1574 Name.consume_front(Prefix: "global") || Name.consume_front(Prefix: "constant") ||
1575 Name.consume_front(Prefix: "param");
1576}
1577
1578static unsigned getFunctionalOpcodeForVP(StringRef Name) {
1579 if (!Name.consume_front(Prefix: "vp."))
1580 return 0;
1581 return StringSwitch<unsigned>(Name)
1582 .StartsWith(S: "select", Value: Instruction::Select)
1583 .StartsWith(S: "add", Value: Instruction::Add)
1584 .StartsWith(S: "sub", Value: Instruction::Sub)
1585 .StartsWith(S: "mul", Value: Instruction::Mul)
1586 .StartsWith(S: "ashr", Value: Instruction::AShr)
1587 .StartsWith(S: "lshr", Value: Instruction::LShr)
1588 .StartsWith(S: "shl", Value: Instruction::Shl)
1589 .StartsWith(S: "or", Value: Instruction::Or)
1590 .StartsWith(S: "and", Value: Instruction::And)
1591 .StartsWith(S: "xor", Value: Instruction::Xor)
1592 .StartsWith(S: "fadd", Value: Instruction::FAdd)
1593 .StartsWith(S: "fsub", Value: Instruction::FSub)
1594 .StartsWith(S: "fmuladd", Value: 0)
1595 .StartsWith(S: "fmul", Value: Instruction::FMul)
1596 .StartsWith(S: "fdiv", Value: Instruction::FDiv)
1597 .StartsWith(S: "frem", Value: Instruction::FRem)
1598 .StartsWith(S: "fneg", Value: Instruction::FNeg)
1599 .StartsWith(S: "trunc", Value: Instruction::Trunc)
1600 .StartsWith(S: "zext", Value: Instruction::ZExt)
1601 .StartsWith(S: "sext", Value: Instruction::SExt)
1602 .StartsWith(S: "fptrunc", Value: Instruction::FPTrunc)
1603 .StartsWith(S: "fpext", Value: Instruction::FPExt)
1604 .StartsWith(S: "fptoui", Value: Instruction::FPToUI)
1605 .StartsWith(S: "fptosi", Value: Instruction::FPToSI)
1606 .StartsWith(S: "uitofp", Value: Instruction::UIToFP)
1607 .StartsWith(S: "sitofp", Value: Instruction::SIToFP)
1608 .StartsWith(S: "ptrtoint", Value: Instruction::PtrToInt)
1609 .StartsWith(S: "inttoptr", Value: Instruction::IntToPtr)
1610 .StartsWith(S: "icmp", Value: Instruction::ICmp)
1611 .StartsWith(S: "fcmp", Value: Instruction::FCmp)
1612 .Default(Value: 0);
1613}
1614
1615static Intrinsic::ID getFunctionalIntrinsicIDForVP(StringRef Name) {
1616 if (!Name.consume_front(Prefix: "vp."))
1617 return 0;
1618 return StringSwitch<Intrinsic::ID>(Name)
1619 .StartsWith(S: "abs", Value: Intrinsic::abs)
1620 .StartsWith(S: "smax", Value: Intrinsic::smax)
1621 .StartsWith(S: "smin", Value: Intrinsic::smin)
1622 .StartsWith(S: "umax", Value: Intrinsic::umax)
1623 .StartsWith(S: "umin", Value: Intrinsic::umin)
1624 .StartsWith(S: "copysign", Value: Intrinsic::copysign)
1625 .StartsWith(S: "minnum", Value: Intrinsic::minnum)
1626 .StartsWith(S: "maxnum", Value: Intrinsic::maxnum)
1627 .StartsWith(S: "minimum", Value: Intrinsic::minimum)
1628 .StartsWith(S: "maximum", Value: Intrinsic::maximum)
1629 .StartsWith(S: "fabs", Value: Intrinsic::fabs)
1630 .StartsWith(S: "sqrt", Value: Intrinsic::sqrt)
1631 .StartsWith(S: "fma", Value: Intrinsic::fma)
1632 .StartsWith(S: "fmuladd", Value: Intrinsic::fmuladd)
1633 .StartsWith(S: "ceil", Value: Intrinsic::ceil)
1634 .StartsWith(S: "floor", Value: Intrinsic::floor)
1635 .StartsWith(S: "rint", Value: Intrinsic::rint)
1636 .StartsWith(S: "nearbyint", Value: Intrinsic::nearbyint)
1637 .StartsWith(S: "roundeven", Value: Intrinsic::roundeven)
1638 .StartsWith(S: "roundtozero", Value: Intrinsic::trunc)
1639 .StartsWith(S: "round", Value: Intrinsic::round)
1640 .StartsWith(S: "lrint", Value: Intrinsic::lrint)
1641 .StartsWith(S: "llrint", Value: Intrinsic::llrint)
1642 .StartsWith(S: "bitreverse", Value: Intrinsic::bitreverse)
1643 .StartsWith(S: "bswap", Value: Intrinsic::bswap)
1644 .StartsWith(S: "ctpop", Value: Intrinsic::ctpop)
1645 .StartsWith(S: "ctlz", Value: Intrinsic::ctlz)
1646 .StartsWith(S: "cttz.elts", Value: 0)
1647 .StartsWith(S: "cttz", Value: Intrinsic::cttz)
1648 .StartsWith(S: "sadd.sat", Value: Intrinsic::sadd_sat)
1649 .StartsWith(S: "uadd.sat", Value: Intrinsic::uadd_sat)
1650 .StartsWith(S: "ssub.sat", Value: Intrinsic::ssub_sat)
1651 .StartsWith(S: "usub.sat", Value: Intrinsic::usub_sat)
1652 .StartsWith(S: "fshl", Value: Intrinsic::fshl)
1653 .StartsWith(S: "fshr", Value: Intrinsic::fshr)
1654 .StartsWith(S: "is.fpclass", Value: Intrinsic::is_fpclass)
1655 .Default(Value: 0);
1656}
1657
1658static bool shouldUpgradeVPIntrinsic(StringRef Name) {
1659 return getFunctionalOpcodeForVP(Name) || getFunctionalIntrinsicIDForVP(Name);
1660}
1661
1662static bool convertIntrinsicValidType(StringRef Name,
1663 const FunctionType *FuncTy) {
1664 Type *HalfTy = Type::getHalfTy(C&: FuncTy->getContext());
1665 if (Name.starts_with(Prefix: "to.fp16")) {
1666 return CastInst::castIsValid(op: Instruction::FPTrunc, SrcTy: FuncTy->getParamType(i: 0),
1667 DstTy: HalfTy) &&
1668 CastInst::castIsValid(op: Instruction::BitCast, SrcTy: HalfTy,
1669 DstTy: FuncTy->getReturnType());
1670 }
1671
1672 if (Name.starts_with(Prefix: "from.fp16")) {
1673 return CastInst::castIsValid(op: Instruction::BitCast, SrcTy: FuncTy->getParamType(i: 0),
1674 DstTy: HalfTy) &&
1675 CastInst::castIsValid(op: Instruction::FPExt, SrcTy: HalfTy,
1676 DstTy: FuncTy->getReturnType());
1677 }
1678
1679 return false;
1680}
1681
1682static unsigned
1683getFullArgCountForDefaultArgUpgrade(Function *F, Intrinsic::ID IID,
1684 SmallVectorImpl<Type *> &OverloadTys) {
1685 auto [FirstDefault, Defaults] = Intrinsic::getAllDefaultArgValues(IID);
1686 if (Defaults.empty())
1687 return 0;
1688
1689 unsigned FullArgCount = FirstDefault + Defaults.size();
1690
1691 // Only trailing default arguments can be missing.
1692 if (F->arg_size() < FirstDefault || F->arg_size() >= FullArgCount)
1693 return 0;
1694
1695 unsigned NumMissingTrailingParams = FullArgCount - F->arg_size();
1696 if (!Intrinsic::isSignatureValid(ID: IID, FT: F->getFunctionType(), OverloadTys,
1697 NumMissingTrailingParams))
1698 return 0;
1699
1700 return FullArgCount;
1701}
1702
1703static bool upgradeIntrinsicWithDefaultArgs(Function *F, Function *&NewFn) {
1704 Intrinsic::ID IID = F->getIntrinsicID();
1705 SmallVector<Type *, 4> OverloadTys;
1706
1707 unsigned FullArgCount =
1708 getFullArgCountForDefaultArgUpgrade(F, IID, OverloadTys);
1709 if (FullArgCount == 0)
1710 return false;
1711
1712 rename(GV: F);
1713 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID, OverloadTys);
1714 assert(NewFn->arg_size() == FullArgCount &&
1715 "total number of default args does not match intrinsic signature");
1716 return true;
1717}
1718
1719static bool upgradeIntrinsicFunction1(Function *F, Function *&NewFn,
1720 bool CanUpgradeDebugIntrinsicsToRecords) {
1721 assert(F && "Illegal to upgrade a non-existent Function.");
1722
1723 StringRef Name = F->getName();
1724
1725 // Quickly eliminate it, if it's not a candidate.
1726 if (!Name.consume_front(Prefix: "llvm.") || Name.empty())
1727 return false;
1728
1729 switch (Name[0]) {
1730 default: break;
1731 case 'a': {
1732 bool IsArm = Name.consume_front(Prefix: "arm.");
1733 if (IsArm || Name.consume_front(Prefix: "aarch64.")) {
1734 if (upgradeArmOrAarch64IntrinsicFunction(IsArm, F, Name, NewFn))
1735 return true;
1736 break;
1737 }
1738
1739 if (Name.consume_front(Prefix: "amdgcn.")) {
1740 if (Name == "alignbit") {
1741 // Target specific intrinsic became redundant
1742 NewFn = Intrinsic::getOrInsertDeclaration(
1743 M: F->getParent(), id: Intrinsic::fshr, OverloadTys: {F->getReturnType()});
1744 return true;
1745 }
1746
1747 if (Name.consume_front(Prefix: "atomic.")) {
1748 if (Name.starts_with(Prefix: "inc") || Name.starts_with(Prefix: "dec") ||
1749 Name.starts_with(Prefix: "cond.sub") || Name.starts_with(Prefix: "csub")) {
1750 // These were replaced with atomicrmw uinc_wrap, udec_wrap, usub_cond
1751 // and usub_sat so there's no new declaration.
1752 NewFn = nullptr;
1753 return true;
1754 }
1755 break; // No other 'amdgcn.atomic.*'
1756 }
1757
1758 if (Name.starts_with(Prefix: "addrspacecast.nonnull")) {
1759 // Replaced with an addrspacecast instruction carrying the nonnull flag,
1760 // so there's no new declaration.
1761 NewFn = nullptr;
1762 return true;
1763 }
1764
1765 switch (F->getIntrinsicID()) {
1766 default:
1767 break;
1768 // Legacy wmma iu intrinsics without the optional clamp operand.
1769 case Intrinsic::amdgcn_wmma_i32_16x16x64_iu8:
1770 if (F->arg_size() == 7) {
1771 NewFn = nullptr;
1772 return true;
1773 }
1774 break;
1775 case Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8:
1776 case Intrinsic::amdgcn_wmma_f32_16x16x4_f32:
1777 case Intrinsic::amdgcn_wmma_f32_16x16x32_bf16:
1778 case Intrinsic::amdgcn_wmma_f32_16x16x32_f16:
1779 case Intrinsic::amdgcn_wmma_f16_16x16x32_f16:
1780 case Intrinsic::amdgcn_wmma_bf16_16x16x32_bf16:
1781 case Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16:
1782 if (F->arg_size() == 8) {
1783 NewFn = nullptr;
1784 return true;
1785 }
1786 break;
1787 }
1788
1789 if (Name.consume_front(Prefix: "ds.") || Name.consume_front(Prefix: "global.atomic.") ||
1790 Name.consume_front(Prefix: "flat.atomic.")) {
1791 if (Name.starts_with(Prefix: "fadd") ||
1792 // FIXME: We should also remove fmin.num and fmax.num intrinsics.
1793 (Name.starts_with(Prefix: "fmin") && !Name.starts_with(Prefix: "fmin.num")) ||
1794 (Name.starts_with(Prefix: "fmax") && !Name.starts_with(Prefix: "fmax.num"))) {
1795 // Replaced with atomicrmw fadd/fmin/fmax, so there's no new
1796 // declaration.
1797 NewFn = nullptr;
1798 return true;
1799 }
1800 }
1801
1802 if (Name.starts_with(Prefix: "fcmp.") || Name.starts_with(Prefix: "icmp.")) {
1803 NewFn = nullptr;
1804 return true;
1805 }
1806
1807 if (Name.starts_with(Prefix: "ldexp.")) {
1808 // Target specific intrinsic became redundant
1809 NewFn = Intrinsic::getOrInsertDeclaration(
1810 M: F->getParent(), id: Intrinsic::ldexp,
1811 OverloadTys: {F->getReturnType(), F->getArg(i: 1)->getType()});
1812 return true;
1813 }
1814 break; // No other 'amdgcn.*'
1815 }
1816
1817 break;
1818 }
1819 case 'c': {
1820 if (F->arg_size() == 1) {
1821 if (Name.consume_front(Prefix: "convert.")) {
1822 if (convertIntrinsicValidType(Name, FuncTy: F->getFunctionType())) {
1823 NewFn = nullptr;
1824 return true;
1825 }
1826 }
1827
1828 Intrinsic::ID ID = StringSwitch<Intrinsic::ID>(Name)
1829 .StartsWith(S: "ctlz.", Value: Intrinsic::ctlz)
1830 .StartsWith(S: "cttz.", Value: Intrinsic::cttz)
1831 .Default(Value: Intrinsic::not_intrinsic);
1832 if (ID != Intrinsic::not_intrinsic) {
1833 rename(GV: F);
1834 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID,
1835 OverloadTys: F->arg_begin()->getType());
1836 return true;
1837 }
1838 }
1839
1840 Intrinsic::ID CoroEndID = Intrinsic::not_intrinsic;
1841 if (Name == "coro.end" &&
1842 (F->arg_size() == 2 || F->getReturnType()->isIntegerTy(BitWidth: 1)))
1843 CoroEndID = Intrinsic::coro_end;
1844 else if (Name == "coro.end.async" && F->getReturnType()->isIntegerTy(BitWidth: 1))
1845 CoroEndID = Intrinsic::coro_end_async;
1846
1847 if (CoroEndID != Intrinsic::not_intrinsic) {
1848 rename(GV: F);
1849 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: CoroEndID);
1850 return true;
1851 }
1852
1853 break;
1854 }
1855 case 'd':
1856 if (Name.consume_front(Prefix: "dbg.")) {
1857 // Mark debug intrinsics for upgrade to new debug format.
1858 if (CanUpgradeDebugIntrinsicsToRecords) {
1859 if (Name == "addr" || Name == "value" || Name == "assign" ||
1860 Name == "declare" || Name == "label") {
1861 // There's no function to replace these with.
1862 NewFn = nullptr;
1863 // But we do want these to get upgraded.
1864 return true;
1865 }
1866 }
1867 // Update llvm.dbg.addr intrinsics even in "new debug mode"; they'll get
1868 // converted to DbgVariableRecords later.
1869 if (Name == "addr" || (Name == "value" && F->arg_size() == 4)) {
1870 rename(GV: F);
1871 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(),
1872 id: Intrinsic::dbg_value);
1873 return true;
1874 }
1875 break; // No other 'dbg.*'.
1876 }
1877 break;
1878 case 'e':
1879 if (Name.consume_front(Prefix: "experimental.vector.")) {
1880 Intrinsic::ID ID =
1881 StringSwitch<Intrinsic::ID>(Name)
1882 // Skip over extract.last.active, otherwise it will be 'upgraded'
1883 // to a regular vector extract which is a different operation.
1884 .StartsWith(S: "extract.last.active.", Value: Intrinsic::not_intrinsic)
1885 .StartsWith(S: "extract.", Value: Intrinsic::vector_extract)
1886 .StartsWith(S: "insert.", Value: Intrinsic::vector_insert)
1887 .StartsWith(S: "reverse.", Value: Intrinsic::vector_reverse)
1888 .StartsWith(S: "interleave2.", Value: Intrinsic::vector_interleave2)
1889 .StartsWith(S: "deinterleave2.", Value: Intrinsic::vector_deinterleave2)
1890 .StartsWith(S: "partial.reduce.add",
1891 Value: Intrinsic::vector_partial_reduce_add)
1892 .Default(Value: Intrinsic::not_intrinsic);
1893 if (ID != Intrinsic::not_intrinsic) {
1894 const auto *FT = F->getFunctionType();
1895 SmallVector<Type *, 2> Tys;
1896 if (ID == Intrinsic::vector_extract ||
1897 ID == Intrinsic::vector_interleave2)
1898 // Extracting overloads the return type.
1899 Tys.push_back(Elt: FT->getReturnType());
1900 if (ID != Intrinsic::vector_interleave2)
1901 Tys.push_back(Elt: FT->getParamType(i: 0));
1902 if (ID == Intrinsic::vector_insert ||
1903 ID == Intrinsic::vector_partial_reduce_add)
1904 // Inserting overloads the inserted type.
1905 Tys.push_back(Elt: FT->getParamType(i: 1));
1906 rename(GV: F);
1907 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID, OverloadTys: Tys);
1908 return true;
1909 }
1910
1911 if (Name.consume_front(Prefix: "reduce.")) {
1912 SmallVector<StringRef, 2> Groups;
1913 static const Regex R("^([a-z]+)\\.[a-z][0-9]+");
1914 if (R.match(String: Name, Matches: &Groups))
1915 ID = StringSwitch<Intrinsic::ID>(Groups[1])
1916 .Case(S: "add", Value: Intrinsic::vector_reduce_add)
1917 .Case(S: "mul", Value: Intrinsic::vector_reduce_mul)
1918 .Case(S: "and", Value: Intrinsic::vector_reduce_and)
1919 .Case(S: "or", Value: Intrinsic::vector_reduce_or)
1920 .Case(S: "xor", Value: Intrinsic::vector_reduce_xor)
1921 .Case(S: "smax", Value: Intrinsic::vector_reduce_smax)
1922 .Case(S: "smin", Value: Intrinsic::vector_reduce_smin)
1923 .Case(S: "umax", Value: Intrinsic::vector_reduce_umax)
1924 .Case(S: "umin", Value: Intrinsic::vector_reduce_umin)
1925 .Case(S: "fmax", Value: Intrinsic::vector_reduce_fmax)
1926 .Case(S: "fmin", Value: Intrinsic::vector_reduce_fmin)
1927 .Default(Value: Intrinsic::not_intrinsic);
1928
1929 bool V2 = false;
1930 if (ID == Intrinsic::not_intrinsic) {
1931 static const Regex R2("^v2\\.([a-z]+)\\.[fi][0-9]+");
1932 Groups.clear();
1933 V2 = true;
1934 if (R2.match(String: Name, Matches: &Groups))
1935 ID = StringSwitch<Intrinsic::ID>(Groups[1])
1936 .Case(S: "fadd", Value: Intrinsic::vector_reduce_fadd)
1937 .Case(S: "fmul", Value: Intrinsic::vector_reduce_fmul)
1938 .Default(Value: Intrinsic::not_intrinsic);
1939 }
1940 if (ID != Intrinsic::not_intrinsic) {
1941 rename(GV: F);
1942 auto Args = F->getFunctionType()->params();
1943 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID,
1944 OverloadTys: {Args[V2 ? 1 : 0]});
1945 return true;
1946 }
1947 break; // No other 'expermental.vector.reduce.*'.
1948 }
1949
1950 if (Name.consume_front(Prefix: "splice"))
1951 return true;
1952 break; // No other 'experimental.vector.*'.
1953 }
1954 if (Name.consume_front(Prefix: "experimental.stepvector.")) {
1955 Intrinsic::ID ID = Intrinsic::stepvector;
1956 rename(GV: F);
1957 NewFn = Intrinsic::getOrInsertDeclaration(
1958 M: F->getParent(), id: ID, OverloadTys: F->getFunctionType()->getReturnType());
1959 return true;
1960 }
1961 break; // No other 'e*'.
1962 case 'f':
1963 if (Name.starts_with(Prefix: "flt.rounds")) {
1964 rename(GV: F);
1965 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(),
1966 id: Intrinsic::get_rounding);
1967 return true;
1968 }
1969 break;
1970 case 'i':
1971 if (Name.starts_with(Prefix: "invariant.group.barrier")) {
1972 // Rename invariant.group.barrier to launder.invariant.group
1973 auto Args = F->getFunctionType()->params();
1974 Type* ObjectPtr[1] = {Args[0]};
1975 rename(GV: F);
1976 NewFn = Intrinsic::getOrInsertDeclaration(
1977 M: F->getParent(), id: Intrinsic::launder_invariant_group, OverloadTys: ObjectPtr);
1978 return true;
1979 }
1980 break;
1981 case 'l': {
1982 bool IsLifetimeStart = Name.consume_front(Prefix: "lifetime.start");
1983 bool IsLifetimeEnd = !IsLifetimeStart && Name.consume_front(Prefix: "lifetime.end");
1984 if (IsLifetimeStart || IsLifetimeEnd) {
1985 if (F->arg_size() == 2) {
1986 Intrinsic::ID IID = IsLifetimeStart ? Intrinsic::lifetime_start
1987 : Intrinsic::lifetime_end;
1988 rename(GV: F);
1989 // Old 2 argument form of these intrinsics have [Size, Ptr] as
1990 // arguments. Use the Ptr argument to create new declaration.
1991 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID,
1992 OverloadTys: F->getArg(i: 1)->getType());
1993 return true;
1994 } else if (F->arg_size() == 1 && Name == ".i64") {
1995 // Matches @llvm.lifetime.{start/end}.i64 which used to be created by
1996 // Autoupgrade prior to
1997 // https://github.com/llvm/llvm-project/pull/204601. This is an invalid
1998 // intrinsic with no expected calls. To allow auto-upgrade process to
1999 // delete such invalid intrinsic declaration, set NewFn = nullptr
2000 // and return true here. If there are actual calls to this intrinsic
2001 // (which is not expected), they will be deleted in
2002 // UpgradeIntrinsicCall.
2003 NewFn = nullptr;
2004 return true;
2005 }
2006 }
2007 break;
2008 }
2009 case 'm': {
2010 // Updating the memory intrinsics (memcpy/memmove/memset) that have an
2011 // alignment parameter to embedding the alignment as an attribute of
2012 // the pointer args.
2013 if (unsigned ID = StringSwitch<unsigned>(Name)
2014 .StartsWith(S: "memcpy.", Value: Intrinsic::memcpy)
2015 .StartsWith(S: "memmove.", Value: Intrinsic::memmove)
2016 .Default(Value: 0)) {
2017 if (F->arg_size() == 5) {
2018 rename(GV: F);
2019 // Get the types of dest, src, and len
2020 ArrayRef<Type *> ParamTypes =
2021 F->getFunctionType()->params().slice(N: 0, M: 3);
2022 NewFn =
2023 Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID, OverloadTys: ParamTypes);
2024 return true;
2025 }
2026 }
2027 if (Name.starts_with(Prefix: "memset.") && F->arg_size() == 5) {
2028 rename(GV: F);
2029 // Get the types of dest, and len
2030 const auto *FT = F->getFunctionType();
2031 Type *ParamTypes[2] = {
2032 FT->getParamType(i: 0), // Dest
2033 FT->getParamType(i: 2) // len
2034 };
2035 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(),
2036 id: Intrinsic::memset, OverloadTys: ParamTypes);
2037 return true;
2038 }
2039
2040 unsigned MaskedID =
2041 StringSwitch<unsigned>(Name)
2042 .StartsWith(S: "masked.load", Value: Intrinsic::masked_load)
2043 .StartsWith(S: "masked.gather", Value: Intrinsic::masked_gather)
2044 .StartsWith(S: "masked.store", Value: Intrinsic::masked_store)
2045 .StartsWith(S: "masked.scatter", Value: Intrinsic::masked_scatter)
2046 .Default(Value: 0);
2047 if (MaskedID && F->arg_size() == 4) {
2048 rename(GV: F);
2049 if (MaskedID == Intrinsic::masked_load ||
2050 MaskedID == Intrinsic::masked_gather) {
2051 NewFn = Intrinsic::getOrInsertDeclaration(
2052 M: F->getParent(), id: MaskedID,
2053 OverloadTys: {F->getReturnType(), F->getArg(i: 0)->getType()});
2054 return true;
2055 }
2056 NewFn = Intrinsic::getOrInsertDeclaration(
2057 M: F->getParent(), id: MaskedID,
2058 OverloadTys: {F->getArg(i: 0)->getType(), F->getArg(i: 1)->getType()});
2059 return true;
2060 }
2061 break;
2062 }
2063 case 'n': {
2064 if (Name.consume_front(Prefix: "nvvm.")) {
2065 // Check for nvvm intrinsics corresponding exactly to an LLVM intrinsic.
2066 if (F->arg_size() == 1) {
2067 Intrinsic::ID IID =
2068 StringSwitch<Intrinsic::ID>(Name)
2069 .Cases(CaseStrings: {"brev32", "brev64"}, Value: Intrinsic::bitreverse)
2070 .Case(S: "clz.i", Value: Intrinsic::ctlz)
2071 .Case(S: "popc.i", Value: Intrinsic::ctpop)
2072 .Default(Value: Intrinsic::not_intrinsic);
2073 if (IID != Intrinsic::not_intrinsic) {
2074 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID,
2075 OverloadTys: {F->getReturnType()});
2076 return true;
2077 }
2078 } else if (F->arg_size() == 2) {
2079 Intrinsic::ID IID =
2080 StringSwitch<Intrinsic::ID>(Name)
2081 .Cases(CaseStrings: {"max.s", "max.i", "max.ll"}, Value: Intrinsic::smax)
2082 .Cases(CaseStrings: {"min.s", "min.i", "min.ll"}, Value: Intrinsic::smin)
2083 .Cases(CaseStrings: {"max.us", "max.ui", "max.ull"}, Value: Intrinsic::umax)
2084 .Cases(CaseStrings: {"min.us", "min.ui", "min.ull"}, Value: Intrinsic::umin)
2085 .Cases(CaseStrings: {"mulhi.s", "mulhi.i", "mulhi.ll"}, Value: Intrinsic::smulh)
2086 .Cases(CaseStrings: {"mulhi.us", "mulhi.ui", "mulhi.ull"}, Value: Intrinsic::umulh)
2087 .Default(Value: Intrinsic::not_intrinsic);
2088 if (IID != Intrinsic::not_intrinsic) {
2089 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID,
2090 OverloadTys: {F->getReturnType()});
2091 return true;
2092 }
2093 }
2094
2095 // Check for nvvm intrinsics that need a return type adjustment.
2096 {
2097 Intrinsic::ID IID = shouldUpgradeNVPTXBF16Intrinsic(Name);
2098 if (IID != Intrinsic::not_intrinsic &&
2099 isLegacyNVPTXBF16IntSignature(F, IID)) {
2100 NewFn = nullptr;
2101 return true;
2102 }
2103 }
2104
2105 // Upgrade Distributed Shared Memory Intrinsics
2106 Intrinsic::ID IID = shouldUpgradeNVPTXSharedClusterIntrinsic(F, Name);
2107 if (IID != Intrinsic::not_intrinsic) {
2108 rename(GV: F);
2109 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
2110 return true;
2111 }
2112
2113 // Upgrade TMA reduction intrinsics
2114 // llvm.nvvm.cp.async.bulk.tensor.reduce.<red_op>* =>
2115 // llvm.nvvm.cp.async.bulk.tensor.reduce.<shape>*
2116 IID = shouldUpgradeNVPTXTMAReductionIntrinsics(Name);
2117 if (IID != Intrinsic::not_intrinsic) {
2118 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
2119 return true;
2120 }
2121
2122 // Upgrade tcgen05.commit shared variants to anyptr intrinsics.
2123 IID = shouldUpgradeNVPTXTcgen05CommitSharedIntrinsic(F, Name);
2124 if (IID != Intrinsic::not_intrinsic) {
2125 rename(GV: F);
2126 NewFn = Intrinsic::getOrInsertDeclaration(
2127 M: F->getParent(), IID, RetTy: F->getReturnType(),
2128 ArgTys: F->getFunctionType()->params());
2129 return true;
2130 }
2131
2132 // Upgrade tcgen05.alloc/dealloc with the is_exclusive argument and
2133 // tcgen05.alloc shared variants to anyptr intrinsics.
2134 IID = shouldUpgradeNVPTXTcgen05AllocDeallocIntrinsic(F, Name);
2135 if (IID != Intrinsic::not_intrinsic) {
2136 rename(GV: F);
2137 if (Intrinsic::isOverloaded(id: IID))
2138 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID,
2139 OverloadTys: {F->getArg(i: 0)->getType()});
2140 else
2141 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
2142 return true;
2143 }
2144
2145 // Upgrade TMA copy G2S CTA intrinsics.
2146 IID = shouldUpgradeNVPTXTMAG2SCTAIntrinsics(F, Name);
2147 if (IID != Intrinsic::not_intrinsic) {
2148 rename(GV: F);
2149 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
2150 return true;
2151 }
2152
2153 // Upgrade TMA copy G2S (cluster) intrinsics.
2154 SmallVector<Type *, 1> OvlTys;
2155 IID = shouldUpgradeNVPTXTMAG2SIntrinsics(F, Name, OvlTys);
2156 if (IID != Intrinsic::not_intrinsic) {
2157 rename(GV: F);
2158 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID, OverloadTys: OvlTys);
2159 return true;
2160 }
2161
2162 // Upgrade the legacy cp.async.bulk.global.to.shared.cluster signature
2163 // (multicast-mask overloading + trailing flag_valid_pattern).
2164 SmallVector<Type *, 1> BulkG2SOvlTys;
2165 IID = shouldUpgradeNVPTXBulkG2SClusterIntrinsic(F, Name, OvlTys&: BulkG2SOvlTys);
2166 if (IID != Intrinsic::not_intrinsic) {
2167 rename(GV: F);
2168 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID,
2169 OverloadTys: BulkG2SOvlTys);
2170 return true;
2171 }
2172
2173 // Upgrade the legacy cp.async.bulk.global.to.shared.cta signature
2174 // (no ignore_bytes_left/right + trailing flag_valid_pattern).
2175 IID = shouldUpgradeNVPTXBulkG2SCTAIntrinsic(F, Name);
2176 if (IID != Intrinsic::not_intrinsic) {
2177 rename(GV: F);
2178 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
2179 return true;
2180 }
2181
2182 // Upgrade mbarrier.init intrinsics missing the layout operand.
2183 IID = shouldUpgradeNVPTXMBarrierInitIntrinsic(Name);
2184 if (IID != Intrinsic::not_intrinsic) {
2185 rename(GV: F);
2186 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID,
2187 OverloadTys: F->getArg(i: 0)->getType());
2188 return true;
2189 }
2190
2191 // The following nvvm intrinsics correspond exactly to an LLVM idiom, but
2192 // not to an intrinsic alone. We expand them in UpgradeIntrinsicCall.
2193 //
2194 // TODO: We could add lohi.i2d.
2195 bool Expand = false;
2196 if (Name.consume_front(Prefix: "abs."))
2197 // nvvm.abs.{i,ii}
2198 Expand =
2199 Name == "i" || Name == "ll" || Name == "bf16" || Name == "bf16x2";
2200 else if (Name.consume_front(Prefix: "fabs."))
2201 // nvvm.fabs.{f,ftz.f,d}
2202 Expand = Name == "f" || Name == "ftz.f" || Name == "d";
2203 else if (Name.consume_front(Prefix: "add."))
2204 // nvvm.add.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
2205 Expand = getNVVMFPArithUpgrade(Name, IIDs: NVVMFAddIIDs).has_value();
2206 else if (Name.consume_front(Prefix: "mul."))
2207 // nvvm.mul.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
2208 Expand = getNVVMFPArithUpgrade(Name, IIDs: NVVMFMulIIDs).has_value();
2209 else if (Name.consume_front(Prefix: "ex2.approx."))
2210 // nvvm.ex2.approx.{f,ftz.f,d,f16x2}
2211 Expand =
2212 Name == "f" || Name == "ftz.f" || Name == "d" || Name == "f16x2";
2213 else if (Name.consume_front(Prefix: "atomic.load."))
2214 // nvvm.atomic.load.add.{f32,f64}.p
2215 // nvvm.atomic.load.{inc,dec}.32.p
2216 Expand = StringSwitch<bool>(Name)
2217 .StartsWith(S: "add.f32.p", Value: true)
2218 .StartsWith(S: "add.f64.p", Value: true)
2219 .StartsWith(S: "inc.32.p", Value: true)
2220 .StartsWith(S: "dec.32.p", Value: true)
2221 .Default(Value: false);
2222 else if (Name.consume_front(Prefix: "atomic."))
2223 // nvvm.atomic.{add,exch,max,min,inc,dec,and,or,xor}.gen.{i,f}.{cta,sys}
2224 // nvvm.atomic.cas.gen.i.{cta,sys}
2225 Expand = StringSwitch<bool>(Name)
2226 .StartsWith(S: "add.gen.", Value: true)
2227 .StartsWith(S: "exch.gen.", Value: true)
2228 .StartsWith(S: "max.gen.", Value: true)
2229 .StartsWith(S: "min.gen.", Value: true)
2230 .StartsWith(S: "inc.gen.", Value: true)
2231 .StartsWith(S: "dec.gen.", Value: true)
2232 .StartsWith(S: "and.gen.", Value: true)
2233 .StartsWith(S: "or.gen.", Value: true)
2234 .StartsWith(S: "xor.gen.", Value: true)
2235 .StartsWith(S: "cas.gen.", Value: true)
2236 .Default(Value: false);
2237 else if (Name.consume_front(Prefix: "bitcast."))
2238 // nvvm.bitcast.{f2i,i2f,ll2d,d2ll}
2239 Expand =
2240 Name == "f2i" || Name == "i2f" || Name == "ll2d" || Name == "d2ll";
2241 else if (Name.consume_front(Prefix: "rotate."))
2242 // nvvm.rotate.{b32,b64,right.b64}
2243 Expand = Name == "b32" || Name == "b64" || Name == "right.b64";
2244 else if (Name.consume_front(Prefix: "ptr.gen.to."))
2245 // nvvm.ptr.gen.to.{local,shared,global,constant,param}
2246 Expand = consumeNVVMPtrAddrSpace(Name);
2247 else if (Name.consume_front(Prefix: "ptr."))
2248 // nvvm.ptr.{local,shared,global,constant,param}.to.gen
2249 Expand = consumeNVVMPtrAddrSpace(Name) && Name.starts_with(Prefix: ".to.gen");
2250 else if (Name.consume_front(Prefix: "ldg.global."))
2251 // nvvm.ldg.global.{i,p,f}
2252 Expand = (Name.starts_with(Prefix: "i.") || Name.starts_with(Prefix: "f.") ||
2253 Name.starts_with(Prefix: "p."));
2254 else
2255 Expand = StringSwitch<bool>(Name)
2256 .Case(S: "barrier0", Value: true)
2257 .Case(S: "barrier.n", Value: true)
2258 .Case(S: "barrier.sync.cnt", Value: true)
2259 .Case(S: "barrier.sync", Value: true)
2260 .Case(S: "barrier", Value: true)
2261 .Case(S: "bar.sync", Value: true)
2262 .Case(S: "barrier0.popc", Value: true)
2263 .Case(S: "barrier0.and", Value: true)
2264 .Case(S: "barrier0.or", Value: true)
2265 .Case(S: "clz.ll", Value: true)
2266 .Case(S: "popc.ll", Value: true)
2267 .Case(S: "h2f", Value: true)
2268 .Case(S: "swap.lo.hi.b64", Value: true)
2269 .Case(S: "tanh.approx.f32", Value: true)
2270 .Default(Value: false);
2271
2272 if (Expand) {
2273 NewFn = nullptr;
2274 return true;
2275 }
2276 break; // No other 'nvvm.*'.
2277 }
2278 break;
2279 }
2280 case 'o':
2281 if (Name.starts_with(Prefix: "objectsize.")) {
2282 Type *Tys[2] = { F->getReturnType(), F->arg_begin()->getType() };
2283 if (F->arg_size() == 2 || F->arg_size() == 3) {
2284 rename(GV: F);
2285 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(),
2286 id: Intrinsic::objectsize, OverloadTys: Tys);
2287 return true;
2288 }
2289 }
2290 break;
2291
2292 case 'p':
2293 if (Name.starts_with(Prefix: "ptr.annotation.") && F->arg_size() == 4) {
2294 rename(GV: F);
2295 NewFn = Intrinsic::getOrInsertDeclaration(
2296 M: F->getParent(), id: Intrinsic::ptr_annotation,
2297 OverloadTys: {F->arg_begin()->getType(), F->getArg(i: 1)->getType()});
2298 return true;
2299 }
2300 break;
2301
2302 case 'r': {
2303 if (Name.consume_front(Prefix: "riscv.")) {
2304 Intrinsic::ID ID;
2305 ID = StringSwitch<Intrinsic::ID>(Name)
2306 .Case(S: "aes32dsi", Value: Intrinsic::riscv_aes32dsi)
2307 .Case(S: "aes32dsmi", Value: Intrinsic::riscv_aes32dsmi)
2308 .Case(S: "aes32esi", Value: Intrinsic::riscv_aes32esi)
2309 .Case(S: "aes32esmi", Value: Intrinsic::riscv_aes32esmi)
2310 .Default(Value: Intrinsic::not_intrinsic);
2311 if (ID != Intrinsic::not_intrinsic) {
2312 if (!F->getFunctionType()->getParamType(i: 2)->isIntegerTy(BitWidth: 32)) {
2313 rename(GV: F);
2314 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
2315 return true;
2316 }
2317 break; // No other applicable upgrades.
2318 }
2319
2320 ID = StringSwitch<Intrinsic::ID>(Name)
2321 .StartsWith(S: "sm4ks", Value: Intrinsic::riscv_sm4ks)
2322 .StartsWith(S: "sm4ed", Value: Intrinsic::riscv_sm4ed)
2323 .Default(Value: Intrinsic::not_intrinsic);
2324 if (ID != Intrinsic::not_intrinsic) {
2325 if (!F->getFunctionType()->getParamType(i: 2)->isIntegerTy(BitWidth: 32) ||
2326 F->getFunctionType()->getReturnType()->isIntegerTy(BitWidth: 64)) {
2327 rename(GV: F);
2328 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
2329 return true;
2330 }
2331 break; // No other applicable upgrades.
2332 }
2333
2334 ID = StringSwitch<Intrinsic::ID>(Name)
2335 .StartsWith(S: "sha256sig0", Value: Intrinsic::riscv_sha256sig0)
2336 .StartsWith(S: "sha256sig1", Value: Intrinsic::riscv_sha256sig1)
2337 .StartsWith(S: "sha256sum0", Value: Intrinsic::riscv_sha256sum0)
2338 .StartsWith(S: "sha256sum1", Value: Intrinsic::riscv_sha256sum1)
2339 .StartsWith(S: "sm3p0", Value: Intrinsic::riscv_sm3p0)
2340 .StartsWith(S: "sm3p1", Value: Intrinsic::riscv_sm3p1)
2341 .Default(Value: Intrinsic::not_intrinsic);
2342 if (ID != Intrinsic::not_intrinsic) {
2343 if (F->getFunctionType()->getReturnType()->isIntegerTy(BitWidth: 64)) {
2344 rename(GV: F);
2345 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
2346 return true;
2347 }
2348 break; // No other applicable upgrades.
2349 }
2350
2351 // Replace llvm.riscv.clmul with llvm.clmul.
2352 if (Name == "clmul.i32" || Name == "clmul.i64") {
2353 NewFn = Intrinsic::getOrInsertDeclaration(
2354 M: F->getParent(), id: Intrinsic::clmul, OverloadTys: {F->getReturnType()});
2355 return true;
2356 }
2357
2358 break; // No other 'riscv.*' intrinsics
2359 }
2360 } break;
2361
2362 case 's':
2363 if (Name == "stackprotectorcheck") {
2364 NewFn = nullptr;
2365 return true;
2366 }
2367 if (Name.starts_with(Prefix: "strip.invariant.group")) {
2368 // For clang's usage it would be safe to just drop the
2369 // strip.invariant.group, but to be conservative replace with the
2370 // stronger launder.invariant.group instead.
2371 NewFn = Intrinsic::getOrInsertDeclaration(
2372 M: F->getParent(), id: Intrinsic::launder_invariant_group,
2373 OverloadTys: F->getReturnType());
2374 return true;
2375 }
2376 break;
2377
2378 case 't':
2379 if (Name == "thread.pointer") {
2380 NewFn = Intrinsic::getOrInsertDeclaration(
2381 M: F->getParent(), id: Intrinsic::thread_pointer, OverloadTys: F->getReturnType());
2382 return true;
2383 }
2384 break;
2385
2386 case 'v': {
2387 if (Name == "var.annotation" && F->arg_size() == 4) {
2388 rename(GV: F);
2389 NewFn = Intrinsic::getOrInsertDeclaration(
2390 M: F->getParent(), id: Intrinsic::var_annotation,
2391 OverloadTys: {{F->arg_begin()->getType(), F->getArg(i: 1)->getType()}});
2392 return true;
2393 }
2394 if (Name.consume_front(Prefix: "vector.splice")) {
2395 if (Name.starts_with(Prefix: ".left") || Name.starts_with(Prefix: ".right"))
2396 break;
2397 return true;
2398 }
2399 if (shouldUpgradeVPIntrinsic(Name))
2400 return true;
2401 break;
2402 }
2403
2404 case 'w':
2405 if (Name.consume_front(Prefix: "wasm.")) {
2406 Intrinsic::ID ID =
2407 StringSwitch<Intrinsic::ID>(Name)
2408 .StartsWith(S: "fma.", Value: Intrinsic::wasm_relaxed_madd)
2409 .StartsWith(S: "fms.", Value: Intrinsic::wasm_relaxed_nmadd)
2410 .StartsWith(S: "laneselect.", Value: Intrinsic::wasm_relaxed_laneselect)
2411 .Default(Value: Intrinsic::not_intrinsic);
2412 if (ID != Intrinsic::not_intrinsic) {
2413 rename(GV: F);
2414 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID,
2415 OverloadTys: F->getReturnType());
2416 return true;
2417 }
2418
2419 if (Name.consume_front(Prefix: "dot.i8x16.i7x16.")) {
2420 ID = StringSwitch<Intrinsic::ID>(Name)
2421 .Case(S: "signed", Value: Intrinsic::wasm_relaxed_dot_i8x16_i7x16_signed)
2422 .Case(S: "add.signed",
2423 Value: Intrinsic::wasm_relaxed_dot_i8x16_i7x16_add_signed)
2424 .Default(Value: Intrinsic::not_intrinsic);
2425 if (ID != Intrinsic::not_intrinsic) {
2426 rename(GV: F);
2427 NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: ID);
2428 return true;
2429 }
2430 break; // No other 'wasm.dot.i8x16.i7x16.*'.
2431 }
2432 break; // No other 'wasm.*'.
2433 }
2434 break;
2435
2436 case 'x':
2437 if (upgradeX86IntrinsicFunction(F, Name, NewFn))
2438 return true;
2439 }
2440
2441 auto *ST = dyn_cast<StructType>(Val: F->getReturnType());
2442 if (ST && (!ST->isLiteral() || ST->isPacked()) &&
2443 F->getIntrinsicID() != Intrinsic::not_intrinsic) {
2444 // Replace return type with literal non-packed struct. Only do this for
2445 // intrinsics declared to return a struct, not for intrinsics with
2446 // overloaded return type, in which case the exact struct type will be
2447 // mangled into the name.
2448 if (Intrinsic::hasStructReturnType(id: F->getIntrinsicID())) {
2449 FunctionType *FT = F->getFunctionType();
2450 auto *NewST = StructType::get(Context&: ST->getContext(), Elements: ST->elements());
2451 auto *NewFT = FunctionType::get(Result: NewST, Params: FT->params(), isVarArg: FT->isVarArg());
2452 std::string Name = F->getName().str();
2453 rename(GV: F);
2454 NewFn = Function::Create(Ty: NewFT, Linkage: F->getLinkage(), AddrSpace: F->getAddressSpace(),
2455 N: Name, M: F->getParent());
2456
2457 // The new function may also need remangling.
2458 if (auto Result = llvm::Intrinsic::remangleIntrinsicFunction(F: NewFn))
2459 NewFn = *Result;
2460 return true;
2461 }
2462 }
2463
2464 // Remangle our intrinsic since we upgrade the mangling
2465 auto Result = llvm::Intrinsic::remangleIntrinsicFunction(F);
2466 if (Result != std::nullopt) {
2467 NewFn = *Result;
2468 return true;
2469 }
2470
2471 if (upgradeIntrinsicWithDefaultArgs(F, NewFn))
2472 return true;
2473
2474 // This may not belong here. This function is effectively being overloaded
2475 // to both detect an intrinsic which needs upgrading, and to provide the
2476 // upgraded form of the intrinsic. We should perhaps have two separate
2477 // functions for this.
2478
2479 return false;
2480}
2481
2482bool llvm::UpgradeIntrinsicFunction(Function *F, Function *&NewFn,
2483 bool CanUpgradeDebugIntrinsicsToRecords) {
2484 NewFn = nullptr;
2485 bool Upgraded =
2486 upgradeIntrinsicFunction1(F, NewFn, CanUpgradeDebugIntrinsicsToRecords);
2487
2488 // Upgrade intrinsic attributes. This does not change the function.
2489 if (NewFn)
2490 F = NewFn;
2491 if (Intrinsic::ID id = F->getIntrinsicID()) {
2492 // Only do this if the intrinsic signature is valid.
2493 SmallVector<Type *> OverloadTys;
2494 if (Intrinsic::isSignatureValid(ID: id, FT: F->getFunctionType(), OverloadTys))
2495 F->setAttributes(
2496 Intrinsic::getAttributes(C&: F->getContext(), id, FT: F->getFunctionType()));
2497 }
2498 return Upgraded;
2499}
2500
2501GlobalVariable *llvm::UpgradeGlobalVariable(GlobalVariable *GV) {
2502 if (!(GV->hasName() && (GV->getName() == "llvm.global_ctors" ||
2503 GV->getName() == "llvm.global_dtors")) ||
2504 !GV->hasInitializer())
2505 return nullptr;
2506 ArrayType *ATy = dyn_cast<ArrayType>(Val: GV->getValueType());
2507 if (!ATy)
2508 return nullptr;
2509 StructType *STy = dyn_cast<StructType>(Val: ATy->getElementType());
2510 if (!STy || STy->getNumElements() != 2)
2511 return nullptr;
2512
2513 IRBuilder<> IRB(*GV->getParent());
2514 auto EltTy = StructType::get(elt1: STy->getElementType(N: 0), elts: STy->getElementType(N: 1),
2515 elts: IRB.getPtrTy());
2516 Constant *Init = GV->getInitializer();
2517 unsigned N = Init->getNumOperands();
2518 std::vector<Constant *> NewCtors(N);
2519 for (unsigned i = 0; i != N; ++i) {
2520 auto Ctor = cast<Constant>(Val: Init->getOperand(i));
2521 NewCtors[i] = ConstantStruct::get(T: EltTy, Vs: Ctor->getAggregateElement(Elt: 0u),
2522 Vs: Ctor->getAggregateElement(Elt: 1),
2523 Vs: ConstantPointerNull::get(T: IRB.getPtrTy()));
2524 }
2525 Constant *NewInit = ConstantArray::get(T: ArrayType::get(ElementType: EltTy, NumElements: N), V: NewCtors);
2526
2527 return new GlobalVariable(NewInit->getType(), false, GV->getLinkage(),
2528 NewInit, GV->getName());
2529}
2530
2531// Handles upgrading SSE2/AVX2/AVX512BW PSLLDQ intrinsics by converting them
2532// to byte shuffles.
2533static Value *upgradeX86PSLLDQIntrinsics(IRBuilder<> &Builder, Value *Op,
2534 unsigned Shift) {
2535 auto *ResultTy = cast<FixedVectorType>(Val: Op->getType());
2536 unsigned NumElts = ResultTy->getNumElements() * 8;
2537
2538 // Bitcast from a 64-bit element type to a byte element type.
2539 Type *VecTy = FixedVectorType::get(ElementType: Builder.getInt8Ty(), NumElts);
2540 Op = Builder.CreateBitCast(V: Op, DestTy: VecTy, Name: "cast");
2541
2542 // We'll be shuffling in zeroes.
2543 Value *Res = Constant::getNullValue(Ty: VecTy);
2544
2545 // If shift is less than 16, emit a shuffle to move the bytes. Otherwise,
2546 // we'll just return the zero vector.
2547 if (Shift < 16) {
2548 int Idxs[64];
2549 // 256/512-bit version is split into 2/4 16-byte lanes.
2550 for (unsigned l = 0; l != NumElts; l += 16)
2551 for (unsigned i = 0; i != 16; ++i) {
2552 unsigned Idx = NumElts + i - Shift;
2553 if (Idx < NumElts)
2554 Idx -= NumElts - 16; // end of lane, switch operand.
2555 Idxs[l + i] = Idx + l;
2556 }
2557
2558 Res = Builder.CreateShuffleVector(V1: Res, V2: Op, Mask: ArrayRef(Idxs, NumElts));
2559 }
2560
2561 // Bitcast back to a 64-bit element type.
2562 return Builder.CreateBitCast(V: Res, DestTy: ResultTy, Name: "cast");
2563}
2564
2565// Handles upgrading SSE2/AVX2/AVX512BW PSRLDQ intrinsics by converting them
2566// to byte shuffles.
2567static Value *upgradeX86PSRLDQIntrinsics(IRBuilder<> &Builder, Value *Op,
2568 unsigned Shift) {
2569 auto *ResultTy = cast<FixedVectorType>(Val: Op->getType());
2570 unsigned NumElts = ResultTy->getNumElements() * 8;
2571
2572 // Bitcast from a 64-bit element type to a byte element type.
2573 Type *VecTy = FixedVectorType::get(ElementType: Builder.getInt8Ty(), NumElts);
2574 Op = Builder.CreateBitCast(V: Op, DestTy: VecTy, Name: "cast");
2575
2576 // We'll be shuffling in zeroes.
2577 Value *Res = Constant::getNullValue(Ty: VecTy);
2578
2579 // If shift is less than 16, emit a shuffle to move the bytes. Otherwise,
2580 // we'll just return the zero vector.
2581 if (Shift < 16) {
2582 int Idxs[64];
2583 // 256/512-bit version is split into 2/4 16-byte lanes.
2584 for (unsigned l = 0; l != NumElts; l += 16)
2585 for (unsigned i = 0; i != 16; ++i) {
2586 unsigned Idx = i + Shift;
2587 if (Idx >= 16)
2588 Idx += NumElts - 16; // end of lane, switch operand.
2589 Idxs[l + i] = Idx + l;
2590 }
2591
2592 Res = Builder.CreateShuffleVector(V1: Op, V2: Res, Mask: ArrayRef(Idxs, NumElts));
2593 }
2594
2595 // Bitcast back to a 64-bit element type.
2596 return Builder.CreateBitCast(V: Res, DestTy: ResultTy, Name: "cast");
2597}
2598
2599static Value *getX86MaskVec(IRBuilder<> &Builder, Value *Mask,
2600 unsigned NumElts) {
2601 assert(isPowerOf2_32(NumElts) && "Expected power-of-2 mask elements");
2602 llvm::VectorType *MaskTy = FixedVectorType::get(
2603 ElementType: Builder.getInt1Ty(), NumElts: cast<IntegerType>(Val: Mask->getType())->getBitWidth());
2604 Mask = Builder.CreateBitCast(V: Mask, DestTy: MaskTy);
2605
2606 // If we have less than 8 elements (1, 2 or 4), then the starting mask was an
2607 // i8 and we need to extract down to the right number of elements.
2608 if (NumElts <= 4) {
2609 int Indices[4];
2610 for (unsigned i = 0; i != NumElts; ++i)
2611 Indices[i] = i;
2612 Mask = Builder.CreateShuffleVector(V1: Mask, V2: Mask, Mask: ArrayRef(Indices, NumElts),
2613 Name: "extract");
2614 }
2615
2616 return Mask;
2617}
2618
2619static Value *emitX86Select(IRBuilder<> &Builder, Value *Mask, Value *Op0,
2620 Value *Op1) {
2621 // If the mask is all ones just emit the first operation.
2622 if (const auto *C = dyn_cast<Constant>(Val: Mask))
2623 if (C->isAllOnesValue())
2624 return Op0;
2625
2626 Mask = getX86MaskVec(Builder, Mask,
2627 NumElts: cast<FixedVectorType>(Val: Op0->getType())->getNumElements());
2628 return Builder.CreateSelect(C: Mask, True: Op0, False: Op1);
2629}
2630
2631static Value *emitX86ScalarSelect(IRBuilder<> &Builder, Value *Mask, Value *Op0,
2632 Value *Op1) {
2633 // If the mask is all ones just emit the first operation.
2634 if (const auto *C = dyn_cast<Constant>(Val: Mask))
2635 if (C->isAllOnesValue())
2636 return Op0;
2637
2638 auto *MaskTy = FixedVectorType::get(ElementType: Builder.getInt1Ty(),
2639 NumElts: Mask->getType()->getIntegerBitWidth());
2640 Mask = Builder.CreateBitCast(V: Mask, DestTy: MaskTy);
2641 Mask = Builder.CreateExtractElement(Vec: Mask, Idx: (uint64_t)0);
2642 return Builder.CreateSelect(C: Mask, True: Op0, False: Op1);
2643}
2644
2645// Handle autoupgrade for masked PALIGNR and VALIGND/Q intrinsics.
2646// PALIGNR handles large immediates by shifting while VALIGN masks the immediate
2647// so we need to handle both cases. VALIGN also doesn't have 128-bit lanes.
2648static Value *upgradeX86ALIGNIntrinsics(IRBuilder<> &Builder, Value *Op0,
2649 Value *Op1, Value *Shift,
2650 Value *Passthru, Value *Mask,
2651 bool IsVALIGN) {
2652 unsigned ShiftVal = cast<llvm::ConstantInt>(Val: Shift)->getZExtValue();
2653
2654 unsigned NumElts = cast<FixedVectorType>(Val: Op0->getType())->getNumElements();
2655 assert((IsVALIGN || NumElts % 16 == 0) && "Illegal NumElts for PALIGNR!");
2656 assert((!IsVALIGN || NumElts <= 16) && "NumElts too large for VALIGN!");
2657 assert(isPowerOf2_32(NumElts) && "NumElts not a power of 2!");
2658
2659 // Mask the immediate for VALIGN.
2660 if (IsVALIGN)
2661 ShiftVal &= (NumElts - 1);
2662
2663 // If palignr is shifting the pair of vectors more than the size of two
2664 // lanes, emit zero.
2665 if (ShiftVal >= 32)
2666 return llvm::Constant::getNullValue(Ty: Op0->getType());
2667
2668 // If palignr is shifting the pair of input vectors more than one lane,
2669 // but less than two lanes, convert to shifting in zeroes.
2670 if (ShiftVal > 16) {
2671 ShiftVal -= 16;
2672 Op1 = Op0;
2673 Op0 = llvm::Constant::getNullValue(Ty: Op0->getType());
2674 }
2675
2676 int Indices[64];
2677 // 256-bit palignr operates on 128-bit lanes so we need to handle that
2678 for (unsigned l = 0; l < NumElts; l += 16) {
2679 for (unsigned i = 0; i != 16; ++i) {
2680 unsigned Idx = ShiftVal + i;
2681 if (!IsVALIGN && Idx >= 16) // Disable wrap for VALIGN.
2682 Idx += NumElts - 16; // End of lane, switch operand.
2683 Indices[l + i] = Idx + l;
2684 }
2685 }
2686
2687 Value *Align = Builder.CreateShuffleVector(
2688 V1: Op1, V2: Op0, Mask: ArrayRef(Indices, NumElts), Name: "palignr");
2689
2690 return emitX86Select(Builder, Mask, Op0: Align, Op1: Passthru);
2691}
2692
2693static Value *upgradeX86VPERMT2Intrinsics(IRBuilder<> &Builder, CallBase &CI,
2694 bool ZeroMask, bool IndexForm) {
2695 Type *Ty = CI.getType();
2696 unsigned VecWidth = Ty->getPrimitiveSizeInBits();
2697 unsigned EltWidth = Ty->getScalarSizeInBits();
2698 bool IsFloat = Ty->isFPOrFPVectorTy();
2699 Intrinsic::ID IID;
2700 if (VecWidth == 128 && EltWidth == 32 && IsFloat)
2701 IID = Intrinsic::x86_avx512_vpermi2var_ps_128;
2702 else if (VecWidth == 128 && EltWidth == 32 && !IsFloat)
2703 IID = Intrinsic::x86_avx512_vpermi2var_d_128;
2704 else if (VecWidth == 128 && EltWidth == 64 && IsFloat)
2705 IID = Intrinsic::x86_avx512_vpermi2var_pd_128;
2706 else if (VecWidth == 128 && EltWidth == 64 && !IsFloat)
2707 IID = Intrinsic::x86_avx512_vpermi2var_q_128;
2708 else if (VecWidth == 256 && EltWidth == 32 && IsFloat)
2709 IID = Intrinsic::x86_avx512_vpermi2var_ps_256;
2710 else if (VecWidth == 256 && EltWidth == 32 && !IsFloat)
2711 IID = Intrinsic::x86_avx512_vpermi2var_d_256;
2712 else if (VecWidth == 256 && EltWidth == 64 && IsFloat)
2713 IID = Intrinsic::x86_avx512_vpermi2var_pd_256;
2714 else if (VecWidth == 256 && EltWidth == 64 && !IsFloat)
2715 IID = Intrinsic::x86_avx512_vpermi2var_q_256;
2716 else if (VecWidth == 512 && EltWidth == 32 && IsFloat)
2717 IID = Intrinsic::x86_avx512_vpermi2var_ps_512;
2718 else if (VecWidth == 512 && EltWidth == 32 && !IsFloat)
2719 IID = Intrinsic::x86_avx512_vpermi2var_d_512;
2720 else if (VecWidth == 512 && EltWidth == 64 && IsFloat)
2721 IID = Intrinsic::x86_avx512_vpermi2var_pd_512;
2722 else if (VecWidth == 512 && EltWidth == 64 && !IsFloat)
2723 IID = Intrinsic::x86_avx512_vpermi2var_q_512;
2724 else if (VecWidth == 128 && EltWidth == 16)
2725 IID = Intrinsic::x86_avx512_vpermi2var_hi_128;
2726 else if (VecWidth == 256 && EltWidth == 16)
2727 IID = Intrinsic::x86_avx512_vpermi2var_hi_256;
2728 else if (VecWidth == 512 && EltWidth == 16)
2729 IID = Intrinsic::x86_avx512_vpermi2var_hi_512;
2730 else if (VecWidth == 128 && EltWidth == 8)
2731 IID = Intrinsic::x86_avx512_vpermi2var_qi_128;
2732 else if (VecWidth == 256 && EltWidth == 8)
2733 IID = Intrinsic::x86_avx512_vpermi2var_qi_256;
2734 else if (VecWidth == 512 && EltWidth == 8)
2735 IID = Intrinsic::x86_avx512_vpermi2var_qi_512;
2736 else
2737 llvm_unreachable("Unexpected intrinsic");
2738
2739 Value *Args[] = { CI.getArgOperand(i: 0) , CI.getArgOperand(i: 1),
2740 CI.getArgOperand(i: 2) };
2741
2742 // If this isn't index form we need to swap operand 0 and 1.
2743 if (!IndexForm)
2744 std::swap(a&: Args[0], b&: Args[1]);
2745
2746 Value *V = Builder.CreateIntrinsic(ID: IID, Args);
2747 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(Ty)
2748 : Builder.CreateBitCast(V: CI.getArgOperand(i: 1),
2749 DestTy: Ty);
2750 return emitX86Select(Builder, Mask: CI.getArgOperand(i: 3), Op0: V, Op1: PassThru);
2751}
2752
2753static Value *upgradeX86BinaryIntrinsics(IRBuilder<> &Builder, CallBase &CI,
2754 Intrinsic::ID IID) {
2755 Type *Ty = CI.getType();
2756 Value *Op0 = CI.getOperand(i_nocapture: 0);
2757 Value *Op1 = CI.getOperand(i_nocapture: 1);
2758 Value *Res = Builder.CreateIntrinsic(ID: IID, OverloadTypes: Ty, Args: {Op0, Op1});
2759
2760 if (CI.arg_size() == 4) { // For masked intrinsics.
2761 Value *VecSrc = CI.getOperand(i_nocapture: 2);
2762 Value *Mask = CI.getOperand(i_nocapture: 3);
2763 Res = emitX86Select(Builder, Mask, Op0: Res, Op1: VecSrc);
2764 }
2765 return Res;
2766}
2767
2768static Value *upgradeX86Rotate(IRBuilder<> &Builder, CallBase &CI,
2769 bool IsRotateRight) {
2770 Type *Ty = CI.getType();
2771 Value *Src = CI.getArgOperand(i: 0);
2772 Value *Amt = CI.getArgOperand(i: 1);
2773
2774 // Amount may be scalar immediate, in which case create a splat vector.
2775 // Funnel shifts amounts are treated as modulo and types are all power-of-2 so
2776 // we only care about the lowest log2 bits anyway.
2777 if (Amt->getType() != Ty) {
2778 unsigned NumElts = cast<FixedVectorType>(Val: Ty)->getNumElements();
2779 Amt = Builder.CreateIntCast(V: Amt, DestTy: Ty->getScalarType(), isSigned: false);
2780 Amt = Builder.CreateVectorSplat(NumElts, V: Amt);
2781 }
2782
2783 Intrinsic::ID IID = IsRotateRight ? Intrinsic::fshr : Intrinsic::fshl;
2784 Value *Res = Builder.CreateIntrinsic(ID: IID, OverloadTypes: Ty, Args: {Src, Src, Amt});
2785
2786 if (CI.arg_size() == 4) { // For masked intrinsics.
2787 Value *VecSrc = CI.getOperand(i_nocapture: 2);
2788 Value *Mask = CI.getOperand(i_nocapture: 3);
2789 Res = emitX86Select(Builder, Mask, Op0: Res, Op1: VecSrc);
2790 }
2791 return Res;
2792}
2793
2794static Value *upgradeX86vpcom(IRBuilder<> &Builder, CallBase &CI, unsigned Imm,
2795 bool IsSigned) {
2796 Type *Ty = CI.getType();
2797 Value *LHS = CI.getArgOperand(i: 0);
2798 Value *RHS = CI.getArgOperand(i: 1);
2799
2800 CmpInst::Predicate Pred;
2801 switch (Imm) {
2802 case 0x0:
2803 Pred = IsSigned ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT;
2804 break;
2805 case 0x1:
2806 Pred = IsSigned ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE;
2807 break;
2808 case 0x2:
2809 Pred = IsSigned ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT;
2810 break;
2811 case 0x3:
2812 Pred = IsSigned ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE;
2813 break;
2814 case 0x4:
2815 Pred = ICmpInst::ICMP_EQ;
2816 break;
2817 case 0x5:
2818 Pred = ICmpInst::ICMP_NE;
2819 break;
2820 case 0x6:
2821 return Constant::getNullValue(Ty); // FALSE
2822 case 0x7:
2823 return Constant::getAllOnesValue(Ty); // TRUE
2824 default:
2825 llvm_unreachable("Unknown XOP vpcom/vpcomu predicate");
2826 }
2827
2828 Value *Cmp = Builder.CreateICmp(P: Pred, LHS, RHS);
2829 Value *Ext = Builder.CreateSExt(V: Cmp, DestTy: Ty);
2830 return Ext;
2831}
2832
2833static Value *upgradeX86ConcatShift(IRBuilder<> &Builder, CallBase &CI,
2834 bool IsShiftRight, bool ZeroMask) {
2835 Type *Ty = CI.getType();
2836 Value *Op0 = CI.getArgOperand(i: 0);
2837 Value *Op1 = CI.getArgOperand(i: 1);
2838 Value *Amt = CI.getArgOperand(i: 2);
2839
2840 if (IsShiftRight)
2841 std::swap(a&: Op0, b&: Op1);
2842
2843 // Amount may be scalar immediate, in which case create a splat vector.
2844 // Funnel shifts amounts are treated as modulo and types are all power-of-2 so
2845 // we only care about the lowest log2 bits anyway.
2846 if (Amt->getType() != Ty) {
2847 unsigned NumElts = cast<FixedVectorType>(Val: Ty)->getNumElements();
2848 Amt = Builder.CreateIntCast(V: Amt, DestTy: Ty->getScalarType(), isSigned: false);
2849 Amt = Builder.CreateVectorSplat(NumElts, V: Amt);
2850 }
2851
2852 Intrinsic::ID IID = IsShiftRight ? Intrinsic::fshr : Intrinsic::fshl;
2853 Value *Res = Builder.CreateIntrinsic(ID: IID, OverloadTypes: Ty, Args: {Op0, Op1, Amt});
2854
2855 unsigned NumArgs = CI.arg_size();
2856 if (NumArgs >= 4) { // For masked intrinsics.
2857 Value *VecSrc = NumArgs == 5 ? CI.getArgOperand(i: 3) :
2858 ZeroMask ? ConstantAggregateZero::get(Ty: CI.getType()) :
2859 CI.getArgOperand(i: 0);
2860 Value *Mask = CI.getOperand(i_nocapture: NumArgs - 1);
2861 Res = emitX86Select(Builder, Mask, Op0: Res, Op1: VecSrc);
2862 }
2863 return Res;
2864}
2865
2866static Value *upgradeMaskedStore(IRBuilder<> &Builder, Value *Ptr, Value *Data,
2867 Value *Mask, bool Aligned) {
2868 const Align Alignment =
2869 Aligned
2870 ? Align(Data->getType()->getPrimitiveSizeInBits().getFixedValue() / 8)
2871 : Align(1);
2872
2873 // If the mask is all ones just emit a regular store.
2874 if (const auto *C = dyn_cast<Constant>(Val: Mask))
2875 if (C->isAllOnesValue())
2876 return Builder.CreateAlignedStore(Val: Data, Ptr, Align: Alignment);
2877
2878 // Convert the mask from an integer type to a vector of i1.
2879 unsigned NumElts = cast<FixedVectorType>(Val: Data->getType())->getNumElements();
2880 Mask = getX86MaskVec(Builder, Mask, NumElts);
2881 return Builder.CreateMaskedStore(Val: Data, Ptr, Alignment, Mask);
2882}
2883
2884static Value *upgradeMaskedLoad(IRBuilder<> &Builder, Value *Ptr,
2885 Value *Passthru, Value *Mask, bool Aligned) {
2886 Type *ValTy = Passthru->getType();
2887 const Align Alignment =
2888 Aligned
2889 ? Align(
2890 Passthru->getType()->getPrimitiveSizeInBits().getFixedValue() /
2891 8)
2892 : Align(1);
2893
2894 // If the mask is all ones just emit a regular store.
2895 if (const auto *C = dyn_cast<Constant>(Val: Mask))
2896 if (C->isAllOnesValue())
2897 return Builder.CreateAlignedLoad(Ty: ValTy, Ptr, Align: Alignment);
2898
2899 // Convert the mask from an integer type to a vector of i1.
2900 unsigned NumElts = cast<FixedVectorType>(Val: ValTy)->getNumElements();
2901 Mask = getX86MaskVec(Builder, Mask, NumElts);
2902 return Builder.CreateMaskedLoad(Ty: ValTy, Ptr, Alignment, Mask, PassThru: Passthru);
2903}
2904
2905static Value *upgradeAbs(IRBuilder<> &Builder, CallBase &CI) {
2906 Type *Ty = CI.getType();
2907 Value *Op0 = CI.getArgOperand(i: 0);
2908 Value *Res = Builder.CreateIntrinsic(ID: Intrinsic::abs, OverloadTypes: Ty,
2909 Args: {Op0, Builder.getInt1(V: false)});
2910 if (CI.arg_size() == 3)
2911 Res = emitX86Select(Builder, Mask: CI.getArgOperand(i: 2), Op0: Res, Op1: CI.getArgOperand(i: 1));
2912 return Res;
2913}
2914
2915static Value *upgradePMULDQ(IRBuilder<> &Builder, CallBase &CI, bool IsSigned) {
2916 Type *Ty = CI.getType();
2917
2918 // Arguments have a vXi32 type so cast to vXi64.
2919 Value *LHS = Builder.CreateBitCast(V: CI.getArgOperand(i: 0), DestTy: Ty);
2920 Value *RHS = Builder.CreateBitCast(V: CI.getArgOperand(i: 1), DestTy: Ty);
2921
2922 if (IsSigned) {
2923 // Shift left then arithmetic shift right.
2924 Constant *ShiftAmt = ConstantInt::get(Ty, V: 32);
2925 LHS = Builder.CreateShl(LHS, RHS: ShiftAmt);
2926 LHS = Builder.CreateAShr(LHS, RHS: ShiftAmt);
2927 RHS = Builder.CreateShl(LHS: RHS, RHS: ShiftAmt);
2928 RHS = Builder.CreateAShr(LHS: RHS, RHS: ShiftAmt);
2929 } else {
2930 // Clear the upper bits.
2931 Constant *Mask = ConstantInt::get(Ty, V: 0xffffffff);
2932 LHS = Builder.CreateAnd(LHS, RHS: Mask);
2933 RHS = Builder.CreateAnd(LHS: RHS, RHS: Mask);
2934 }
2935
2936 Value *Res = Builder.CreateMul(LHS, RHS);
2937
2938 if (CI.arg_size() == 4)
2939 Res = emitX86Select(Builder, Mask: CI.getArgOperand(i: 3), Op0: Res, Op1: CI.getArgOperand(i: 2));
2940
2941 return Res;
2942}
2943
2944// Applying mask on vector of i1's and make sure result is at least 8 bits wide.
2945static Value *applyX86MaskOn1BitsVec(IRBuilder<> &Builder, Value *Vec,
2946 Value *Mask) {
2947 unsigned NumElts = cast<FixedVectorType>(Val: Vec->getType())->getNumElements();
2948 if (Mask) {
2949 const auto *C = dyn_cast<Constant>(Val: Mask);
2950 if (!C || !C->isAllOnesValue())
2951 Vec = Builder.CreateAnd(LHS: Vec, RHS: getX86MaskVec(Builder, Mask, NumElts));
2952 }
2953
2954 if (NumElts < 8) {
2955 int Indices[8];
2956 for (unsigned i = 0; i != NumElts; ++i)
2957 Indices[i] = i;
2958 for (unsigned i = NumElts; i != 8; ++i)
2959 Indices[i] = NumElts + i % NumElts;
2960 Vec = Builder.CreateShuffleVector(V1: Vec,
2961 V2: Constant::getNullValue(Ty: Vec->getType()),
2962 Mask: Indices);
2963 }
2964 return Builder.CreateBitCast(V: Vec, DestTy: Builder.getIntNTy(N: std::max(a: NumElts, b: 8U)));
2965}
2966
2967static Value *upgradeMaskedCompare(IRBuilder<> &Builder, CallBase &CI,
2968 unsigned CC, bool Signed) {
2969 Value *Op0 = CI.getArgOperand(i: 0);
2970 unsigned NumElts = cast<FixedVectorType>(Val: Op0->getType())->getNumElements();
2971
2972 Value *Cmp;
2973 if (CC == 3) {
2974 Cmp = Constant::getNullValue(
2975 Ty: FixedVectorType::get(ElementType: Builder.getInt1Ty(), NumElts));
2976 } else if (CC == 7) {
2977 Cmp = Constant::getAllOnesValue(
2978 Ty: FixedVectorType::get(ElementType: Builder.getInt1Ty(), NumElts));
2979 } else {
2980 ICmpInst::Predicate Pred;
2981 switch (CC) {
2982 default: llvm_unreachable("Unknown condition code");
2983 case 0: Pred = ICmpInst::ICMP_EQ; break;
2984 case 1: Pred = Signed ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT; break;
2985 case 2: Pred = Signed ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE; break;
2986 case 4: Pred = ICmpInst::ICMP_NE; break;
2987 case 5: Pred = Signed ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE; break;
2988 case 6: Pred = Signed ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT; break;
2989 }
2990 Cmp = Builder.CreateICmp(P: Pred, LHS: Op0, RHS: CI.getArgOperand(i: 1));
2991 }
2992
2993 Value *Mask = CI.getArgOperand(i: CI.arg_size() - 1);
2994
2995 return applyX86MaskOn1BitsVec(Builder, Vec: Cmp, Mask);
2996}
2997
2998// Replace a masked intrinsic with an older unmasked intrinsic.
2999static Value *upgradeX86MaskedShift(IRBuilder<> &Builder, CallBase &CI,
3000 Intrinsic::ID IID) {
3001 Value *Rep =
3002 Builder.CreateIntrinsic(ID: IID, Args: {CI.getArgOperand(i: 0), CI.getArgOperand(i: 1)});
3003 return emitX86Select(Builder, Mask: CI.getArgOperand(i: 3), Op0: Rep, Op1: CI.getArgOperand(i: 2));
3004}
3005
3006static Value *upgradeMaskedMove(IRBuilder<> &Builder, CallBase &CI) {
3007 Value* A = CI.getArgOperand(i: 0);
3008 Value* B = CI.getArgOperand(i: 1);
3009 Value* Src = CI.getArgOperand(i: 2);
3010 Value* Mask = CI.getArgOperand(i: 3);
3011
3012 Value* AndNode = Builder.CreateAnd(LHS: Mask, RHS: APInt(8, 1));
3013 Value* Cmp = Builder.CreateIsNotNull(Arg: AndNode);
3014 Value* Extract1 = Builder.CreateExtractElement(Vec: B, Idx: (uint64_t)0);
3015 Value* Extract2 = Builder.CreateExtractElement(Vec: Src, Idx: (uint64_t)0);
3016 Value* Select = Builder.CreateSelect(C: Cmp, True: Extract1, False: Extract2);
3017 return Builder.CreateInsertElement(Vec: A, NewElt: Select, Idx: (uint64_t)0);
3018}
3019
3020static Value *upgradeMaskToInt(IRBuilder<> &Builder, CallBase &CI) {
3021 Value* Op = CI.getArgOperand(i: 0);
3022 Type* ReturnOp = CI.getType();
3023 unsigned NumElts = cast<FixedVectorType>(Val: CI.getType())->getNumElements();
3024 Value *Mask = getX86MaskVec(Builder, Mask: Op, NumElts);
3025 return Builder.CreateSExt(V: Mask, DestTy: ReturnOp, Name: "vpmovm2");
3026}
3027
3028// Replace intrinsic with unmasked version and a select.
3029static bool upgradeAVX512MaskToSelect(StringRef Name, IRBuilder<> &Builder,
3030 CallBase &CI, Value *&Rep) {
3031 Name = Name.substr(Start: 12); // Remove avx512.mask.
3032
3033 unsigned VecWidth = CI.getType()->getPrimitiveSizeInBits();
3034 unsigned EltWidth = CI.getType()->getScalarSizeInBits();
3035 Intrinsic::ID IID;
3036 if (Name.starts_with(Prefix: "max.p")) {
3037 if (VecWidth == 128 && EltWidth == 32)
3038 IID = Intrinsic::x86_sse_max_ps;
3039 else if (VecWidth == 128 && EltWidth == 64)
3040 IID = Intrinsic::x86_sse2_max_pd;
3041 else if (VecWidth == 256 && EltWidth == 32)
3042 IID = Intrinsic::x86_avx_max_ps_256;
3043 else if (VecWidth == 256 && EltWidth == 64)
3044 IID = Intrinsic::x86_avx_max_pd_256;
3045 else
3046 llvm_unreachable("Unexpected intrinsic");
3047 } else if (Name.starts_with(Prefix: "min.p")) {
3048 if (VecWidth == 128 && EltWidth == 32)
3049 IID = Intrinsic::x86_sse_min_ps;
3050 else if (VecWidth == 128 && EltWidth == 64)
3051 IID = Intrinsic::x86_sse2_min_pd;
3052 else if (VecWidth == 256 && EltWidth == 32)
3053 IID = Intrinsic::x86_avx_min_ps_256;
3054 else if (VecWidth == 256 && EltWidth == 64)
3055 IID = Intrinsic::x86_avx_min_pd_256;
3056 else
3057 llvm_unreachable("Unexpected intrinsic");
3058 } else if (Name.starts_with(Prefix: "pshuf.b.")) {
3059 if (VecWidth == 128)
3060 IID = Intrinsic::x86_ssse3_pshuf_b_128;
3061 else if (VecWidth == 256)
3062 IID = Intrinsic::x86_avx2_pshuf_b;
3063 else if (VecWidth == 512)
3064 IID = Intrinsic::x86_avx512_pshuf_b_512;
3065 else
3066 llvm_unreachable("Unexpected intrinsic");
3067 } else if (Name.starts_with(Prefix: "pmul.hr.sw.")) {
3068 if (VecWidth == 128)
3069 IID = Intrinsic::x86_ssse3_pmul_hr_sw_128;
3070 else if (VecWidth == 256)
3071 IID = Intrinsic::x86_avx2_pmul_hr_sw;
3072 else if (VecWidth == 512)
3073 IID = Intrinsic::x86_avx512_pmul_hr_sw_512;
3074 else
3075 llvm_unreachable("Unexpected intrinsic");
3076 } else if (Name.starts_with(Prefix: "pmulh.w")) {
3077 assert((VecWidth == 128 || VecWidth == 256 || VecWidth == 512) &&
3078 "Unexpected intrinsic");
3079 Rep = upgradeX86BinaryIntrinsics(Builder, CI, IID: Intrinsic::smulh);
3080 return true;
3081 } else if (Name.starts_with(Prefix: "pmulhu.w")) {
3082 assert((VecWidth == 128 || VecWidth == 256 || VecWidth == 512) &&
3083 "Unexpected intrinsic");
3084 Rep = upgradeX86BinaryIntrinsics(Builder, CI, IID: Intrinsic::umulh);
3085 return true;
3086 } else if (Name.starts_with(Prefix: "pmaddw.d.")) {
3087 if (VecWidth == 128)
3088 IID = Intrinsic::x86_sse2_pmadd_wd;
3089 else if (VecWidth == 256)
3090 IID = Intrinsic::x86_avx2_pmadd_wd;
3091 else if (VecWidth == 512)
3092 IID = Intrinsic::x86_avx512_pmaddw_d_512;
3093 else
3094 llvm_unreachable("Unexpected intrinsic");
3095 } else if (Name.starts_with(Prefix: "pmaddubs.w.")) {
3096 if (VecWidth == 128)
3097 IID = Intrinsic::x86_ssse3_pmadd_ub_sw_128;
3098 else if (VecWidth == 256)
3099 IID = Intrinsic::x86_avx2_pmadd_ub_sw;
3100 else if (VecWidth == 512)
3101 IID = Intrinsic::x86_avx512_pmaddubs_w_512;
3102 else
3103 llvm_unreachable("Unexpected intrinsic");
3104 } else if (Name.starts_with(Prefix: "packsswb.")) {
3105 if (VecWidth == 128)
3106 IID = Intrinsic::x86_sse2_packsswb_128;
3107 else if (VecWidth == 256)
3108 IID = Intrinsic::x86_avx2_packsswb;
3109 else if (VecWidth == 512)
3110 IID = Intrinsic::x86_avx512_packsswb_512;
3111 else
3112 llvm_unreachable("Unexpected intrinsic");
3113 } else if (Name.starts_with(Prefix: "packssdw.")) {
3114 if (VecWidth == 128)
3115 IID = Intrinsic::x86_sse2_packssdw_128;
3116 else if (VecWidth == 256)
3117 IID = Intrinsic::x86_avx2_packssdw;
3118 else if (VecWidth == 512)
3119 IID = Intrinsic::x86_avx512_packssdw_512;
3120 else
3121 llvm_unreachable("Unexpected intrinsic");
3122 } else if (Name.starts_with(Prefix: "packuswb.")) {
3123 if (VecWidth == 128)
3124 IID = Intrinsic::x86_sse2_packuswb_128;
3125 else if (VecWidth == 256)
3126 IID = Intrinsic::x86_avx2_packuswb;
3127 else if (VecWidth == 512)
3128 IID = Intrinsic::x86_avx512_packuswb_512;
3129 else
3130 llvm_unreachable("Unexpected intrinsic");
3131 } else if (Name.starts_with(Prefix: "packusdw.")) {
3132 if (VecWidth == 128)
3133 IID = Intrinsic::x86_sse41_packusdw;
3134 else if (VecWidth == 256)
3135 IID = Intrinsic::x86_avx2_packusdw;
3136 else if (VecWidth == 512)
3137 IID = Intrinsic::x86_avx512_packusdw_512;
3138 else
3139 llvm_unreachable("Unexpected intrinsic");
3140 } else if (Name.starts_with(Prefix: "vpermilvar.")) {
3141 if (VecWidth == 128 && EltWidth == 32)
3142 IID = Intrinsic::x86_avx_vpermilvar_ps;
3143 else if (VecWidth == 128 && EltWidth == 64)
3144 IID = Intrinsic::x86_avx_vpermilvar_pd;
3145 else if (VecWidth == 256 && EltWidth == 32)
3146 IID = Intrinsic::x86_avx_vpermilvar_ps_256;
3147 else if (VecWidth == 256 && EltWidth == 64)
3148 IID = Intrinsic::x86_avx_vpermilvar_pd_256;
3149 else if (VecWidth == 512 && EltWidth == 32)
3150 IID = Intrinsic::x86_avx512_vpermilvar_ps_512;
3151 else if (VecWidth == 512 && EltWidth == 64)
3152 IID = Intrinsic::x86_avx512_vpermilvar_pd_512;
3153 else
3154 llvm_unreachable("Unexpected intrinsic");
3155 } else if (Name == "cvtpd2dq.256") {
3156 IID = Intrinsic::x86_avx_cvt_pd2dq_256;
3157 } else if (Name == "cvtpd2ps.256") {
3158 IID = Intrinsic::x86_avx_cvt_pd2_ps_256;
3159 } else if (Name == "cvttpd2dq.256") {
3160 IID = Intrinsic::x86_avx_cvtt_pd2dq_256;
3161 } else if (Name == "cvttps2dq.128") {
3162 IID = Intrinsic::x86_sse2_cvttps2dq;
3163 } else if (Name == "cvttps2dq.256") {
3164 IID = Intrinsic::x86_avx_cvtt_ps2dq_256;
3165 } else if (Name.starts_with(Prefix: "permvar.")) {
3166 bool IsFloat = CI.getType()->isFPOrFPVectorTy();
3167 if (VecWidth == 256 && EltWidth == 32 && IsFloat)
3168 IID = Intrinsic::x86_avx2_permps;
3169 else if (VecWidth == 256 && EltWidth == 32 && !IsFloat)
3170 IID = Intrinsic::x86_avx2_permd;
3171 else if (VecWidth == 256 && EltWidth == 64 && IsFloat)
3172 IID = Intrinsic::x86_avx512_permvar_df_256;
3173 else if (VecWidth == 256 && EltWidth == 64 && !IsFloat)
3174 IID = Intrinsic::x86_avx512_permvar_di_256;
3175 else if (VecWidth == 512 && EltWidth == 32 && IsFloat)
3176 IID = Intrinsic::x86_avx512_permvar_sf_512;
3177 else if (VecWidth == 512 && EltWidth == 32 && !IsFloat)
3178 IID = Intrinsic::x86_avx512_permvar_si_512;
3179 else if (VecWidth == 512 && EltWidth == 64 && IsFloat)
3180 IID = Intrinsic::x86_avx512_permvar_df_512;
3181 else if (VecWidth == 512 && EltWidth == 64 && !IsFloat)
3182 IID = Intrinsic::x86_avx512_permvar_di_512;
3183 else if (VecWidth == 128 && EltWidth == 16)
3184 IID = Intrinsic::x86_avx512_permvar_hi_128;
3185 else if (VecWidth == 256 && EltWidth == 16)
3186 IID = Intrinsic::x86_avx512_permvar_hi_256;
3187 else if (VecWidth == 512 && EltWidth == 16)
3188 IID = Intrinsic::x86_avx512_permvar_hi_512;
3189 else if (VecWidth == 128 && EltWidth == 8)
3190 IID = Intrinsic::x86_avx512_permvar_qi_128;
3191 else if (VecWidth == 256 && EltWidth == 8)
3192 IID = Intrinsic::x86_avx512_permvar_qi_256;
3193 else if (VecWidth == 512 && EltWidth == 8)
3194 IID = Intrinsic::x86_avx512_permvar_qi_512;
3195 else
3196 llvm_unreachable("Unexpected intrinsic");
3197 } else if (Name.starts_with(Prefix: "dbpsadbw.")) {
3198 if (VecWidth == 128)
3199 IID = Intrinsic::x86_avx512_dbpsadbw_128;
3200 else if (VecWidth == 256)
3201 IID = Intrinsic::x86_avx512_dbpsadbw_256;
3202 else if (VecWidth == 512)
3203 IID = Intrinsic::x86_avx512_dbpsadbw_512;
3204 else
3205 llvm_unreachable("Unexpected intrinsic");
3206 } else if (Name.starts_with(Prefix: "pmultishift.qb.")) {
3207 if (VecWidth == 128)
3208 IID = Intrinsic::x86_avx512_pmultishift_qb_128;
3209 else if (VecWidth == 256)
3210 IID = Intrinsic::x86_avx512_pmultishift_qb_256;
3211 else if (VecWidth == 512)
3212 IID = Intrinsic::x86_avx512_pmultishift_qb_512;
3213 else
3214 llvm_unreachable("Unexpected intrinsic");
3215 } else if (Name.starts_with(Prefix: "conflict.")) {
3216 if (Name[9] == 'd' && VecWidth == 128)
3217 IID = Intrinsic::x86_avx512_conflict_d_128;
3218 else if (Name[9] == 'd' && VecWidth == 256)
3219 IID = Intrinsic::x86_avx512_conflict_d_256;
3220 else if (Name[9] == 'd' && VecWidth == 512)
3221 IID = Intrinsic::x86_avx512_conflict_d_512;
3222 else if (Name[9] == 'q' && VecWidth == 128)
3223 IID = Intrinsic::x86_avx512_conflict_q_128;
3224 else if (Name[9] == 'q' && VecWidth == 256)
3225 IID = Intrinsic::x86_avx512_conflict_q_256;
3226 else if (Name[9] == 'q' && VecWidth == 512)
3227 IID = Intrinsic::x86_avx512_conflict_q_512;
3228 else
3229 llvm_unreachable("Unexpected intrinsic");
3230 } else if (Name.starts_with(Prefix: "pavg.")) {
3231 if (Name[5] == 'b' && VecWidth == 128)
3232 IID = Intrinsic::x86_sse2_pavg_b;
3233 else if (Name[5] == 'b' && VecWidth == 256)
3234 IID = Intrinsic::x86_avx2_pavg_b;
3235 else if (Name[5] == 'b' && VecWidth == 512)
3236 IID = Intrinsic::x86_avx512_pavg_b_512;
3237 else if (Name[5] == 'w' && VecWidth == 128)
3238 IID = Intrinsic::x86_sse2_pavg_w;
3239 else if (Name[5] == 'w' && VecWidth == 256)
3240 IID = Intrinsic::x86_avx2_pavg_w;
3241 else if (Name[5] == 'w' && VecWidth == 512)
3242 IID = Intrinsic::x86_avx512_pavg_w_512;
3243 else
3244 llvm_unreachable("Unexpected intrinsic");
3245 } else
3246 return false;
3247
3248 SmallVector<Value *, 4> Args(CI.args());
3249 Args.pop_back();
3250 Args.pop_back();
3251 Rep = Builder.CreateIntrinsic(ID: IID, Args);
3252 unsigned NumArgs = CI.arg_size();
3253 Rep = emitX86Select(Builder, Mask: CI.getArgOperand(i: NumArgs - 1), Op0: Rep,
3254 Op1: CI.getArgOperand(i: NumArgs - 2));
3255 return true;
3256}
3257
3258/// Upgrade comment in call to inline asm that represents an objc retain release
3259/// marker.
3260void llvm::UpgradeInlineAsmString(std::string *AsmStr) {
3261 size_t Pos;
3262 if (AsmStr->find(s: "mov\tfp") == 0 &&
3263 AsmStr->find(s: "objc_retainAutoreleaseReturnValue") != std::string::npos &&
3264 (Pos = AsmStr->find(s: "# marker")) != std::string::npos) {
3265 AsmStr->replace(pos: Pos, n1: 1, s: ";");
3266 }
3267}
3268
3269static Value *upgradeNVVMFPArithCall(IRBuilder<> &Builder, CallBase *CI,
3270 StringRef Name,
3271 const Intrinsic::ID (&IIDs)[2][2]) {
3272 auto Result = getNVVMFPArithUpgrade(Name, IIDs);
3273 assert(Result && "unsupported nvvm.add.*/nvvm.mul.* intrinsic");
3274 auto [IID, RoundingMode] = *Result;
3275 Value *A = CI->getArgOperand(i: 0);
3276 return Builder.CreateIntrinsic(
3277 RetTy: A->getType(), ID: IID,
3278 Args: {A, CI->getArgOperand(i: 1),
3279 Builder.getInt32(C: static_cast<int>(RoundingMode))});
3280}
3281
3282static Value *upgradeNVVMIntrinsicCall(StringRef Name, CallBase *CI,
3283 Function *F, IRBuilder<> &Builder) {
3284 Value *Rep = nullptr;
3285
3286 if (Name == "abs.i" || Name == "abs.ll") {
3287 Value *Arg = CI->getArgOperand(i: 0);
3288 Rep = Builder.CreateIntrinsic(ID: Intrinsic::abs, OverloadTypes: {Arg->getType()},
3289 Args: {Arg, Builder.getTrue()},
3290 /*FMFSource=*/nullptr, Name: "abs");
3291 } else if (Name == "abs.bf16" || Name == "abs.bf16x2") {
3292 Type *Ty = (Name == "abs.bf16")
3293 ? Builder.getBFloatTy()
3294 : FixedVectorType::get(ElementType: Builder.getBFloatTy(), NumElts: 2);
3295 Value *Arg = Builder.CreateBitCast(V: CI->getArgOperand(i: 0), DestTy: Ty);
3296 Value *Abs = Builder.CreateUnaryIntrinsic(ID: Intrinsic::nvvm_fabs, Op: Arg);
3297 Rep = Builder.CreateBitCast(V: Abs, DestTy: CI->getType());
3298 } else if (Name == "fabs.f" || Name == "fabs.ftz.f" || Name == "fabs.d") {
3299 Intrinsic::ID IID = (Name == "fabs.ftz.f") ? Intrinsic::nvvm_fabs_ftz
3300 : Intrinsic::nvvm_fabs;
3301 Rep = Builder.CreateUnaryIntrinsic(ID: IID, Op: CI->getArgOperand(i: 0));
3302 } else if (Name.consume_front(Prefix: "add.")) {
3303 // nvvm.add.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
3304 Rep = upgradeNVVMFPArithCall(Builder, CI, Name, IIDs: NVVMFAddIIDs);
3305 } else if (Name.consume_front(Prefix: "mul.")) {
3306 // nvvm.mul.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
3307 Rep = upgradeNVVMFPArithCall(Builder, CI, Name, IIDs: NVVMFMulIIDs);
3308 } else if (Name.consume_front(Prefix: "ex2.approx.")) {
3309 // nvvm.ex2.approx.{f,ftz.f,d,f16x2}
3310 Intrinsic::ID IID = Name.starts_with(Prefix: "ftz") ? Intrinsic::nvvm_ex2_approx_ftz
3311 : Intrinsic::nvvm_ex2_approx;
3312 Rep = Builder.CreateUnaryIntrinsic(ID: IID, Op: CI->getArgOperand(i: 0));
3313 } else if (Name.starts_with(Prefix: "atomic.load.add.f32.p") ||
3314 Name.starts_with(Prefix: "atomic.load.add.f64.p")) {
3315 Value *Ptr = CI->getArgOperand(i: 0);
3316 Value *Val = CI->getArgOperand(i: 1);
3317 Rep = Builder.CreateAtomicRMW(
3318 Op: AtomicRMWInst::FAdd, Ptr, Val, Align: MaybeAlign(), Ordering: AtomicOrdering::Monotonic,
3319 SSID: CI->getContext().getOrInsertSyncScopeID(SSN: "device"));
3320 // The default scope for atomic.load.* intrinsics is device
3321 // (= gpu scope in ptx), but the default LLVM atomic scope is
3322 // "system"
3323 } else if (Name.starts_with(Prefix: "atomic.load.inc.32.p") ||
3324 Name.starts_with(Prefix: "atomic.load.dec.32.p")) {
3325 Value *Ptr = CI->getArgOperand(i: 0);
3326 Value *Val = CI->getArgOperand(i: 1);
3327 auto Op = Name.starts_with(Prefix: "atomic.load.inc") ? AtomicRMWInst::UIncWrap
3328 : AtomicRMWInst::UDecWrap;
3329 Rep = Builder.CreateAtomicRMW(
3330 Op, Ptr, Val, Align: MaybeAlign(), Ordering: AtomicOrdering::Monotonic,
3331 SSID: CI->getContext().getOrInsertSyncScopeID(SSN: "device"));
3332 // See comment above.
3333 } else if (Name.starts_with(Prefix: "atomic.") && Name.contains(Other: ".gen.")) {
3334 // nvvm.atomic.{op}.gen.{i,f}.{cta,sys} -> atomicrmw / cmpxchg.
3335 StringRef Op = Name.substr(Start: StringRef("atomic.").size());
3336 Value *Ptr = CI->getArgOperand(i: 0);
3337 Value *Val = CI->getArgOperand(i: 1);
3338 SyncScope::ID SSID = CI->getContext().getOrInsertSyncScopeID(
3339 SSN: Op.contains(Other: ".cta.") ? "block" : "");
3340 if (Op.starts_with(Prefix: "cas.")) {
3341 Value *New = CI->getArgOperand(i: 2);
3342 Value *Pair = Builder.CreateAtomicCmpXchg(
3343 Ptr, Cmp: Val, New, Align: MaybeAlign(), SuccessOrdering: AtomicOrdering::Monotonic,
3344 FailureOrdering: AtomicOrdering::Monotonic, SSID);
3345 Rep = Builder.CreateExtractValue(Agg: Pair, Idxs: 0);
3346 } else {
3347 // Note we don't upgrade anything to AtomicRMWInst::UMin/UMax. This is
3348 // because we were actually missing those intrinsics!
3349 AtomicRMWInst::BinOp BinOp =
3350 StringSwitch<AtomicRMWInst::BinOp>(Op)
3351 .StartsWith(S: "add.gen.f", Value: AtomicRMWInst::FAdd)
3352 .StartsWith(S: "add.gen.i", Value: AtomicRMWInst::Add)
3353 .StartsWith(S: "exch.", Value: AtomicRMWInst::Xchg)
3354 .StartsWith(S: "max.", Value: AtomicRMWInst::Max)
3355 .StartsWith(S: "min.", Value: AtomicRMWInst::Min)
3356 .StartsWith(S: "inc.", Value: AtomicRMWInst::UIncWrap)
3357 .StartsWith(S: "dec.", Value: AtomicRMWInst::UDecWrap)
3358 .StartsWith(S: "and.", Value: AtomicRMWInst::And)
3359 .StartsWith(S: "or.", Value: AtomicRMWInst::Or)
3360 .StartsWith(S: "xor.", Value: AtomicRMWInst::Xor)
3361 .Default(Value: AtomicRMWInst::BAD_BINOP);
3362 assert(BinOp != AtomicRMWInst::BAD_BINOP &&
3363 "unexpected nvvm scoped atomic intrinsic");
3364 Rep = Builder.CreateAtomicRMW(Op: BinOp, Ptr, Val, Align: MaybeAlign(),
3365 Ordering: AtomicOrdering::Monotonic, SSID);
3366 }
3367 } else if (Name == "clz.ll") {
3368 // llvm.nvvm.clz.ll returns an i32, but llvm.ctlz.i64 returns an i64.
3369 Value *Arg = CI->getArgOperand(i: 0);
3370 Value *Ctlz = Builder.CreateIntrinsic(ID: Intrinsic::ctlz, OverloadTypes: {Arg->getType()},
3371 Args: {Arg, Builder.getFalse()},
3372 /*FMFSource=*/nullptr, Name: "ctlz");
3373 Rep = Builder.CreateTrunc(V: Ctlz, DestTy: Builder.getInt32Ty(), Name: "ctlz.trunc");
3374 } else if (Name == "popc.ll") {
3375 // llvm.nvvm.popc.ll returns an i32, but llvm.ctpop.i64 returns an
3376 // i64.
3377 Value *Arg = CI->getArgOperand(i: 0);
3378 Value *Popc = Builder.CreateIntrinsic(ID: Intrinsic::ctpop, OverloadTypes: {Arg->getType()},
3379 Args: Arg, /*FMFSource=*/nullptr, Name: "ctpop");
3380 Rep = Builder.CreateTrunc(V: Popc, DestTy: Builder.getInt32Ty(), Name: "ctpop.trunc");
3381 } else if (Name == "h2f") {
3382 Value *Cast =
3383 Builder.CreateBitCast(V: CI->getArgOperand(i: 0), DestTy: Builder.getHalfTy());
3384 Rep = Builder.CreateFPExt(V: Cast, DestTy: Builder.getFloatTy());
3385 } else if (Name.consume_front(Prefix: "bitcast.") &&
3386 (Name == "f2i" || Name == "i2f" || Name == "ll2d" ||
3387 Name == "d2ll")) {
3388 Rep = Builder.CreateBitCast(V: CI->getArgOperand(i: 0), DestTy: CI->getType());
3389 } else if (Name == "rotate.b32") {
3390 Value *Arg = CI->getOperand(i_nocapture: 0);
3391 Value *ShiftAmt = CI->getOperand(i_nocapture: 1);
3392 Rep = Builder.CreateIntrinsic(RetTy: Builder.getInt32Ty(), ID: Intrinsic::fshl,
3393 Args: {Arg, Arg, ShiftAmt});
3394 } else if (Name == "rotate.b64") {
3395 Type *Int64Ty = Builder.getInt64Ty();
3396 Value *Arg = CI->getOperand(i_nocapture: 0);
3397 Value *ZExtShiftAmt = Builder.CreateZExt(V: CI->getOperand(i_nocapture: 1), DestTy: Int64Ty);
3398 Rep = Builder.CreateIntrinsic(RetTy: Int64Ty, ID: Intrinsic::fshl,
3399 Args: {Arg, Arg, ZExtShiftAmt});
3400 } else if (Name == "rotate.right.b64") {
3401 Type *Int64Ty = Builder.getInt64Ty();
3402 Value *Arg = CI->getOperand(i_nocapture: 0);
3403 Value *ZExtShiftAmt = Builder.CreateZExt(V: CI->getOperand(i_nocapture: 1), DestTy: Int64Ty);
3404 Rep = Builder.CreateIntrinsic(RetTy: Int64Ty, ID: Intrinsic::fshr,
3405 Args: {Arg, Arg, ZExtShiftAmt});
3406 } else if (Name == "swap.lo.hi.b64") {
3407 Type *Int64Ty = Builder.getInt64Ty();
3408 Value *Arg = CI->getOperand(i_nocapture: 0);
3409 Rep = Builder.CreateIntrinsic(RetTy: Int64Ty, ID: Intrinsic::fshl,
3410 Args: {Arg, Arg, Builder.getInt64(C: 32)});
3411 } else if ((Name.consume_front(Prefix: "ptr.gen.to.") &&
3412 consumeNVVMPtrAddrSpace(Name)) ||
3413 (Name.consume_front(Prefix: "ptr.") && consumeNVVMPtrAddrSpace(Name) &&
3414 Name.starts_with(Prefix: ".to.gen"))) {
3415 Rep = Builder.CreateAddrSpaceCast(V: CI->getArgOperand(i: 0), DestTy: CI->getType());
3416 } else if (Name.consume_front(Prefix: "ldg.global")) {
3417 Value *Ptr = CI->getArgOperand(i: 0);
3418 Align PtrAlign = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getAlignValue();
3419 // Use addrspace(1) for NVPTX ADDRESS_SPACE_GLOBAL
3420 Value *ASC = Builder.CreateAddrSpaceCast(V: Ptr, DestTy: Builder.getPtrTy(AddrSpace: 1));
3421 Instruction *LD = Builder.CreateAlignedLoad(Ty: CI->getType(), Ptr: ASC, Align: PtrAlign);
3422 MDNode *MD = MDNode::get(Context&: Builder.getContext(), MDs: {});
3423 LD->setMetadata(KindID: LLVMContext::MD_invariant_load, Node: MD);
3424 return LD;
3425 } else if (Name == "tanh.approx.f32") {
3426 // nvvm.tanh.approx.f32 -> afn llvm.tanh.f32
3427 FastMathFlags FMF;
3428 FMF.setApproxFunc();
3429 Rep = Builder.CreateUnaryIntrinsic(ID: Intrinsic::tanh, Op: CI->getArgOperand(i: 0),
3430 FMFSource: FMF);
3431 } else if (Name == "barrier0" || Name == "barrier.n" || Name == "bar.sync") {
3432 Value *Arg =
3433 Name.ends_with(Suffix: '0') ? Builder.getInt32(C: 0) : CI->getArgOperand(i: 0);
3434 Rep = Builder.CreateIntrinsic(ID: Intrinsic::nvvm_barrier_cta_sync_aligned_all,
3435 OverloadTypes: {}, Args: {Arg});
3436 } else if (Name == "barrier") {
3437 Rep = Builder.CreateIntrinsic(
3438 ID: Intrinsic::nvvm_barrier_cta_sync_aligned_count, OverloadTypes: {},
3439 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1)});
3440 } else if (Name == "barrier.sync") {
3441 Rep = Builder.CreateIntrinsic(ID: Intrinsic::nvvm_barrier_cta_sync_all, OverloadTypes: {},
3442 Args: {CI->getArgOperand(i: 0)});
3443 } else if (Name == "barrier.sync.cnt") {
3444 Rep = Builder.CreateIntrinsic(ID: Intrinsic::nvvm_barrier_cta_sync_count, OverloadTypes: {},
3445 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1)});
3446 } else if (Name == "barrier0.popc" || Name == "barrier0.and" ||
3447 Name == "barrier0.or") {
3448 Value *C = CI->getArgOperand(i: 0);
3449 C = Builder.CreateICmpNE(LHS: C, RHS: Builder.getInt32(C: 0));
3450
3451 Intrinsic::ID IID =
3452 StringSwitch<Intrinsic::ID>(Name)
3453 .Case(S: "barrier0.popc",
3454 Value: Intrinsic::nvvm_barrier_cta_red_popc_aligned_all)
3455 .Case(S: "barrier0.and",
3456 Value: Intrinsic::nvvm_barrier_cta_red_and_aligned_all)
3457 .Case(S: "barrier0.or",
3458 Value: Intrinsic::nvvm_barrier_cta_red_or_aligned_all);
3459 Value *Bar = Builder.CreateIntrinsic(ID: IID, OverloadTypes: {}, Args: {Builder.getInt32(C: 0), C});
3460 Rep = Builder.CreateZExt(V: Bar, DestTy: CI->getType());
3461 } else {
3462 Intrinsic::ID IID = shouldUpgradeNVPTXBF16Intrinsic(Name);
3463 if (IID != Intrinsic::not_intrinsic &&
3464 isLegacyNVPTXBF16IntSignature(F, IID)) {
3465 rename(GV: F);
3466 Function *NewFn = Intrinsic::getOrInsertDeclaration(M: F->getParent(), id: IID);
3467 SmallVector<Value *, 2> Args;
3468 for (size_t I = 0; I < NewFn->arg_size(); ++I) {
3469 Value *Arg = CI->getArgOperand(i: I);
3470 Type *OldType = Arg->getType();
3471 Type *NewType = NewFn->getArg(i: I)->getType();
3472 Args.push_back(
3473 Elt: (OldType->isIntegerTy() && NewType->getScalarType()->isBFloatTy())
3474 ? Builder.CreateBitCast(V: Arg, DestTy: NewType)
3475 : Arg);
3476 }
3477 Rep = Builder.CreateCall(Callee: NewFn, Args);
3478 if (F->getReturnType()->isIntegerTy())
3479 Rep = Builder.CreateBitCast(V: Rep, DestTy: F->getReturnType());
3480 }
3481 }
3482
3483 return Rep;
3484}
3485
3486static Value *upgradeX86IntrinsicCall(StringRef Name, CallBase *CI, Function *F,
3487 IRBuilder<> &Builder) {
3488 LLVMContext &C = F->getContext();
3489 Value *Rep = nullptr;
3490
3491 if (Name.starts_with(Prefix: "sse4a.movnt.")) {
3492 SmallVector<Metadata *, 1> Elts;
3493 Elts.push_back(
3494 Elt: ConstantAsMetadata::get(C: ConstantInt::get(Ty: Type::getInt32Ty(C), V: 1)));
3495 MDNode *Node = MDNode::get(Context&: C, MDs: Elts);
3496
3497 Value *Arg0 = CI->getArgOperand(i: 0);
3498 Value *Arg1 = CI->getArgOperand(i: 1);
3499
3500 // Nontemporal (unaligned) store of the 0'th element of the float/double
3501 // vector.
3502 Value *Extract =
3503 Builder.CreateExtractElement(Vec: Arg1, Idx: (uint64_t)0, Name: "extractelement");
3504
3505 StoreInst *SI = Builder.CreateAlignedStore(Val: Extract, Ptr: Arg0, Align: Align(1));
3506 SI->setMetadata(KindID: LLVMContext::MD_nontemporal, Node);
3507 } else if (Name.starts_with(Prefix: "avx.movnt.") ||
3508 Name.starts_with(Prefix: "avx512.storent.")) {
3509 SmallVector<Metadata *, 1> Elts;
3510 Elts.push_back(
3511 Elt: ConstantAsMetadata::get(C: ConstantInt::get(Ty: Type::getInt32Ty(C), V: 1)));
3512 MDNode *Node = MDNode::get(Context&: C, MDs: Elts);
3513
3514 Value *Arg0 = CI->getArgOperand(i: 0);
3515 Value *Arg1 = CI->getArgOperand(i: 1);
3516
3517 StoreInst *SI = Builder.CreateAlignedStore(
3518 Val: Arg1, Ptr: Arg0,
3519 Align: Align(Arg1->getType()->getPrimitiveSizeInBits().getFixedValue() / 8));
3520 SI->setMetadata(KindID: LLVMContext::MD_nontemporal, Node);
3521 } else if (Name == "sse2.storel.dq") {
3522 Value *Arg0 = CI->getArgOperand(i: 0);
3523 Value *Arg1 = CI->getArgOperand(i: 1);
3524
3525 auto *NewVecTy = FixedVectorType::get(ElementType: Type::getInt64Ty(C), NumElts: 2);
3526 Value *BC0 = Builder.CreateBitCast(V: Arg1, DestTy: NewVecTy, Name: "cast");
3527 Value *Elt = Builder.CreateExtractElement(Vec: BC0, Idx: (uint64_t)0);
3528 Builder.CreateAlignedStore(Val: Elt, Ptr: Arg0, Align: Align(1));
3529 } else if (Name.starts_with(Prefix: "sse.storeu.") ||
3530 Name.starts_with(Prefix: "sse2.storeu.") ||
3531 Name.starts_with(Prefix: "avx.storeu.")) {
3532 Value *Arg0 = CI->getArgOperand(i: 0);
3533 Value *Arg1 = CI->getArgOperand(i: 1);
3534 Builder.CreateAlignedStore(Val: Arg1, Ptr: Arg0, Align: Align(1));
3535 } else if (Name == "avx512.mask.store.ss") {
3536 Value *Mask = Builder.CreateAnd(LHS: CI->getArgOperand(i: 2), RHS: Builder.getInt8(C: 1));
3537 upgradeMaskedStore(Builder, Ptr: CI->getArgOperand(i: 0), Data: CI->getArgOperand(i: 1),
3538 Mask, Aligned: false);
3539 } else if (Name.starts_with(Prefix: "avx512.mask.store")) {
3540 // "avx512.mask.storeu." or "avx512.mask.store."
3541 bool Aligned = Name[17] != 'u'; // "avx512.mask.storeu".
3542 upgradeMaskedStore(Builder, Ptr: CI->getArgOperand(i: 0), Data: CI->getArgOperand(i: 1),
3543 Mask: CI->getArgOperand(i: 2), Aligned);
3544 } else if (Name.starts_with(Prefix: "sse2.pcmp") || Name.starts_with(Prefix: "avx2.pcmp")) {
3545 // Upgrade packed integer vector compare intrinsics to compare instructions.
3546 // "sse2.pcpmpeq." "sse2.pcmpgt." "avx2.pcmpeq." or "avx2.pcmpgt."
3547 bool CmpEq = Name[9] == 'e';
3548 Rep = Builder.CreateICmp(P: CmpEq ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_SGT,
3549 LHS: CI->getArgOperand(i: 0), RHS: CI->getArgOperand(i: 1));
3550 Rep = Builder.CreateSExt(V: Rep, DestTy: CI->getType(), Name: "");
3551 } else if (Name.starts_with(Prefix: "avx512.broadcastm")) {
3552 Type *ExtTy = Type::getInt32Ty(C);
3553 if (CI->getOperand(i_nocapture: 0)->getType()->isIntegerTy(BitWidth: 8))
3554 ExtTy = Type::getInt64Ty(C);
3555 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() /
3556 ExtTy->getPrimitiveSizeInBits();
3557 Rep = Builder.CreateZExt(V: CI->getArgOperand(i: 0), DestTy: ExtTy);
3558 Rep = Builder.CreateVectorSplat(NumElts, V: Rep);
3559 } else if (Name == "sse.sqrt.ss" || Name == "sse2.sqrt.sd") {
3560 Value *Vec = CI->getArgOperand(i: 0);
3561 Value *Elt0 = Builder.CreateExtractElement(Vec, Idx: (uint64_t)0);
3562 Elt0 = Builder.CreateIntrinsic(ID: Intrinsic::sqrt, OverloadTypes: Elt0->getType(), Args: Elt0);
3563 Rep = Builder.CreateInsertElement(Vec, NewElt: Elt0, Idx: (uint64_t)0);
3564 } else if (Name.starts_with(Prefix: "avx.sqrt.p") ||
3565 Name.starts_with(Prefix: "sse2.sqrt.p") ||
3566 Name.starts_with(Prefix: "sse.sqrt.p")) {
3567 Rep = Builder.CreateIntrinsic(ID: Intrinsic::sqrt, OverloadTypes: CI->getType(),
3568 Args: {CI->getArgOperand(i: 0)});
3569 } else if (Name.starts_with(Prefix: "avx512.mask.sqrt.p")) {
3570 if (CI->arg_size() == 4 &&
3571 (!isa<ConstantInt>(Val: CI->getArgOperand(i: 3)) ||
3572 cast<ConstantInt>(Val: CI->getArgOperand(i: 3))->getZExtValue() != 4)) {
3573 Intrinsic::ID IID = Name[18] == 's' ? Intrinsic::x86_avx512_sqrt_ps_512
3574 : Intrinsic::x86_avx512_sqrt_pd_512;
3575
3576 Value *Args[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 3)};
3577 Rep = Builder.CreateIntrinsic(ID: IID, Args);
3578 } else {
3579 Rep = Builder.CreateIntrinsic(ID: Intrinsic::sqrt, OverloadTypes: CI->getType(),
3580 Args: {CI->getArgOperand(i: 0)});
3581 }
3582 Rep =
3583 emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep, Op1: CI->getArgOperand(i: 1));
3584 } else if (Name.starts_with(Prefix: "avx512.ptestm") ||
3585 Name.starts_with(Prefix: "avx512.ptestnm")) {
3586 Value *Op0 = CI->getArgOperand(i: 0);
3587 Value *Op1 = CI->getArgOperand(i: 1);
3588 Value *Mask = CI->getArgOperand(i: 2);
3589 Rep = Builder.CreateAnd(LHS: Op0, RHS: Op1);
3590 llvm::Type *Ty = Op0->getType();
3591 Value *Zero = llvm::Constant::getNullValue(Ty);
3592 ICmpInst::Predicate Pred = Name.starts_with(Prefix: "avx512.ptestm")
3593 ? ICmpInst::ICMP_NE
3594 : ICmpInst::ICMP_EQ;
3595 Rep = Builder.CreateICmp(P: Pred, LHS: Rep, RHS: Zero);
3596 Rep = applyX86MaskOn1BitsVec(Builder, Vec: Rep, Mask);
3597 } else if (Name.starts_with(Prefix: "avx512.mask.pbroadcast")) {
3598 unsigned NumElts = cast<FixedVectorType>(Val: CI->getArgOperand(i: 1)->getType())
3599 ->getNumElements();
3600 Rep = Builder.CreateVectorSplat(NumElts, V: CI->getArgOperand(i: 0));
3601 Rep =
3602 emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep, Op1: CI->getArgOperand(i: 1));
3603 } else if (Name.starts_with(Prefix: "avx512.kunpck")) {
3604 unsigned NumElts = CI->getType()->getScalarSizeInBits();
3605 Value *LHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 0), NumElts);
3606 Value *RHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 1), NumElts);
3607 int Indices[64];
3608 for (unsigned i = 0; i != NumElts; ++i)
3609 Indices[i] = i;
3610
3611 // First extract half of each vector. This gives better codegen than
3612 // doing it in a single shuffle.
3613 LHS = Builder.CreateShuffleVector(V1: LHS, V2: LHS, Mask: ArrayRef(Indices, NumElts / 2));
3614 RHS = Builder.CreateShuffleVector(V1: RHS, V2: RHS, Mask: ArrayRef(Indices, NumElts / 2));
3615 // Concat the vectors.
3616 // NOTE: Operands have to be swapped to match intrinsic definition.
3617 Rep = Builder.CreateShuffleVector(V1: RHS, V2: LHS, Mask: ArrayRef(Indices, NumElts));
3618 Rep = Builder.CreateBitCast(V: Rep, DestTy: CI->getType());
3619 } else if (Name == "avx512.kand.w") {
3620 Value *LHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 0), NumElts: 16);
3621 Value *RHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 1), NumElts: 16);
3622 Rep = Builder.CreateAnd(LHS, RHS);
3623 Rep = Builder.CreateBitCast(V: Rep, DestTy: CI->getType());
3624 } else if (Name == "avx512.kandn.w") {
3625 Value *LHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 0), NumElts: 16);
3626 Value *RHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 1), NumElts: 16);
3627 LHS = Builder.CreateNot(V: LHS);
3628 Rep = Builder.CreateAnd(LHS, RHS);
3629 Rep = Builder.CreateBitCast(V: Rep, DestTy: CI->getType());
3630 } else if (Name == "avx512.kor.w") {
3631 Value *LHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 0), NumElts: 16);
3632 Value *RHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 1), NumElts: 16);
3633 Rep = Builder.CreateOr(LHS, RHS);
3634 Rep = Builder.CreateBitCast(V: Rep, DestTy: CI->getType());
3635 } else if (Name == "avx512.kxor.w") {
3636 Value *LHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 0), NumElts: 16);
3637 Value *RHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 1), NumElts: 16);
3638 Rep = Builder.CreateXor(LHS, RHS);
3639 Rep = Builder.CreateBitCast(V: Rep, DestTy: CI->getType());
3640 } else if (Name == "avx512.kxnor.w") {
3641 Value *LHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 0), NumElts: 16);
3642 Value *RHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 1), NumElts: 16);
3643 LHS = Builder.CreateNot(V: LHS);
3644 Rep = Builder.CreateXor(LHS, RHS);
3645 Rep = Builder.CreateBitCast(V: Rep, DestTy: CI->getType());
3646 } else if (Name == "avx512.knot.w") {
3647 Rep = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 0), NumElts: 16);
3648 Rep = Builder.CreateNot(V: Rep);
3649 Rep = Builder.CreateBitCast(V: Rep, DestTy: CI->getType());
3650 } else if (Name == "avx512.kortestz.w" || Name == "avx512.kortestc.w") {
3651 Value *LHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 0), NumElts: 16);
3652 Value *RHS = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 1), NumElts: 16);
3653 Rep = Builder.CreateOr(LHS, RHS);
3654 Rep = Builder.CreateBitCast(V: Rep, DestTy: Builder.getInt16Ty());
3655 Value *C;
3656 if (Name[14] == 'c')
3657 C = ConstantInt::getAllOnesValue(Ty: Builder.getInt16Ty());
3658 else
3659 C = ConstantInt::getNullValue(Ty: Builder.getInt16Ty());
3660 Rep = Builder.CreateICmpEQ(LHS: Rep, RHS: C);
3661 Rep = Builder.CreateZExt(V: Rep, DestTy: Builder.getInt32Ty());
3662 } else if (Name == "sse.add.ss" || Name == "sse2.add.sd" ||
3663 Name == "sse.sub.ss" || Name == "sse2.sub.sd" ||
3664 Name == "sse.mul.ss" || Name == "sse2.mul.sd" ||
3665 Name == "sse.div.ss" || Name == "sse2.div.sd") {
3666 Type *I32Ty = Type::getInt32Ty(C);
3667 Value *Elt0 = Builder.CreateExtractElement(Vec: CI->getArgOperand(i: 0),
3668 Idx: ConstantInt::get(Ty: I32Ty, V: 0));
3669 Value *Elt1 = Builder.CreateExtractElement(Vec: CI->getArgOperand(i: 1),
3670 Idx: ConstantInt::get(Ty: I32Ty, V: 0));
3671 Value *EltOp;
3672 if (Name.contains(Other: ".add."))
3673 EltOp = Builder.CreateFAdd(L: Elt0, R: Elt1);
3674 else if (Name.contains(Other: ".sub."))
3675 EltOp = Builder.CreateFSub(L: Elt0, R: Elt1);
3676 else if (Name.contains(Other: ".mul."))
3677 EltOp = Builder.CreateFMul(L: Elt0, R: Elt1);
3678 else
3679 EltOp = Builder.CreateFDiv(L: Elt0, R: Elt1);
3680 Rep = Builder.CreateInsertElement(Vec: CI->getArgOperand(i: 0), NewElt: EltOp,
3681 Idx: ConstantInt::get(Ty: I32Ty, V: 0));
3682 } else if (Name.starts_with(Prefix: "avx512.mask.pcmp")) {
3683 // "avx512.mask.pcmpeq." or "avx512.mask.pcmpgt."
3684 bool CmpEq = Name[16] == 'e';
3685 Rep = upgradeMaskedCompare(Builder, CI&: *CI, CC: CmpEq ? 0 : 6, Signed: true);
3686 } else if (Name.starts_with(Prefix: "avx512.mask.vpshufbitqmb.")) {
3687 Type *OpTy = CI->getArgOperand(i: 0)->getType();
3688 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3689 Intrinsic::ID IID;
3690 switch (VecWidth) {
3691 default:
3692 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
3693 break;
3694 case 128:
3695 IID = Intrinsic::x86_avx512_vpshufbitqmb_128;
3696 break;
3697 case 256:
3698 IID = Intrinsic::x86_avx512_vpshufbitqmb_256;
3699 break;
3700 case 512:
3701 IID = Intrinsic::x86_avx512_vpshufbitqmb_512;
3702 break;
3703 }
3704
3705 Rep =
3706 Builder.CreateIntrinsic(ID: IID, Args: {CI->getOperand(i_nocapture: 0), CI->getArgOperand(i: 1)});
3707 Rep = applyX86MaskOn1BitsVec(Builder, Vec: Rep, Mask: CI->getArgOperand(i: 2));
3708 } else if (Name.starts_with(Prefix: "avx512.mask.fpclass.p")) {
3709 Type *OpTy = CI->getArgOperand(i: 0)->getType();
3710 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3711 unsigned EltWidth = OpTy->getScalarSizeInBits();
3712 Intrinsic::ID IID;
3713 if (VecWidth == 128 && EltWidth == 32)
3714 IID = Intrinsic::x86_avx512_fpclass_ps_128;
3715 else if (VecWidth == 256 && EltWidth == 32)
3716 IID = Intrinsic::x86_avx512_fpclass_ps_256;
3717 else if (VecWidth == 512 && EltWidth == 32)
3718 IID = Intrinsic::x86_avx512_fpclass_ps_512;
3719 else if (VecWidth == 128 && EltWidth == 64)
3720 IID = Intrinsic::x86_avx512_fpclass_pd_128;
3721 else if (VecWidth == 256 && EltWidth == 64)
3722 IID = Intrinsic::x86_avx512_fpclass_pd_256;
3723 else if (VecWidth == 512 && EltWidth == 64)
3724 IID = Intrinsic::x86_avx512_fpclass_pd_512;
3725 else
3726 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
3727
3728 Rep =
3729 Builder.CreateIntrinsic(ID: IID, Args: {CI->getOperand(i_nocapture: 0), CI->getArgOperand(i: 1)});
3730 Rep = applyX86MaskOn1BitsVec(Builder, Vec: Rep, Mask: CI->getArgOperand(i: 2));
3731 } else if (Name.starts_with(Prefix: "avx512.cmp.p")) {
3732 SmallVector<Value *, 4> Args(CI->args());
3733 Type *OpTy = Args[0]->getType();
3734 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3735 unsigned EltWidth = OpTy->getScalarSizeInBits();
3736 Intrinsic::ID IID;
3737 if (VecWidth == 128 && EltWidth == 32)
3738 IID = Intrinsic::x86_avx512_mask_cmp_ps_128;
3739 else if (VecWidth == 256 && EltWidth == 32)
3740 IID = Intrinsic::x86_avx512_mask_cmp_ps_256;
3741 else if (VecWidth == 512 && EltWidth == 32)
3742 IID = Intrinsic::x86_avx512_mask_cmp_ps_512;
3743 else if (VecWidth == 128 && EltWidth == 64)
3744 IID = Intrinsic::x86_avx512_mask_cmp_pd_128;
3745 else if (VecWidth == 256 && EltWidth == 64)
3746 IID = Intrinsic::x86_avx512_mask_cmp_pd_256;
3747 else if (VecWidth == 512 && EltWidth == 64)
3748 IID = Intrinsic::x86_avx512_mask_cmp_pd_512;
3749 else
3750 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
3751
3752 Value *Mask = Constant::getAllOnesValue(Ty: CI->getType());
3753 if (VecWidth == 512)
3754 std::swap(a&: Mask, b&: Args.back());
3755 Args.push_back(Elt: Mask);
3756
3757 Rep = Builder.CreateIntrinsic(ID: IID, Args);
3758 } else if (Name.starts_with(Prefix: "avx512.mask.cmp.")) {
3759 // Integer compare intrinsics.
3760 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue();
3761 Rep = upgradeMaskedCompare(Builder, CI&: *CI, CC: Imm, Signed: true);
3762 } else if (Name.starts_with(Prefix: "avx512.mask.ucmp.")) {
3763 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue();
3764 Rep = upgradeMaskedCompare(Builder, CI&: *CI, CC: Imm, Signed: false);
3765 } else if (Name.starts_with(Prefix: "avx512.cvtb2mask.") ||
3766 Name.starts_with(Prefix: "avx512.cvtw2mask.") ||
3767 Name.starts_with(Prefix: "avx512.cvtd2mask.") ||
3768 Name.starts_with(Prefix: "avx512.cvtq2mask.")) {
3769 Value *Op = CI->getArgOperand(i: 0);
3770 Value *Zero = llvm::Constant::getNullValue(Ty: Op->getType());
3771 Rep = Builder.CreateICmp(P: ICmpInst::ICMP_SLT, LHS: Op, RHS: Zero);
3772 Rep = applyX86MaskOn1BitsVec(Builder, Vec: Rep, Mask: nullptr);
3773 } else if (Name == "ssse3.pabs.b.128" || Name == "ssse3.pabs.w.128" ||
3774 Name == "ssse3.pabs.d.128" || Name.starts_with(Prefix: "avx2.pabs") ||
3775 Name.starts_with(Prefix: "avx512.mask.pabs")) {
3776 Rep = upgradeAbs(Builder, CI&: *CI);
3777 } else if (Name == "sse41.pmaxsb" || Name == "sse2.pmaxs.w" ||
3778 Name == "sse41.pmaxsd" || Name.starts_with(Prefix: "avx2.pmaxs") ||
3779 Name.starts_with(Prefix: "avx512.mask.pmaxs")) {
3780 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::smax);
3781 } else if (Name == "sse2.pmaxu.b" || Name == "sse41.pmaxuw" ||
3782 Name == "sse41.pmaxud" || Name.starts_with(Prefix: "avx2.pmaxu") ||
3783 Name.starts_with(Prefix: "avx512.mask.pmaxu")) {
3784 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::umax);
3785 } else if (Name == "sse41.pminsb" || Name == "sse2.pmins.w" ||
3786 Name == "sse41.pminsd" || Name.starts_with(Prefix: "avx2.pmins") ||
3787 Name.starts_with(Prefix: "avx512.mask.pmins")) {
3788 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::smin);
3789 } else if (Name == "sse2.pminu.b" || Name == "sse41.pminuw" ||
3790 Name == "sse41.pminud" || Name.starts_with(Prefix: "avx2.pminu") ||
3791 Name.starts_with(Prefix: "avx512.mask.pminu")) {
3792 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::umin);
3793 } else if (Name == "sse2.pmulh.w" || Name.starts_with(Prefix: "avx2.pmulh.w") ||
3794 Name.starts_with(Prefix: "avx512.pmulh.w")) {
3795 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::smulh);
3796 } else if (Name == "sse2.pmulhu.w" || Name.starts_with(Prefix: "avx2.pmulhu.w") ||
3797 Name.starts_with(Prefix: "avx512.pmulhu.w")) {
3798 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::umulh);
3799 } else if (Name == "sse2.pmulu.dq" || Name == "avx2.pmulu.dq" ||
3800 Name == "avx512.pmulu.dq.512" ||
3801 Name.starts_with(Prefix: "avx512.mask.pmulu.dq.")) {
3802 Rep = upgradePMULDQ(Builder, CI&: *CI, /*Signed*/ IsSigned: false);
3803 } else if (Name == "sse41.pmuldq" || Name == "avx2.pmul.dq" ||
3804 Name == "avx512.pmul.dq.512" ||
3805 Name.starts_with(Prefix: "avx512.mask.pmul.dq.")) {
3806 Rep = upgradePMULDQ(Builder, CI&: *CI, /*Signed*/ IsSigned: true);
3807 } else if (Name == "sse.cvtsi2ss" || Name == "sse2.cvtsi2sd" ||
3808 Name == "sse.cvtsi642ss" || Name == "sse2.cvtsi642sd") {
3809 Rep =
3810 Builder.CreateSIToFP(V: CI->getArgOperand(i: 1),
3811 DestTy: cast<VectorType>(Val: CI->getType())->getElementType());
3812 Rep = Builder.CreateInsertElement(Vec: CI->getArgOperand(i: 0), NewElt: Rep, Idx: (uint64_t)0);
3813 } else if (Name == "avx512.cvtusi2sd") {
3814 Rep =
3815 Builder.CreateUIToFP(V: CI->getArgOperand(i: 1),
3816 DestTy: cast<VectorType>(Val: CI->getType())->getElementType());
3817 Rep = Builder.CreateInsertElement(Vec: CI->getArgOperand(i: 0), NewElt: Rep, Idx: (uint64_t)0);
3818 } else if (Name == "sse2.cvtss2sd") {
3819 Rep = Builder.CreateExtractElement(Vec: CI->getArgOperand(i: 1), Idx: (uint64_t)0);
3820 Rep = Builder.CreateFPExt(
3821 V: Rep, DestTy: cast<VectorType>(Val: CI->getType())->getElementType());
3822 Rep = Builder.CreateInsertElement(Vec: CI->getArgOperand(i: 0), NewElt: Rep, Idx: (uint64_t)0);
3823 } else if (Name == "sse2.cvtdq2pd" || Name == "sse2.cvtdq2ps" ||
3824 Name == "avx.cvtdq2.pd.256" || Name == "avx.cvtdq2.ps.256" ||
3825 Name.starts_with(Prefix: "avx512.mask.cvtdq2pd.") ||
3826 Name.starts_with(Prefix: "avx512.mask.cvtudq2pd.") ||
3827 Name.starts_with(Prefix: "avx512.mask.cvtdq2ps.") ||
3828 Name.starts_with(Prefix: "avx512.mask.cvtudq2ps.") ||
3829 Name.starts_with(Prefix: "avx512.mask.cvtqq2pd.") ||
3830 Name.starts_with(Prefix: "avx512.mask.cvtuqq2pd.") ||
3831 Name == "avx512.mask.cvtqq2ps.256" ||
3832 Name == "avx512.mask.cvtqq2ps.512" ||
3833 Name == "avx512.mask.cvtuqq2ps.256" ||
3834 Name == "avx512.mask.cvtuqq2ps.512" || Name == "sse2.cvtps2pd" ||
3835 Name == "avx.cvt.ps2.pd.256" ||
3836 Name == "avx512.mask.cvtps2pd.128" ||
3837 Name == "avx512.mask.cvtps2pd.256") {
3838 auto *DstTy = cast<FixedVectorType>(Val: CI->getType());
3839 Rep = CI->getArgOperand(i: 0);
3840 auto *SrcTy = cast<FixedVectorType>(Val: Rep->getType());
3841
3842 unsigned NumDstElts = DstTy->getNumElements();
3843 if (NumDstElts < SrcTy->getNumElements()) {
3844 assert(NumDstElts == 2 && "Unexpected vector size");
3845 Rep = Builder.CreateShuffleVector(V1: Rep, V2: Rep, Mask: ArrayRef<int>{0, 1});
3846 }
3847
3848 bool IsPS2PD = SrcTy->getElementType()->isFloatTy();
3849 bool IsUnsigned = Name.contains(Other: "cvtu");
3850 if (IsPS2PD)
3851 Rep = Builder.CreateFPExt(V: Rep, DestTy: DstTy, Name: "cvtps2pd");
3852 else if (CI->arg_size() == 4 &&
3853 (!isa<ConstantInt>(Val: CI->getArgOperand(i: 3)) ||
3854 cast<ConstantInt>(Val: CI->getArgOperand(i: 3))->getZExtValue() != 4)) {
3855 Intrinsic::ID IID = IsUnsigned ? Intrinsic::x86_avx512_uitofp_round
3856 : Intrinsic::x86_avx512_sitofp_round;
3857 Rep = Builder.CreateIntrinsic(ID: IID, OverloadTypes: {DstTy, SrcTy},
3858 Args: {Rep, CI->getArgOperand(i: 3)});
3859 } else {
3860 Rep = IsUnsigned ? Builder.CreateUIToFP(V: Rep, DestTy: DstTy, Name: "cvt")
3861 : Builder.CreateSIToFP(V: Rep, DestTy: DstTy, Name: "cvt");
3862 }
3863
3864 if (CI->arg_size() >= 3)
3865 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep,
3866 Op1: CI->getArgOperand(i: 1));
3867 } else if (Name.starts_with(Prefix: "avx512.mask.vcvtph2ps.") ||
3868 Name.starts_with(Prefix: "vcvtph2ps.")) {
3869 auto *DstTy = cast<FixedVectorType>(Val: CI->getType());
3870 Rep = CI->getArgOperand(i: 0);
3871 auto *SrcTy = cast<FixedVectorType>(Val: Rep->getType());
3872 unsigned NumDstElts = DstTy->getNumElements();
3873 if (NumDstElts != SrcTy->getNumElements()) {
3874 assert(NumDstElts == 4 && "Unexpected vector size");
3875 Rep = Builder.CreateShuffleVector(V1: Rep, V2: Rep, Mask: ArrayRef<int>{0, 1, 2, 3});
3876 }
3877 Rep = Builder.CreateBitCast(
3878 V: Rep, DestTy: FixedVectorType::get(ElementType: Type::getHalfTy(C), NumElts: NumDstElts));
3879 Rep = Builder.CreateFPExt(V: Rep, DestTy: DstTy, Name: "cvtph2ps");
3880 if (CI->arg_size() >= 3)
3881 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep,
3882 Op1: CI->getArgOperand(i: 1));
3883 } else if (Name.starts_with(Prefix: "avx512.mask.load")) {
3884 // "avx512.mask.loadu." or "avx512.mask.load."
3885 bool Aligned = Name[16] != 'u'; // "avx512.mask.loadu".
3886 Rep = upgradeMaskedLoad(Builder, Ptr: CI->getArgOperand(i: 0), Passthru: CI->getArgOperand(i: 1),
3887 Mask: CI->getArgOperand(i: 2), Aligned);
3888 } else if (Name.starts_with(Prefix: "avx512.mask.expand.load.")) {
3889 auto *ResultTy = cast<FixedVectorType>(Val: CI->getType());
3890 auto *PtrTy = CI->getOperand(i_nocapture: 0)->getType();
3891 Value *MaskVec = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 2),
3892 NumElts: ResultTy->getNumElements());
3893 Rep = Builder.CreateIntrinsic(
3894 ID: Intrinsic::masked_expandload, OverloadTypes: {ResultTy, PtrTy},
3895 Args: {CI->getOperand(i_nocapture: 0), MaskVec, CI->getOperand(i_nocapture: 1)});
3896 } else if (Name.starts_with(Prefix: "avx512.mask.compress.store.")) {
3897 auto *ResultTy = cast<VectorType>(Val: CI->getArgOperand(i: 1)->getType());
3898 auto *PtrTy = CI->getArgOperand(i: 0)->getType();
3899 Value *MaskVec =
3900 getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 2),
3901 NumElts: cast<FixedVectorType>(Val: ResultTy)->getNumElements());
3902 Rep = Builder.CreateIntrinsic(
3903 ID: Intrinsic::masked_compressstore, OverloadTypes: {ResultTy, PtrTy},
3904 Args: {CI->getArgOperand(i: 1), CI->getArgOperand(i: 0), MaskVec});
3905 } else if (Name.starts_with(Prefix: "avx512.mask.compress.") ||
3906 Name.starts_with(Prefix: "avx512.mask.expand.")) {
3907 auto *ResultTy = cast<FixedVectorType>(Val: CI->getType());
3908
3909 Value *MaskVec = getX86MaskVec(Builder, Mask: CI->getArgOperand(i: 2),
3910 NumElts: ResultTy->getNumElements());
3911
3912 bool IsCompress = Name[12] == 'c';
3913 Intrinsic::ID IID = IsCompress ? Intrinsic::x86_avx512_mask_compress
3914 : Intrinsic::x86_avx512_mask_expand;
3915 Rep = Builder.CreateIntrinsic(
3916 ID: IID, OverloadTypes: ResultTy, Args: {CI->getOperand(i_nocapture: 0), CI->getOperand(i_nocapture: 1), MaskVec});
3917 } else if (Name.starts_with(Prefix: "xop.vpcom")) {
3918 bool IsSigned;
3919 if (Name.ends_with(Suffix: "ub") || Name.ends_with(Suffix: "uw") || Name.ends_with(Suffix: "ud") ||
3920 Name.ends_with(Suffix: "uq"))
3921 IsSigned = false;
3922 else if (Name.ends_with(Suffix: "b") || Name.ends_with(Suffix: "w") ||
3923 Name.ends_with(Suffix: "d") || Name.ends_with(Suffix: "q"))
3924 IsSigned = true;
3925 else
3926 reportFatalUsageErrorWithCI(reason: "Intrinsic has unknown suffix", CI);
3927
3928 unsigned Imm;
3929 if (CI->arg_size() == 3) {
3930 Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue();
3931 } else {
3932 Name = Name.substr(Start: 9); // strip off "xop.vpcom"
3933 if (Name.starts_with(Prefix: "lt"))
3934 Imm = 0;
3935 else if (Name.starts_with(Prefix: "le"))
3936 Imm = 1;
3937 else if (Name.starts_with(Prefix: "gt"))
3938 Imm = 2;
3939 else if (Name.starts_with(Prefix: "ge"))
3940 Imm = 3;
3941 else if (Name.starts_with(Prefix: "eq"))
3942 Imm = 4;
3943 else if (Name.starts_with(Prefix: "ne"))
3944 Imm = 5;
3945 else if (Name.starts_with(Prefix: "false"))
3946 Imm = 6;
3947 else if (Name.starts_with(Prefix: "true"))
3948 Imm = 7;
3949 else
3950 llvm_unreachable("Unknown condition");
3951 }
3952
3953 Rep = upgradeX86vpcom(Builder, CI&: *CI, Imm, IsSigned);
3954 } else if (Name.starts_with(Prefix: "xop.vpcmov")) {
3955 Value *Sel = CI->getArgOperand(i: 2);
3956 Value *NotSel = Builder.CreateNot(V: Sel);
3957 Value *Sel0 = Builder.CreateAnd(LHS: CI->getArgOperand(i: 0), RHS: Sel);
3958 Value *Sel1 = Builder.CreateAnd(LHS: CI->getArgOperand(i: 1), RHS: NotSel);
3959 Rep = Builder.CreateOr(LHS: Sel0, RHS: Sel1);
3960 } else if (Name.starts_with(Prefix: "xop.vprot") || Name.starts_with(Prefix: "avx512.prol") ||
3961 Name.starts_with(Prefix: "avx512.mask.prol")) {
3962 Rep = upgradeX86Rotate(Builder, CI&: *CI, IsRotateRight: false);
3963 } else if (Name.starts_with(Prefix: "avx512.pror") ||
3964 Name.starts_with(Prefix: "avx512.mask.pror")) {
3965 Rep = upgradeX86Rotate(Builder, CI&: *CI, IsRotateRight: true);
3966 } else if (Name.starts_with(Prefix: "avx512.vpshld.") ||
3967 Name.starts_with(Prefix: "avx512.mask.vpshld") ||
3968 Name.starts_with(Prefix: "avx512.maskz.vpshld")) {
3969 bool ZeroMask = Name[11] == 'z';
3970 Rep = upgradeX86ConcatShift(Builder, CI&: *CI, IsShiftRight: false, ZeroMask);
3971 } else if (Name.starts_with(Prefix: "avx512.vpshrd.") ||
3972 Name.starts_with(Prefix: "avx512.mask.vpshrd") ||
3973 Name.starts_with(Prefix: "avx512.maskz.vpshrd")) {
3974 bool ZeroMask = Name[11] == 'z';
3975 Rep = upgradeX86ConcatShift(Builder, CI&: *CI, IsShiftRight: true, ZeroMask);
3976 } else if (Name == "sse42.crc32.64.8") {
3977 Value *Trunc0 =
3978 Builder.CreateTrunc(V: CI->getArgOperand(i: 0), DestTy: Type::getInt32Ty(C));
3979 Rep = Builder.CreateIntrinsic(ID: Intrinsic::x86_sse42_crc32_32_8,
3980 Args: {Trunc0, CI->getArgOperand(i: 1)});
3981 Rep = Builder.CreateZExt(V: Rep, DestTy: CI->getType(), Name: "");
3982 } else if (Name.starts_with(Prefix: "avx.vbroadcast.s") ||
3983 Name.starts_with(Prefix: "avx512.vbroadcast.s")) {
3984 // Replace broadcasts with a series of insertelements.
3985 auto *VecTy = cast<FixedVectorType>(Val: CI->getType());
3986 Type *EltTy = VecTy->getElementType();
3987 unsigned EltNum = VecTy->getNumElements();
3988 Value *Load = Builder.CreateLoad(Ty: EltTy, Ptr: CI->getArgOperand(i: 0));
3989 Type *I32Ty = Type::getInt32Ty(C);
3990 Rep = PoisonValue::get(T: VecTy);
3991 for (unsigned I = 0; I < EltNum; ++I)
3992 Rep = Builder.CreateInsertElement(Vec: Rep, NewElt: Load, Idx: ConstantInt::get(Ty: I32Ty, V: I));
3993 } else if (Name.starts_with(Prefix: "sse41.pmovsx") ||
3994 Name.starts_with(Prefix: "sse41.pmovzx") ||
3995 Name.starts_with(Prefix: "avx2.pmovsx") ||
3996 Name.starts_with(Prefix: "avx2.pmovzx") ||
3997 Name.starts_with(Prefix: "avx512.mask.pmovsx") ||
3998 Name.starts_with(Prefix: "avx512.mask.pmovzx")) {
3999 auto *DstTy = cast<FixedVectorType>(Val: CI->getType());
4000 unsigned NumDstElts = DstTy->getNumElements();
4001
4002 // Extract a subvector of the first NumDstElts lanes and sign/zero extend.
4003 SmallVector<int, 8> ShuffleMask(NumDstElts);
4004 for (unsigned i = 0; i != NumDstElts; ++i)
4005 ShuffleMask[i] = i;
4006
4007 Value *SV = Builder.CreateShuffleVector(V: CI->getArgOperand(i: 0), Mask: ShuffleMask);
4008
4009 bool DoSext = Name.contains(Other: "pmovsx");
4010 Rep =
4011 DoSext ? Builder.CreateSExt(V: SV, DestTy: DstTy) : Builder.CreateZExt(V: SV, DestTy: DstTy);
4012 // If there are 3 arguments, it's a masked intrinsic so we need a select.
4013 if (CI->arg_size() == 3)
4014 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep,
4015 Op1: CI->getArgOperand(i: 1));
4016 } else if (Name == "avx512.mask.pmov.qd.256" ||
4017 Name == "avx512.mask.pmov.qd.512" ||
4018 Name == "avx512.mask.pmov.wb.256" ||
4019 Name == "avx512.mask.pmov.wb.512") {
4020 Type *Ty = CI->getArgOperand(i: 1)->getType();
4021 Rep = Builder.CreateTrunc(V: CI->getArgOperand(i: 0), DestTy: Ty);
4022 Rep =
4023 emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep, Op1: CI->getArgOperand(i: 1));
4024 } else if (Name.starts_with(Prefix: "avx.vbroadcastf128") ||
4025 Name == "avx2.vbroadcasti128") {
4026 // Replace vbroadcastf128/vbroadcasti128 with a vector load+shuffle.
4027 Type *EltTy = cast<VectorType>(Val: CI->getType())->getElementType();
4028 unsigned NumSrcElts = 128 / EltTy->getPrimitiveSizeInBits();
4029 auto *VT = FixedVectorType::get(ElementType: EltTy, NumElts: NumSrcElts);
4030 Value *Load = Builder.CreateAlignedLoad(Ty: VT, Ptr: CI->getArgOperand(i: 0), Align: Align(1));
4031 if (NumSrcElts == 2)
4032 Rep = Builder.CreateShuffleVector(V: Load, Mask: ArrayRef<int>{0, 1, 0, 1});
4033 else
4034 Rep = Builder.CreateShuffleVector(V: Load,
4035 Mask: ArrayRef<int>{0, 1, 2, 3, 0, 1, 2, 3});
4036 } else if (Name.starts_with(Prefix: "avx512.mask.shuf.i") ||
4037 Name.starts_with(Prefix: "avx512.mask.shuf.f")) {
4038 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue();
4039 Type *VT = CI->getType();
4040 unsigned NumLanes = VT->getPrimitiveSizeInBits() / 128;
4041 unsigned NumElementsInLane = 128 / VT->getScalarSizeInBits();
4042 unsigned ControlBitsMask = NumLanes - 1;
4043 unsigned NumControlBits = NumLanes / 2;
4044 SmallVector<int, 8> ShuffleMask(0);
4045
4046 for (unsigned l = 0; l != NumLanes; ++l) {
4047 unsigned LaneMask = (Imm >> (l * NumControlBits)) & ControlBitsMask;
4048 // We actually need the other source.
4049 if (l >= NumLanes / 2)
4050 LaneMask += NumLanes;
4051 for (unsigned i = 0; i != NumElementsInLane; ++i)
4052 ShuffleMask.push_back(Elt: LaneMask * NumElementsInLane + i);
4053 }
4054 Rep = Builder.CreateShuffleVector(V1: CI->getArgOperand(i: 0),
4055 V2: CI->getArgOperand(i: 1), Mask: ShuffleMask);
4056 Rep =
4057 emitX86Select(Builder, Mask: CI->getArgOperand(i: 4), Op0: Rep, Op1: CI->getArgOperand(i: 3));
4058 } else if (Name.starts_with(Prefix: "avx512.mask.broadcastf") ||
4059 Name.starts_with(Prefix: "avx512.mask.broadcasti")) {
4060 unsigned NumSrcElts = cast<FixedVectorType>(Val: CI->getArgOperand(i: 0)->getType())
4061 ->getNumElements();
4062 unsigned NumDstElts =
4063 cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4064
4065 SmallVector<int, 8> ShuffleMask(NumDstElts);
4066 for (unsigned i = 0; i != NumDstElts; ++i)
4067 ShuffleMask[i] = i % NumSrcElts;
4068
4069 Rep = Builder.CreateShuffleVector(V1: CI->getArgOperand(i: 0),
4070 V2: CI->getArgOperand(i: 0), Mask: ShuffleMask);
4071 Rep =
4072 emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep, Op1: CI->getArgOperand(i: 1));
4073 } else if (Name.starts_with(Prefix: "avx2.pbroadcast") ||
4074 Name.starts_with(Prefix: "avx2.vbroadcast") ||
4075 Name.starts_with(Prefix: "avx512.pbroadcast") ||
4076 Name.starts_with(Prefix: "avx512.mask.broadcast.s")) {
4077 // Replace vp?broadcasts with a vector shuffle.
4078 Value *Op = CI->getArgOperand(i: 0);
4079 ElementCount EC = cast<VectorType>(Val: CI->getType())->getElementCount();
4080 Type *MaskTy = VectorType::get(ElementType: Type::getInt32Ty(C), EC);
4081 SmallVector<int, 8> M;
4082 ShuffleVectorInst::getShuffleMask(Mask: Constant::getNullValue(Ty: MaskTy), Result&: M);
4083 Rep = Builder.CreateShuffleVector(V: Op, Mask: M);
4084
4085 if (CI->arg_size() == 3)
4086 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep,
4087 Op1: CI->getArgOperand(i: 1));
4088 } else if (Name.starts_with(Prefix: "sse2.padds.") ||
4089 Name.starts_with(Prefix: "avx2.padds.") ||
4090 Name.starts_with(Prefix: "avx512.padds.") ||
4091 Name.starts_with(Prefix: "avx512.mask.padds.")) {
4092 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::sadd_sat);
4093 } else if (Name.starts_with(Prefix: "sse2.psubs.") ||
4094 Name.starts_with(Prefix: "avx2.psubs.") ||
4095 Name.starts_with(Prefix: "avx512.psubs.") ||
4096 Name.starts_with(Prefix: "avx512.mask.psubs.")) {
4097 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::ssub_sat);
4098 } else if (Name.starts_with(Prefix: "sse2.paddus.") ||
4099 Name.starts_with(Prefix: "avx2.paddus.") ||
4100 Name.starts_with(Prefix: "avx512.mask.paddus.")) {
4101 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::uadd_sat);
4102 } else if (Name.starts_with(Prefix: "sse2.psubus.") ||
4103 Name.starts_with(Prefix: "avx2.psubus.") ||
4104 Name.starts_with(Prefix: "avx512.mask.psubus.")) {
4105 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::usub_sat);
4106 } else if (Name.starts_with(Prefix: "avx512.mask.palignr.")) {
4107 Rep = upgradeX86ALIGNIntrinsics(Builder, Op0: CI->getArgOperand(i: 0),
4108 Op1: CI->getArgOperand(i: 1), Shift: CI->getArgOperand(i: 2),
4109 Passthru: CI->getArgOperand(i: 3), Mask: CI->getArgOperand(i: 4),
4110 IsVALIGN: false);
4111 } else if (Name.starts_with(Prefix: "avx512.mask.valign.")) {
4112 Rep = upgradeX86ALIGNIntrinsics(
4113 Builder, Op0: CI->getArgOperand(i: 0), Op1: CI->getArgOperand(i: 1),
4114 Shift: CI->getArgOperand(i: 2), Passthru: CI->getArgOperand(i: 3), Mask: CI->getArgOperand(i: 4), IsVALIGN: true);
4115 } else if (Name == "sse2.psll.dq" || Name == "avx2.psll.dq") {
4116 // 128/256-bit shift left specified in bits.
4117 unsigned Shift = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4118 Rep = upgradeX86PSLLDQIntrinsics(Builder, Op: CI->getArgOperand(i: 0),
4119 Shift: Shift / 8); // Shift is in bits.
4120 } else if (Name == "sse2.psrl.dq" || Name == "avx2.psrl.dq") {
4121 // 128/256-bit shift right specified in bits.
4122 unsigned Shift = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4123 Rep = upgradeX86PSRLDQIntrinsics(Builder, Op: CI->getArgOperand(i: 0),
4124 Shift: Shift / 8); // Shift is in bits.
4125 } else if (Name == "sse2.psll.dq.bs" || Name == "avx2.psll.dq.bs" ||
4126 Name == "avx512.psll.dq.512") {
4127 // 128/256/512-bit shift left specified in bytes.
4128 unsigned Shift = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4129 Rep = upgradeX86PSLLDQIntrinsics(Builder, Op: CI->getArgOperand(i: 0), Shift);
4130 } else if (Name == "sse2.psrl.dq.bs" || Name == "avx2.psrl.dq.bs" ||
4131 Name == "avx512.psrl.dq.512") {
4132 // 128/256/512-bit shift right specified in bytes.
4133 unsigned Shift = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4134 Rep = upgradeX86PSRLDQIntrinsics(Builder, Op: CI->getArgOperand(i: 0), Shift);
4135 } else if (Name == "sse41.pblendw" || Name.starts_with(Prefix: "sse41.blendp") ||
4136 Name.starts_with(Prefix: "avx.blend.p") || Name == "avx2.pblendw" ||
4137 Name.starts_with(Prefix: "avx2.pblendd.")) {
4138 Value *Op0 = CI->getArgOperand(i: 0);
4139 Value *Op1 = CI->getArgOperand(i: 1);
4140 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue();
4141 auto *VecTy = cast<FixedVectorType>(Val: CI->getType());
4142 unsigned NumElts = VecTy->getNumElements();
4143
4144 SmallVector<int, 16> Idxs(NumElts);
4145 for (unsigned i = 0; i != NumElts; ++i)
4146 Idxs[i] = ((Imm >> (i % 8)) & 1) ? i + NumElts : i;
4147
4148 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op1, Mask: Idxs);
4149 } else if (Name.starts_with(Prefix: "avx.vinsertf128.") ||
4150 Name == "avx2.vinserti128" ||
4151 Name.starts_with(Prefix: "avx512.mask.insert")) {
4152 Value *Op0 = CI->getArgOperand(i: 0);
4153 Value *Op1 = CI->getArgOperand(i: 1);
4154 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue();
4155 unsigned DstNumElts =
4156 cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4157 unsigned SrcNumElts =
4158 cast<FixedVectorType>(Val: Op1->getType())->getNumElements();
4159 unsigned Scale = DstNumElts / SrcNumElts;
4160
4161 // Mask off the high bits of the immediate value; hardware ignores those.
4162 Imm = Imm % Scale;
4163
4164 // Extend the second operand into a vector the size of the destination.
4165 SmallVector<int, 8> Idxs(DstNumElts);
4166 for (unsigned i = 0; i != SrcNumElts; ++i)
4167 Idxs[i] = i;
4168 for (unsigned i = SrcNumElts; i != DstNumElts; ++i)
4169 Idxs[i] = SrcNumElts;
4170 Rep = Builder.CreateShuffleVector(V: Op1, Mask: Idxs);
4171
4172 // Insert the second operand into the first operand.
4173
4174 // Note that there is no guarantee that instruction lowering will actually
4175 // produce a vinsertf128 instruction for the created shuffles. In
4176 // particular, the 0 immediate case involves no lane changes, so it can
4177 // be handled as a blend.
4178
4179 // Example of shuffle mask for 32-bit elements:
4180 // Imm = 1 <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
4181 // Imm = 0 <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7 >
4182
4183 // First fill with identify mask.
4184 for (unsigned i = 0; i != DstNumElts; ++i)
4185 Idxs[i] = i;
4186 // Then replace the elements where we need to insert.
4187 for (unsigned i = 0; i != SrcNumElts; ++i)
4188 Idxs[i + Imm * SrcNumElts] = i + DstNumElts;
4189 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Rep, Mask: Idxs);
4190
4191 // If the intrinsic has a mask operand, handle that.
4192 if (CI->arg_size() == 5)
4193 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 4), Op0: Rep,
4194 Op1: CI->getArgOperand(i: 3));
4195 } else if (Name.starts_with(Prefix: "avx.vextractf128.") ||
4196 Name == "avx2.vextracti128" ||
4197 Name.starts_with(Prefix: "avx512.mask.vextract")) {
4198 Value *Op0 = CI->getArgOperand(i: 0);
4199 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4200 unsigned DstNumElts =
4201 cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4202 unsigned SrcNumElts =
4203 cast<FixedVectorType>(Val: Op0->getType())->getNumElements();
4204 unsigned Scale = SrcNumElts / DstNumElts;
4205
4206 // Mask off the high bits of the immediate value; hardware ignores those.
4207 Imm = Imm % Scale;
4208
4209 // Get indexes for the subvector of the input vector.
4210 SmallVector<int, 8> Idxs(DstNumElts);
4211 for (unsigned i = 0; i != DstNumElts; ++i) {
4212 Idxs[i] = i + (Imm * DstNumElts);
4213 }
4214 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op0, Mask: Idxs);
4215
4216 // If the intrinsic has a mask operand, handle that.
4217 if (CI->arg_size() == 4)
4218 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep,
4219 Op1: CI->getArgOperand(i: 2));
4220 } else if (Name.starts_with(Prefix: "avx512.mask.perm.df.") ||
4221 Name.starts_with(Prefix: "avx512.mask.perm.di.")) {
4222 Value *Op0 = CI->getArgOperand(i: 0);
4223 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4224 auto *VecTy = cast<FixedVectorType>(Val: CI->getType());
4225 unsigned NumElts = VecTy->getNumElements();
4226
4227 SmallVector<int, 8> Idxs(NumElts);
4228 for (unsigned i = 0; i != NumElts; ++i)
4229 Idxs[i] = (i & ~0x3) + ((Imm >> (2 * (i & 0x3))) & 3);
4230
4231 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op0, Mask: Idxs);
4232
4233 if (CI->arg_size() == 4)
4234 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep,
4235 Op1: CI->getArgOperand(i: 2));
4236 } else if (Name.starts_with(Prefix: "avx.vperm2f128.") || Name == "avx2.vperm2i128") {
4237 // The immediate permute control byte looks like this:
4238 // [1:0] - select 128 bits from sources for low half of destination
4239 // [2] - ignore
4240 // [3] - zero low half of destination
4241 // [5:4] - select 128 bits from sources for high half of destination
4242 // [6] - ignore
4243 // [7] - zero high half of destination
4244
4245 uint8_t Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue();
4246
4247 unsigned NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4248 unsigned HalfSize = NumElts / 2;
4249 SmallVector<int, 8> ShuffleMask(NumElts);
4250
4251 // Determine which operand(s) are actually in use for this instruction.
4252 Value *V0 = (Imm & 0x02) ? CI->getArgOperand(i: 1) : CI->getArgOperand(i: 0);
4253 Value *V1 = (Imm & 0x20) ? CI->getArgOperand(i: 1) : CI->getArgOperand(i: 0);
4254
4255 // If needed, replace operands based on zero mask.
4256 V0 = (Imm & 0x08) ? ConstantAggregateZero::get(Ty: CI->getType()) : V0;
4257 V1 = (Imm & 0x80) ? ConstantAggregateZero::get(Ty: CI->getType()) : V1;
4258
4259 // Permute low half of result.
4260 unsigned StartIndex = (Imm & 0x01) ? HalfSize : 0;
4261 for (unsigned i = 0; i < HalfSize; ++i)
4262 ShuffleMask[i] = StartIndex + i;
4263
4264 // Permute high half of result.
4265 StartIndex = (Imm & 0x10) ? HalfSize : 0;
4266 for (unsigned i = 0; i < HalfSize; ++i)
4267 ShuffleMask[i + HalfSize] = NumElts + StartIndex + i;
4268
4269 Rep = Builder.CreateShuffleVector(V1: V0, V2: V1, Mask: ShuffleMask);
4270
4271 } else if (Name.starts_with(Prefix: "avx.vpermil.") || Name == "sse2.pshuf.d" ||
4272 Name.starts_with(Prefix: "avx512.mask.vpermil.p") ||
4273 Name.starts_with(Prefix: "avx512.mask.pshuf.d.")) {
4274 Value *Op0 = CI->getArgOperand(i: 0);
4275 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4276 auto *VecTy = cast<FixedVectorType>(Val: CI->getType());
4277 unsigned NumElts = VecTy->getNumElements();
4278 // Calculate the size of each index in the immediate.
4279 unsigned IdxSize = 64 / VecTy->getScalarSizeInBits();
4280 unsigned IdxMask = ((1 << IdxSize) - 1);
4281
4282 SmallVector<int, 8> Idxs(NumElts);
4283 // Lookup the bits for this element, wrapping around the immediate every
4284 // 8-bits. Elements are grouped into sets of 2 or 4 elements so we need
4285 // to offset by the first index of each group.
4286 for (unsigned i = 0; i != NumElts; ++i)
4287 Idxs[i] = ((Imm >> ((i * IdxSize) % 8)) & IdxMask) | (i & ~IdxMask);
4288
4289 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op0, Mask: Idxs);
4290
4291 if (CI->arg_size() == 4)
4292 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep,
4293 Op1: CI->getArgOperand(i: 2));
4294 } else if (Name == "sse2.pshufl.w" ||
4295 Name.starts_with(Prefix: "avx512.mask.pshufl.w.")) {
4296 Value *Op0 = CI->getArgOperand(i: 0);
4297 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4298 unsigned NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4299
4300 if (Name == "sse2.pshufl.w" && NumElts % 8 != 0)
4301 reportFatalUsageErrorWithCI(reason: "Intrinsic has invalid signature", CI);
4302
4303 SmallVector<int, 16> Idxs(NumElts);
4304 for (unsigned l = 0; l != NumElts; l += 8) {
4305 for (unsigned i = 0; i != 4; ++i)
4306 Idxs[i + l] = ((Imm >> (2 * i)) & 0x3) + l;
4307 for (unsigned i = 4; i != 8; ++i)
4308 Idxs[i + l] = i + l;
4309 }
4310
4311 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op0, Mask: Idxs);
4312
4313 if (CI->arg_size() == 4)
4314 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep,
4315 Op1: CI->getArgOperand(i: 2));
4316 } else if (Name == "sse2.pshufh.w" ||
4317 Name.starts_with(Prefix: "avx512.mask.pshufh.w.")) {
4318 Value *Op0 = CI->getArgOperand(i: 0);
4319 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
4320 unsigned NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4321
4322 if (Name == "sse2.pshufh.w" && NumElts % 8 != 0)
4323 reportFatalUsageErrorWithCI(reason: "Intrinsic has invalid signature", CI);
4324
4325 SmallVector<int, 16> Idxs(NumElts);
4326 for (unsigned l = 0; l != NumElts; l += 8) {
4327 for (unsigned i = 0; i != 4; ++i)
4328 Idxs[i + l] = i + l;
4329 for (unsigned i = 0; i != 4; ++i)
4330 Idxs[i + l + 4] = ((Imm >> (2 * i)) & 0x3) + 4 + l;
4331 }
4332
4333 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op0, Mask: Idxs);
4334
4335 if (CI->arg_size() == 4)
4336 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep,
4337 Op1: CI->getArgOperand(i: 2));
4338 } else if (Name.starts_with(Prefix: "avx512.mask.shuf.p")) {
4339 Value *Op0 = CI->getArgOperand(i: 0);
4340 Value *Op1 = CI->getArgOperand(i: 1);
4341 unsigned Imm = cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue();
4342 unsigned NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4343
4344 unsigned NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4345 unsigned HalfLaneElts = NumLaneElts / 2;
4346
4347 SmallVector<int, 16> Idxs(NumElts);
4348 for (unsigned i = 0; i != NumElts; ++i) {
4349 // Base index is the starting element of the lane.
4350 Idxs[i] = i - (i % NumLaneElts);
4351 // If we are half way through the lane switch to the other source.
4352 if ((i % NumLaneElts) >= HalfLaneElts)
4353 Idxs[i] += NumElts;
4354 // Now select the specific element. By adding HalfLaneElts bits from
4355 // the immediate. Wrapping around the immediate every 8-bits.
4356 Idxs[i] += (Imm >> ((i * HalfLaneElts) % 8)) & ((1 << HalfLaneElts) - 1);
4357 }
4358
4359 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op1, Mask: Idxs);
4360
4361 Rep =
4362 emitX86Select(Builder, Mask: CI->getArgOperand(i: 4), Op0: Rep, Op1: CI->getArgOperand(i: 3));
4363 } else if (Name.starts_with(Prefix: "avx512.mask.movddup") ||
4364 Name.starts_with(Prefix: "avx512.mask.movshdup") ||
4365 Name.starts_with(Prefix: "avx512.mask.movsldup")) {
4366 Value *Op0 = CI->getArgOperand(i: 0);
4367 unsigned NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4368 unsigned NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4369
4370 unsigned Offset = 0;
4371 if (Name.starts_with(Prefix: "avx512.mask.movshdup."))
4372 Offset = 1;
4373
4374 SmallVector<int, 16> Idxs(NumElts);
4375 for (unsigned l = 0; l != NumElts; l += NumLaneElts)
4376 for (unsigned i = 0; i != NumLaneElts; i += 2) {
4377 Idxs[i + l + 0] = i + l + Offset;
4378 Idxs[i + l + 1] = i + l + Offset;
4379 }
4380
4381 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op0, Mask: Idxs);
4382
4383 Rep =
4384 emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep, Op1: CI->getArgOperand(i: 1));
4385 } else if (Name.starts_with(Prefix: "avx512.mask.punpckl") ||
4386 Name.starts_with(Prefix: "avx512.mask.unpckl.")) {
4387 Value *Op0 = CI->getArgOperand(i: 0);
4388 Value *Op1 = CI->getArgOperand(i: 1);
4389 int NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4390 int NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4391
4392 SmallVector<int, 64> Idxs(NumElts);
4393 for (int l = 0; l != NumElts; l += NumLaneElts)
4394 for (int i = 0; i != NumLaneElts; ++i)
4395 Idxs[i + l] = l + (i / 2) + NumElts * (i % 2);
4396
4397 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op1, Mask: Idxs);
4398
4399 Rep =
4400 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4401 } else if (Name.starts_with(Prefix: "avx512.mask.punpckh") ||
4402 Name.starts_with(Prefix: "avx512.mask.unpckh.")) {
4403 Value *Op0 = CI->getArgOperand(i: 0);
4404 Value *Op1 = CI->getArgOperand(i: 1);
4405 int NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4406 int NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4407
4408 SmallVector<int, 64> Idxs(NumElts);
4409 for (int l = 0; l != NumElts; l += NumLaneElts)
4410 for (int i = 0; i != NumLaneElts; ++i)
4411 Idxs[i + l] = (NumLaneElts / 2) + l + (i / 2) + NumElts * (i % 2);
4412
4413 Rep = Builder.CreateShuffleVector(V1: Op0, V2: Op1, Mask: Idxs);
4414
4415 Rep =
4416 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4417 } else if (Name.starts_with(Prefix: "avx512.mask.and.") ||
4418 Name.starts_with(Prefix: "avx512.mask.pand.")) {
4419 VectorType *FTy = cast<VectorType>(Val: CI->getType());
4420 VectorType *ITy = VectorType::getInteger(VTy: FTy);
4421 Rep = Builder.CreateAnd(LHS: Builder.CreateBitCast(V: CI->getArgOperand(i: 0), DestTy: ITy),
4422 RHS: Builder.CreateBitCast(V: CI->getArgOperand(i: 1), DestTy: ITy));
4423 Rep = Builder.CreateBitCast(V: Rep, DestTy: FTy);
4424 Rep =
4425 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4426 } else if (Name.starts_with(Prefix: "avx512.mask.andn.") ||
4427 Name.starts_with(Prefix: "avx512.mask.pandn.")) {
4428 VectorType *FTy = cast<VectorType>(Val: CI->getType());
4429 VectorType *ITy = VectorType::getInteger(VTy: FTy);
4430 Rep = Builder.CreateNot(V: Builder.CreateBitCast(V: CI->getArgOperand(i: 0), DestTy: ITy));
4431 Rep = Builder.CreateAnd(LHS: Rep,
4432 RHS: Builder.CreateBitCast(V: CI->getArgOperand(i: 1), DestTy: ITy));
4433 Rep = Builder.CreateBitCast(V: Rep, DestTy: FTy);
4434 Rep =
4435 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4436 } else if (Name.starts_with(Prefix: "avx512.mask.or.") ||
4437 Name.starts_with(Prefix: "avx512.mask.por.")) {
4438 VectorType *FTy = cast<VectorType>(Val: CI->getType());
4439 VectorType *ITy = VectorType::getInteger(VTy: FTy);
4440 Rep = Builder.CreateOr(LHS: Builder.CreateBitCast(V: CI->getArgOperand(i: 0), DestTy: ITy),
4441 RHS: Builder.CreateBitCast(V: CI->getArgOperand(i: 1), DestTy: ITy));
4442 Rep = Builder.CreateBitCast(V: Rep, DestTy: FTy);
4443 Rep =
4444 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4445 } else if (Name.starts_with(Prefix: "avx512.mask.xor.") ||
4446 Name.starts_with(Prefix: "avx512.mask.pxor.")) {
4447 VectorType *FTy = cast<VectorType>(Val: CI->getType());
4448 VectorType *ITy = VectorType::getInteger(VTy: FTy);
4449 Rep = Builder.CreateXor(LHS: Builder.CreateBitCast(V: CI->getArgOperand(i: 0), DestTy: ITy),
4450 RHS: Builder.CreateBitCast(V: CI->getArgOperand(i: 1), DestTy: ITy));
4451 Rep = Builder.CreateBitCast(V: Rep, DestTy: FTy);
4452 Rep =
4453 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4454 } else if (Name.starts_with(Prefix: "avx512.mask.padd.")) {
4455 Rep = Builder.CreateAdd(LHS: CI->getArgOperand(i: 0), RHS: CI->getArgOperand(i: 1));
4456 Rep =
4457 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4458 } else if (Name.starts_with(Prefix: "avx512.mask.psub.")) {
4459 Rep = Builder.CreateSub(LHS: CI->getArgOperand(i: 0), RHS: CI->getArgOperand(i: 1));
4460 Rep =
4461 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4462 } else if (Name.starts_with(Prefix: "avx512.mask.pmull.")) {
4463 Rep = Builder.CreateMul(LHS: CI->getArgOperand(i: 0), RHS: CI->getArgOperand(i: 1));
4464 Rep =
4465 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4466 } else if (Name.starts_with(Prefix: "avx512.mask.add.p")) {
4467 if (Name.ends_with(Suffix: ".512")) {
4468 Intrinsic::ID IID;
4469 if (Name[17] == 's')
4470 IID = Intrinsic::x86_avx512_add_ps_512;
4471 else
4472 IID = Intrinsic::x86_avx512_add_pd_512;
4473
4474 Rep = Builder.CreateIntrinsic(
4475 ID: IID,
4476 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1), CI->getArgOperand(i: 4)});
4477 } else {
4478 Rep = Builder.CreateFAdd(L: CI->getArgOperand(i: 0), R: CI->getArgOperand(i: 1));
4479 }
4480 Rep =
4481 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4482 } else if (Name.starts_with(Prefix: "avx512.mask.div.p")) {
4483 if (Name.ends_with(Suffix: ".512")) {
4484 Intrinsic::ID IID;
4485 if (Name[17] == 's')
4486 IID = Intrinsic::x86_avx512_div_ps_512;
4487 else
4488 IID = Intrinsic::x86_avx512_div_pd_512;
4489
4490 Rep = Builder.CreateIntrinsic(
4491 ID: IID,
4492 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1), CI->getArgOperand(i: 4)});
4493 } else {
4494 Rep = Builder.CreateFDiv(L: CI->getArgOperand(i: 0), R: CI->getArgOperand(i: 1));
4495 }
4496 Rep =
4497 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4498 } else if (Name.starts_with(Prefix: "avx512.mask.mul.p")) {
4499 if (Name.ends_with(Suffix: ".512")) {
4500 Intrinsic::ID IID;
4501 if (Name[17] == 's')
4502 IID = Intrinsic::x86_avx512_mul_ps_512;
4503 else
4504 IID = Intrinsic::x86_avx512_mul_pd_512;
4505
4506 Rep = Builder.CreateIntrinsic(
4507 ID: IID,
4508 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1), CI->getArgOperand(i: 4)});
4509 } else {
4510 Rep = Builder.CreateFMul(L: CI->getArgOperand(i: 0), R: CI->getArgOperand(i: 1));
4511 }
4512 Rep =
4513 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4514 } else if (Name.starts_with(Prefix: "avx512.mask.sub.p")) {
4515 if (Name.ends_with(Suffix: ".512")) {
4516 Intrinsic::ID IID;
4517 if (Name[17] == 's')
4518 IID = Intrinsic::x86_avx512_sub_ps_512;
4519 else
4520 IID = Intrinsic::x86_avx512_sub_pd_512;
4521
4522 Rep = Builder.CreateIntrinsic(
4523 ID: IID,
4524 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1), CI->getArgOperand(i: 4)});
4525 } else {
4526 Rep = Builder.CreateFSub(L: CI->getArgOperand(i: 0), R: CI->getArgOperand(i: 1));
4527 }
4528 Rep =
4529 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4530 } else if ((Name.starts_with(Prefix: "avx512.mask.max.p") ||
4531 Name.starts_with(Prefix: "avx512.mask.min.p")) &&
4532 Name.drop_front(N: 18) == ".512") {
4533 bool IsDouble = Name[17] == 'd';
4534 bool IsMin = Name[13] == 'i';
4535 static const Intrinsic::ID MinMaxTbl[2][2] = {
4536 {Intrinsic::x86_avx512_max_ps_512, Intrinsic::x86_avx512_max_pd_512},
4537 {Intrinsic::x86_avx512_min_ps_512, Intrinsic::x86_avx512_min_pd_512}};
4538 Intrinsic::ID IID = MinMaxTbl[IsMin][IsDouble];
4539
4540 Rep = Builder.CreateIntrinsic(
4541 ID: IID,
4542 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1), CI->getArgOperand(i: 4)});
4543 Rep =
4544 emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: CI->getArgOperand(i: 2));
4545 } else if (Name.starts_with(Prefix: "avx512.mask.lzcnt.")) {
4546 Rep =
4547 Builder.CreateIntrinsic(ID: Intrinsic::ctlz, OverloadTypes: CI->getType(),
4548 Args: {CI->getArgOperand(i: 0), Builder.getInt1(V: false)});
4549 Rep =
4550 emitX86Select(Builder, Mask: CI->getArgOperand(i: 2), Op0: Rep, Op1: CI->getArgOperand(i: 1));
4551 } else if (Name.starts_with(Prefix: "avx512.mask.psll")) {
4552 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4553 bool IsVariable = Name[16] == 'v';
4554 char Size = Name[16] == '.' ? Name[17]
4555 : Name[17] == '.' ? Name[18]
4556 : Name[18] == '.' ? Name[19]
4557 : Name[20];
4558
4559 Intrinsic::ID IID;
4560 if (IsVariable && Name[17] != '.') {
4561 if (Size == 'd' && Name[17] == '2') // avx512.mask.psllv2.di
4562 IID = Intrinsic::x86_avx2_psllv_q;
4563 else if (Size == 'd' && Name[17] == '4') // avx512.mask.psllv4.di
4564 IID = Intrinsic::x86_avx2_psllv_q_256;
4565 else if (Size == 's' && Name[17] == '4') // avx512.mask.psllv4.si
4566 IID = Intrinsic::x86_avx2_psllv_d;
4567 else if (Size == 's' && Name[17] == '8') // avx512.mask.psllv8.si
4568 IID = Intrinsic::x86_avx2_psllv_d_256;
4569 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psllv8.hi
4570 IID = Intrinsic::x86_avx512_psllv_w_128;
4571 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psllv16.hi
4572 IID = Intrinsic::x86_avx512_psllv_w_256;
4573 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psllv32hi
4574 IID = Intrinsic::x86_avx512_psllv_w_512;
4575 else
4576 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4577 } else if (Name.ends_with(Suffix: ".128")) {
4578 if (Size == 'd') // avx512.mask.psll.d.128, avx512.mask.psll.di.128
4579 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_d
4580 : Intrinsic::x86_sse2_psll_d;
4581 else if (Size == 'q') // avx512.mask.psll.q.128, avx512.mask.psll.qi.128
4582 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_q
4583 : Intrinsic::x86_sse2_psll_q;
4584 else if (Size == 'w') // avx512.mask.psll.w.128, avx512.mask.psll.wi.128
4585 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_w
4586 : Intrinsic::x86_sse2_psll_w;
4587 else
4588 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4589 } else if (Name.ends_with(Suffix: ".256")) {
4590 if (Size == 'd') // avx512.mask.psll.d.256, avx512.mask.psll.di.256
4591 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_d
4592 : Intrinsic::x86_avx2_psll_d;
4593 else if (Size == 'q') // avx512.mask.psll.q.256, avx512.mask.psll.qi.256
4594 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_q
4595 : Intrinsic::x86_avx2_psll_q;
4596 else if (Size == 'w') // avx512.mask.psll.w.256, avx512.mask.psll.wi.256
4597 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_w
4598 : Intrinsic::x86_avx2_psll_w;
4599 else
4600 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4601 } else {
4602 if (Size == 'd') // psll.di.512, pslli.d, psll.d, psllv.d.512
4603 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_d_512
4604 : IsVariable ? Intrinsic::x86_avx512_psllv_d_512
4605 : Intrinsic::x86_avx512_psll_d_512;
4606 else if (Size == 'q') // psll.qi.512, pslli.q, psll.q, psllv.q.512
4607 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_q_512
4608 : IsVariable ? Intrinsic::x86_avx512_psllv_q_512
4609 : Intrinsic::x86_avx512_psll_q_512;
4610 else if (Size == 'w') // psll.wi.512, pslli.w, psll.w
4611 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_w_512
4612 : Intrinsic::x86_avx512_psll_w_512;
4613 else
4614 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4615 }
4616
4617 Rep = upgradeX86MaskedShift(Builder, CI&: *CI, IID);
4618 } else if (Name.starts_with(Prefix: "avx512.mask.psrl")) {
4619 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4620 bool IsVariable = Name[16] == 'v';
4621 char Size = Name[16] == '.' ? Name[17]
4622 : Name[17] == '.' ? Name[18]
4623 : Name[18] == '.' ? Name[19]
4624 : Name[20];
4625
4626 Intrinsic::ID IID;
4627 if (IsVariable && Name[17] != '.') {
4628 if (Size == 'd' && Name[17] == '2') // avx512.mask.psrlv2.di
4629 IID = Intrinsic::x86_avx2_psrlv_q;
4630 else if (Size == 'd' && Name[17] == '4') // avx512.mask.psrlv4.di
4631 IID = Intrinsic::x86_avx2_psrlv_q_256;
4632 else if (Size == 's' && Name[17] == '4') // avx512.mask.psrlv4.si
4633 IID = Intrinsic::x86_avx2_psrlv_d;
4634 else if (Size == 's' && Name[17] == '8') // avx512.mask.psrlv8.si
4635 IID = Intrinsic::x86_avx2_psrlv_d_256;
4636 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psrlv8.hi
4637 IID = Intrinsic::x86_avx512_psrlv_w_128;
4638 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psrlv16.hi
4639 IID = Intrinsic::x86_avx512_psrlv_w_256;
4640 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psrlv32hi
4641 IID = Intrinsic::x86_avx512_psrlv_w_512;
4642 else
4643 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4644 } else if (Name.ends_with(Suffix: ".128")) {
4645 if (Size == 'd') // avx512.mask.psrl.d.128, avx512.mask.psrl.di.128
4646 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_d
4647 : Intrinsic::x86_sse2_psrl_d;
4648 else if (Size == 'q') // avx512.mask.psrl.q.128, avx512.mask.psrl.qi.128
4649 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_q
4650 : Intrinsic::x86_sse2_psrl_q;
4651 else if (Size == 'w') // avx512.mask.psrl.w.128, avx512.mask.psrl.wi.128
4652 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_w
4653 : Intrinsic::x86_sse2_psrl_w;
4654 else
4655 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4656 } else if (Name.ends_with(Suffix: ".256")) {
4657 if (Size == 'd') // avx512.mask.psrl.d.256, avx512.mask.psrl.di.256
4658 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_d
4659 : Intrinsic::x86_avx2_psrl_d;
4660 else if (Size == 'q') // avx512.mask.psrl.q.256, avx512.mask.psrl.qi.256
4661 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_q
4662 : Intrinsic::x86_avx2_psrl_q;
4663 else if (Size == 'w') // avx512.mask.psrl.w.256, avx512.mask.psrl.wi.256
4664 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_w
4665 : Intrinsic::x86_avx2_psrl_w;
4666 else
4667 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4668 } else {
4669 if (Size == 'd') // psrl.di.512, psrli.d, psrl.d, psrl.d.512
4670 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_d_512
4671 : IsVariable ? Intrinsic::x86_avx512_psrlv_d_512
4672 : Intrinsic::x86_avx512_psrl_d_512;
4673 else if (Size == 'q') // psrl.qi.512, psrli.q, psrl.q, psrl.q.512
4674 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_q_512
4675 : IsVariable ? Intrinsic::x86_avx512_psrlv_q_512
4676 : Intrinsic::x86_avx512_psrl_q_512;
4677 else if (Size == 'w') // psrl.wi.512, psrli.w, psrl.w)
4678 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_w_512
4679 : Intrinsic::x86_avx512_psrl_w_512;
4680 else
4681 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4682 }
4683
4684 Rep = upgradeX86MaskedShift(Builder, CI&: *CI, IID);
4685 } else if (Name.starts_with(Prefix: "avx512.mask.psra")) {
4686 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4687 bool IsVariable = Name[16] == 'v';
4688 char Size = Name[16] == '.' ? Name[17]
4689 : Name[17] == '.' ? Name[18]
4690 : Name[18] == '.' ? Name[19]
4691 : Name[20];
4692
4693 Intrinsic::ID IID;
4694 if (IsVariable && Name[17] != '.') {
4695 if (Size == 's' && Name[17] == '4') // avx512.mask.psrav4.si
4696 IID = Intrinsic::x86_avx2_psrav_d;
4697 else if (Size == 's' && Name[17] == '8') // avx512.mask.psrav8.si
4698 IID = Intrinsic::x86_avx2_psrav_d_256;
4699 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psrav8.hi
4700 IID = Intrinsic::x86_avx512_psrav_w_128;
4701 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psrav16.hi
4702 IID = Intrinsic::x86_avx512_psrav_w_256;
4703 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psrav32hi
4704 IID = Intrinsic::x86_avx512_psrav_w_512;
4705 else
4706 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4707 } else if (Name.ends_with(Suffix: ".128")) {
4708 if (Size == 'd') // avx512.mask.psra.d.128, avx512.mask.psra.di.128
4709 IID = IsImmediate ? Intrinsic::x86_sse2_psrai_d
4710 : Intrinsic::x86_sse2_psra_d;
4711 else if (Size == 'q') // avx512.mask.psra.q.128, avx512.mask.psra.qi.128
4712 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_128
4713 : IsVariable ? Intrinsic::x86_avx512_psrav_q_128
4714 : Intrinsic::x86_avx512_psra_q_128;
4715 else if (Size == 'w') // avx512.mask.psra.w.128, avx512.mask.psra.wi.128
4716 IID = IsImmediate ? Intrinsic::x86_sse2_psrai_w
4717 : Intrinsic::x86_sse2_psra_w;
4718 else
4719 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4720 } else if (Name.ends_with(Suffix: ".256")) {
4721 if (Size == 'd') // avx512.mask.psra.d.256, avx512.mask.psra.di.256
4722 IID = IsImmediate ? Intrinsic::x86_avx2_psrai_d
4723 : Intrinsic::x86_avx2_psra_d;
4724 else if (Size == 'q') // avx512.mask.psra.q.256, avx512.mask.psra.qi.256
4725 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_256
4726 : IsVariable ? Intrinsic::x86_avx512_psrav_q_256
4727 : Intrinsic::x86_avx512_psra_q_256;
4728 else if (Size == 'w') // avx512.mask.psra.w.256, avx512.mask.psra.wi.256
4729 IID = IsImmediate ? Intrinsic::x86_avx2_psrai_w
4730 : Intrinsic::x86_avx2_psra_w;
4731 else
4732 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4733 } else {
4734 if (Size == 'd') // psra.di.512, psrai.d, psra.d, psrav.d.512
4735 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_d_512
4736 : IsVariable ? Intrinsic::x86_avx512_psrav_d_512
4737 : Intrinsic::x86_avx512_psra_d_512;
4738 else if (Size == 'q') // psra.qi.512, psrai.q, psra.q
4739 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_512
4740 : IsVariable ? Intrinsic::x86_avx512_psrav_q_512
4741 : Intrinsic::x86_avx512_psra_q_512;
4742 else if (Size == 'w') // psra.wi.512, psrai.w, psra.w
4743 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_w_512
4744 : Intrinsic::x86_avx512_psra_w_512;
4745 else
4746 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected size", CI);
4747 }
4748
4749 Rep = upgradeX86MaskedShift(Builder, CI&: *CI, IID);
4750 } else if (Name.starts_with(Prefix: "avx512.mask.move.s")) {
4751 Rep = upgradeMaskedMove(Builder, CI&: *CI);
4752 } else if (Name.starts_with(Prefix: "avx512.cvtmask2")) {
4753 Rep = upgradeMaskToInt(Builder, CI&: *CI);
4754 } else if (Name.ends_with(Suffix: ".movntdqa")) {
4755 MDNode *Node = MDNode::get(
4756 Context&: C, MDs: ConstantAsMetadata::get(C: ConstantInt::get(Ty: Type::getInt32Ty(C), V: 1)));
4757
4758 LoadInst *LI = Builder.CreateAlignedLoad(
4759 Ty: CI->getType(), Ptr: CI->getArgOperand(i: 0),
4760 Align: Align(CI->getType()->getPrimitiveSizeInBits().getFixedValue() / 8));
4761 LI->setMetadata(KindID: LLVMContext::MD_nontemporal, Node);
4762 Rep = LI;
4763 } else if (Name.starts_with(Prefix: "fma.vfmadd.") ||
4764 Name.starts_with(Prefix: "fma.vfmsub.") ||
4765 Name.starts_with(Prefix: "fma.vfnmadd.") ||
4766 Name.starts_with(Prefix: "fma.vfnmsub.")) {
4767 bool NegMul = Name[6] == 'n';
4768 bool NegAcc = NegMul ? Name[8] == 's' : Name[7] == 's';
4769 bool IsScalar = NegMul ? Name[12] == 's' : Name[11] == 's';
4770
4771 Value *Ops[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
4772 CI->getArgOperand(i: 2)};
4773
4774 if (IsScalar) {
4775 Ops[0] = Builder.CreateExtractElement(Vec: Ops[0], Idx: (uint64_t)0);
4776 Ops[1] = Builder.CreateExtractElement(Vec: Ops[1], Idx: (uint64_t)0);
4777 Ops[2] = Builder.CreateExtractElement(Vec: Ops[2], Idx: (uint64_t)0);
4778 }
4779
4780 if (NegMul && !IsScalar)
4781 Ops[0] = Builder.CreateFNeg(V: Ops[0]);
4782 if (NegMul && IsScalar)
4783 Ops[1] = Builder.CreateFNeg(V: Ops[1]);
4784 if (NegAcc)
4785 Ops[2] = Builder.CreateFNeg(V: Ops[2]);
4786
4787 Rep = Builder.CreateIntrinsic(ID: Intrinsic::fma, OverloadTypes: Ops[0]->getType(), Args: Ops);
4788
4789 if (IsScalar)
4790 Rep = Builder.CreateInsertElement(Vec: CI->getArgOperand(i: 0), NewElt: Rep, Idx: (uint64_t)0);
4791 } else if (Name.starts_with(Prefix: "fma4.vfmadd.s")) {
4792 Value *Ops[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
4793 CI->getArgOperand(i: 2)};
4794
4795 Ops[0] = Builder.CreateExtractElement(Vec: Ops[0], Idx: (uint64_t)0);
4796 Ops[1] = Builder.CreateExtractElement(Vec: Ops[1], Idx: (uint64_t)0);
4797 Ops[2] = Builder.CreateExtractElement(Vec: Ops[2], Idx: (uint64_t)0);
4798
4799 Rep = Builder.CreateIntrinsic(ID: Intrinsic::fma, OverloadTypes: Ops[0]->getType(), Args: Ops);
4800
4801 Rep = Builder.CreateInsertElement(Vec: Constant::getNullValue(Ty: CI->getType()),
4802 NewElt: Rep, Idx: (uint64_t)0);
4803 } else if (Name.starts_with(Prefix: "avx512.mask.vfmadd.s") ||
4804 Name.starts_with(Prefix: "avx512.maskz.vfmadd.s") ||
4805 Name.starts_with(Prefix: "avx512.mask3.vfmadd.s") ||
4806 Name.starts_with(Prefix: "avx512.mask3.vfmsub.s") ||
4807 Name.starts_with(Prefix: "avx512.mask3.vfnmsub.s")) {
4808 bool IsMask3 = Name[11] == '3';
4809 bool IsMaskZ = Name[11] == 'z';
4810 // Drop the "avx512.mask." to make it easier.
4811 Name = Name.drop_front(N: IsMask3 || IsMaskZ ? 13 : 12);
4812 bool NegMul = Name[2] == 'n';
4813 bool NegAcc = NegMul ? Name[4] == 's' : Name[3] == 's';
4814
4815 Value *A = CI->getArgOperand(i: 0);
4816 Value *B = CI->getArgOperand(i: 1);
4817 Value *C = CI->getArgOperand(i: 2);
4818
4819 if (NegMul && (IsMask3 || IsMaskZ))
4820 A = Builder.CreateFNeg(V: A);
4821 if (NegMul && !(IsMask3 || IsMaskZ))
4822 B = Builder.CreateFNeg(V: B);
4823 if (NegAcc)
4824 C = Builder.CreateFNeg(V: C);
4825
4826 A = Builder.CreateExtractElement(Vec: A, Idx: (uint64_t)0);
4827 B = Builder.CreateExtractElement(Vec: B, Idx: (uint64_t)0);
4828 C = Builder.CreateExtractElement(Vec: C, Idx: (uint64_t)0);
4829
4830 if (!isa<ConstantInt>(Val: CI->getArgOperand(i: 4)) ||
4831 cast<ConstantInt>(Val: CI->getArgOperand(i: 4))->getZExtValue() != 4) {
4832 Value *Ops[] = {A, B, C, CI->getArgOperand(i: 4)};
4833
4834 Intrinsic::ID IID;
4835 if (Name.back() == 'd')
4836 IID = Intrinsic::x86_avx512_vfmadd_f64;
4837 else
4838 IID = Intrinsic::x86_avx512_vfmadd_f32;
4839 Rep = Builder.CreateIntrinsic(ID: IID, Args: Ops);
4840 } else {
4841 Rep = Builder.CreateFMA(Factor1: A, Factor2: B, Summand: C);
4842 }
4843
4844 Value *PassThru = IsMaskZ ? Constant::getNullValue(Ty: Rep->getType())
4845 : IsMask3 ? C
4846 : A;
4847
4848 // For Mask3 with NegAcc, we need to create a new extractelement that
4849 // avoids the negation above.
4850 if (NegAcc && IsMask3)
4851 PassThru =
4852 Builder.CreateExtractElement(Vec: CI->getArgOperand(i: 2), Idx: (uint64_t)0);
4853
4854 Rep = emitX86ScalarSelect(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: PassThru);
4855 Rep = Builder.CreateInsertElement(Vec: CI->getArgOperand(i: IsMask3 ? 2 : 0), NewElt: Rep,
4856 Idx: (uint64_t)0);
4857 } else if (Name.starts_with(Prefix: "avx512.mask.vfmadd.p") ||
4858 Name.starts_with(Prefix: "avx512.mask.vfnmadd.p") ||
4859 Name.starts_with(Prefix: "avx512.mask.vfnmsub.p") ||
4860 Name.starts_with(Prefix: "avx512.mask3.vfmadd.p") ||
4861 Name.starts_with(Prefix: "avx512.mask3.vfmsub.p") ||
4862 Name.starts_with(Prefix: "avx512.mask3.vfnmsub.p") ||
4863 Name.starts_with(Prefix: "avx512.maskz.vfmadd.p")) {
4864 bool IsMask3 = Name[11] == '3';
4865 bool IsMaskZ = Name[11] == 'z';
4866 // Drop the "avx512.mask." to make it easier.
4867 Name = Name.drop_front(N: IsMask3 || IsMaskZ ? 13 : 12);
4868 bool NegMul = Name[2] == 'n';
4869 bool NegAcc = NegMul ? Name[4] == 's' : Name[3] == 's';
4870
4871 Value *A = CI->getArgOperand(i: 0);
4872 Value *B = CI->getArgOperand(i: 1);
4873 Value *C = CI->getArgOperand(i: 2);
4874
4875 if (NegMul && (IsMask3 || IsMaskZ))
4876 A = Builder.CreateFNeg(V: A);
4877 if (NegMul && !(IsMask3 || IsMaskZ))
4878 B = Builder.CreateFNeg(V: B);
4879 if (NegAcc)
4880 C = Builder.CreateFNeg(V: C);
4881
4882 if (CI->arg_size() == 5 &&
4883 (!isa<ConstantInt>(Val: CI->getArgOperand(i: 4)) ||
4884 cast<ConstantInt>(Val: CI->getArgOperand(i: 4))->getZExtValue() != 4)) {
4885 Intrinsic::ID IID;
4886 // Check the character before ".512" in string.
4887 if (Name[Name.size() - 5] == 's')
4888 IID = Intrinsic::x86_avx512_vfmadd_ps_512;
4889 else
4890 IID = Intrinsic::x86_avx512_vfmadd_pd_512;
4891
4892 Rep = Builder.CreateIntrinsic(ID: IID, Args: {A, B, C, CI->getArgOperand(i: 4)});
4893 } else {
4894 Rep = Builder.CreateFMA(Factor1: A, Factor2: B, Summand: C);
4895 }
4896
4897 Value *PassThru = IsMaskZ ? llvm::Constant::getNullValue(Ty: CI->getType())
4898 : IsMask3 ? CI->getArgOperand(i: 2)
4899 : CI->getArgOperand(i: 0);
4900
4901 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: PassThru);
4902 } else if (Name.starts_with(Prefix: "fma.vfmsubadd.p")) {
4903 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
4904 unsigned EltWidth = CI->getType()->getScalarSizeInBits();
4905 Intrinsic::ID IID;
4906 if (VecWidth == 128 && EltWidth == 32)
4907 IID = Intrinsic::x86_fma_vfmaddsub_ps;
4908 else if (VecWidth == 256 && EltWidth == 32)
4909 IID = Intrinsic::x86_fma_vfmaddsub_ps_256;
4910 else if (VecWidth == 128 && EltWidth == 64)
4911 IID = Intrinsic::x86_fma_vfmaddsub_pd;
4912 else if (VecWidth == 256 && EltWidth == 64)
4913 IID = Intrinsic::x86_fma_vfmaddsub_pd_256;
4914 else
4915 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
4916
4917 Value *Ops[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
4918 CI->getArgOperand(i: 2)};
4919 Ops[2] = Builder.CreateFNeg(V: Ops[2]);
4920 Rep = Builder.CreateIntrinsic(ID: IID, Args: Ops);
4921 } else if (Name.starts_with(Prefix: "avx512.mask.vfmaddsub.p") ||
4922 Name.starts_with(Prefix: "avx512.mask3.vfmaddsub.p") ||
4923 Name.starts_with(Prefix: "avx512.maskz.vfmaddsub.p") ||
4924 Name.starts_with(Prefix: "avx512.mask3.vfmsubadd.p")) {
4925 bool IsMask3 = Name[11] == '3';
4926 bool IsMaskZ = Name[11] == 'z';
4927 // Drop the "avx512.mask." to make it easier.
4928 Name = Name.drop_front(N: IsMask3 || IsMaskZ ? 13 : 12);
4929 bool IsSubAdd = Name[3] == 's';
4930 if (CI->arg_size() == 5) {
4931 Intrinsic::ID IID;
4932 // Check the character before ".512" in string.
4933 if (Name[Name.size() - 5] == 's')
4934 IID = Intrinsic::x86_avx512_vfmaddsub_ps_512;
4935 else
4936 IID = Intrinsic::x86_avx512_vfmaddsub_pd_512;
4937
4938 Value *Ops[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
4939 CI->getArgOperand(i: 2), CI->getArgOperand(i: 4)};
4940 if (IsSubAdd)
4941 Ops[2] = Builder.CreateFNeg(V: Ops[2]);
4942
4943 Rep = Builder.CreateIntrinsic(ID: IID, Args: Ops);
4944 } else {
4945 int NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
4946
4947 Value *Ops[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
4948 CI->getArgOperand(i: 2)};
4949
4950 Function *FMA = Intrinsic::getOrInsertDeclaration(
4951 M: CI->getModule(), id: Intrinsic::fma, OverloadTys: Ops[0]->getType());
4952 Value *Odd = Builder.CreateCall(Callee: FMA, Args: Ops);
4953 Ops[2] = Builder.CreateFNeg(V: Ops[2]);
4954 Value *Even = Builder.CreateCall(Callee: FMA, Args: Ops);
4955
4956 if (IsSubAdd)
4957 std::swap(a&: Even, b&: Odd);
4958
4959 SmallVector<int, 32> Idxs(NumElts);
4960 for (int i = 0; i != NumElts; ++i)
4961 Idxs[i] = i + (i % 2) * NumElts;
4962
4963 Rep = Builder.CreateShuffleVector(V1: Even, V2: Odd, Mask: Idxs);
4964 }
4965
4966 Value *PassThru = IsMaskZ ? llvm::Constant::getNullValue(Ty: CI->getType())
4967 : IsMask3 ? CI->getArgOperand(i: 2)
4968 : CI->getArgOperand(i: 0);
4969
4970 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: PassThru);
4971 } else if (Name.starts_with(Prefix: "avx512.mask.pternlog.") ||
4972 Name.starts_with(Prefix: "avx512.maskz.pternlog.")) {
4973 bool ZeroMask = Name[11] == 'z';
4974 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
4975 unsigned EltWidth = CI->getType()->getScalarSizeInBits();
4976 Intrinsic::ID IID;
4977 if (VecWidth == 128 && EltWidth == 32)
4978 IID = Intrinsic::x86_avx512_pternlog_d_128;
4979 else if (VecWidth == 256 && EltWidth == 32)
4980 IID = Intrinsic::x86_avx512_pternlog_d_256;
4981 else if (VecWidth == 512 && EltWidth == 32)
4982 IID = Intrinsic::x86_avx512_pternlog_d_512;
4983 else if (VecWidth == 128 && EltWidth == 64)
4984 IID = Intrinsic::x86_avx512_pternlog_q_128;
4985 else if (VecWidth == 256 && EltWidth == 64)
4986 IID = Intrinsic::x86_avx512_pternlog_q_256;
4987 else if (VecWidth == 512 && EltWidth == 64)
4988 IID = Intrinsic::x86_avx512_pternlog_q_512;
4989 else
4990 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
4991
4992 Value *Args[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
4993 CI->getArgOperand(i: 2), CI->getArgOperand(i: 3)};
4994 Rep = Builder.CreateIntrinsic(ID: IID, Args);
4995 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(Ty: CI->getType())
4996 : CI->getArgOperand(i: 0);
4997 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 4), Op0: Rep, Op1: PassThru);
4998 } else if (Name.starts_with(Prefix: "avx512.mask.vpmadd52") ||
4999 Name.starts_with(Prefix: "avx512.maskz.vpmadd52")) {
5000 bool ZeroMask = Name[11] == 'z';
5001 bool High = Name[20] == 'h' || Name[21] == 'h';
5002 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
5003 Intrinsic::ID IID;
5004 if (VecWidth == 128 && !High)
5005 IID = Intrinsic::x86_avx512_vpmadd52l_uq_128;
5006 else if (VecWidth == 256 && !High)
5007 IID = Intrinsic::x86_avx512_vpmadd52l_uq_256;
5008 else if (VecWidth == 512 && !High)
5009 IID = Intrinsic::x86_avx512_vpmadd52l_uq_512;
5010 else if (VecWidth == 128 && High)
5011 IID = Intrinsic::x86_avx512_vpmadd52h_uq_128;
5012 else if (VecWidth == 256 && High)
5013 IID = Intrinsic::x86_avx512_vpmadd52h_uq_256;
5014 else if (VecWidth == 512 && High)
5015 IID = Intrinsic::x86_avx512_vpmadd52h_uq_512;
5016 else
5017 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
5018
5019 Value *Args[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
5020 CI->getArgOperand(i: 2)};
5021 Rep = Builder.CreateIntrinsic(ID: IID, Args);
5022 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(Ty: CI->getType())
5023 : CI->getArgOperand(i: 0);
5024 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: PassThru);
5025 } else if (Name.starts_with(Prefix: "avx512.mask.vpermi2var.") ||
5026 Name.starts_with(Prefix: "avx512.mask.vpermt2var.") ||
5027 Name.starts_with(Prefix: "avx512.maskz.vpermt2var.")) {
5028 bool ZeroMask = Name[11] == 'z';
5029 bool IndexForm = Name[17] == 'i';
5030 Rep = upgradeX86VPERMT2Intrinsics(Builder, CI&: *CI, ZeroMask, IndexForm);
5031 } else if (Name.starts_with(Prefix: "avx512.mask.vpdpbusd.") ||
5032 Name.starts_with(Prefix: "avx512.maskz.vpdpbusd.") ||
5033 Name.starts_with(Prefix: "avx512.mask.vpdpbusds.") ||
5034 Name.starts_with(Prefix: "avx512.maskz.vpdpbusds.")) {
5035 bool ZeroMask = Name[11] == 'z';
5036 bool IsSaturating = Name[ZeroMask ? 21 : 20] == 's';
5037 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
5038 Intrinsic::ID IID;
5039 if (VecWidth == 128 && !IsSaturating)
5040 IID = Intrinsic::x86_avx512_vpdpbusd_128;
5041 else if (VecWidth == 256 && !IsSaturating)
5042 IID = Intrinsic::x86_avx512_vpdpbusd_256;
5043 else if (VecWidth == 512 && !IsSaturating)
5044 IID = Intrinsic::x86_avx512_vpdpbusd_512;
5045 else if (VecWidth == 128 && IsSaturating)
5046 IID = Intrinsic::x86_avx512_vpdpbusds_128;
5047 else if (VecWidth == 256 && IsSaturating)
5048 IID = Intrinsic::x86_avx512_vpdpbusds_256;
5049 else if (VecWidth == 512 && IsSaturating)
5050 IID = Intrinsic::x86_avx512_vpdpbusds_512;
5051 else
5052 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
5053
5054 Value *Args[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
5055 CI->getArgOperand(i: 2)};
5056
5057 // Input arguments types were incorrectly set to vectors of i32 before but
5058 // they should be vectors of i8. Insert bit cast when encountering the old
5059 // types
5060 if (Args[1]->getType()->isVectorTy() &&
5061 cast<VectorType>(Val: Args[1]->getType())
5062 ->getElementType()
5063 ->isIntegerTy(BitWidth: 32) &&
5064 Args[2]->getType()->isVectorTy() &&
5065 cast<VectorType>(Val: Args[2]->getType())
5066 ->getElementType()
5067 ->isIntegerTy(BitWidth: 32)) {
5068 Type *NewArgType = nullptr;
5069 if (VecWidth == 128)
5070 NewArgType = VectorType::get(ElementType: Builder.getInt8Ty(), NumElements: 16, Scalable: false);
5071 else if (VecWidth == 256)
5072 NewArgType = VectorType::get(ElementType: Builder.getInt8Ty(), NumElements: 32, Scalable: false);
5073 else if (VecWidth == 512)
5074 NewArgType = VectorType::get(ElementType: Builder.getInt8Ty(), NumElements: 64, Scalable: false);
5075 else
5076 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected vector bit width",
5077 CI);
5078
5079 Args[1] = Builder.CreateBitCast(V: Args[1], DestTy: NewArgType);
5080 Args[2] = Builder.CreateBitCast(V: Args[2], DestTy: NewArgType);
5081 }
5082
5083 Rep = Builder.CreateIntrinsic(ID: IID, Args);
5084 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(Ty: CI->getType())
5085 : CI->getArgOperand(i: 0);
5086 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: PassThru);
5087 } else if (Name.starts_with(Prefix: "avx512.mask.vpdpwssd.") ||
5088 Name.starts_with(Prefix: "avx512.maskz.vpdpwssd.") ||
5089 Name.starts_with(Prefix: "avx512.mask.vpdpwssds.") ||
5090 Name.starts_with(Prefix: "avx512.maskz.vpdpwssds.")) {
5091 bool ZeroMask = Name[11] == 'z';
5092 bool IsSaturating = Name[ZeroMask ? 21 : 20] == 's';
5093 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
5094 Intrinsic::ID IID;
5095 if (VecWidth == 128 && !IsSaturating)
5096 IID = Intrinsic::x86_avx512_vpdpwssd_128;
5097 else if (VecWidth == 256 && !IsSaturating)
5098 IID = Intrinsic::x86_avx512_vpdpwssd_256;
5099 else if (VecWidth == 512 && !IsSaturating)
5100 IID = Intrinsic::x86_avx512_vpdpwssd_512;
5101 else if (VecWidth == 128 && IsSaturating)
5102 IID = Intrinsic::x86_avx512_vpdpwssds_128;
5103 else if (VecWidth == 256 && IsSaturating)
5104 IID = Intrinsic::x86_avx512_vpdpwssds_256;
5105 else if (VecWidth == 512 && IsSaturating)
5106 IID = Intrinsic::x86_avx512_vpdpwssds_512;
5107 else
5108 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
5109
5110 Value *Args[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
5111 CI->getArgOperand(i: 2)};
5112
5113 // Input arguments types were incorrectly set to vectors of i32 before but
5114 // they should be vectors of i16. Insert bit cast when encountering the old
5115 // types
5116 if (Args[1]->getType()->isVectorTy() &&
5117 cast<VectorType>(Val: Args[1]->getType())
5118 ->getElementType()
5119 ->isIntegerTy(BitWidth: 32) &&
5120 Args[2]->getType()->isVectorTy() &&
5121 cast<VectorType>(Val: Args[2]->getType())
5122 ->getElementType()
5123 ->isIntegerTy(BitWidth: 32)) {
5124 Type *NewArgType = nullptr;
5125 if (VecWidth == 128)
5126 NewArgType = VectorType::get(ElementType: Builder.getInt16Ty(), NumElements: 8, Scalable: false);
5127 else if (VecWidth == 256)
5128 NewArgType = VectorType::get(ElementType: Builder.getInt16Ty(), NumElements: 16, Scalable: false);
5129 else if (VecWidth == 512)
5130 NewArgType = VectorType::get(ElementType: Builder.getInt16Ty(), NumElements: 32, Scalable: false);
5131 else
5132 reportFatalUsageErrorWithCI(reason: "Intrinsic has unexpected vector bit width",
5133 CI);
5134
5135 Args[1] = Builder.CreateBitCast(V: Args[1], DestTy: NewArgType);
5136 Args[2] = Builder.CreateBitCast(V: Args[2], DestTy: NewArgType);
5137 }
5138
5139 Rep = Builder.CreateIntrinsic(ID: IID, Args);
5140 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(Ty: CI->getType())
5141 : CI->getArgOperand(i: 0);
5142 Rep = emitX86Select(Builder, Mask: CI->getArgOperand(i: 3), Op0: Rep, Op1: PassThru);
5143 } else if (Name == "addcarryx.u32" || Name == "addcarryx.u64" ||
5144 Name == "addcarry.u32" || Name == "addcarry.u64" ||
5145 Name == "subborrow.u32" || Name == "subborrow.u64") {
5146 Intrinsic::ID IID;
5147 if (Name[0] == 'a' && Name.back() == '2')
5148 IID = Intrinsic::x86_addcarry_32;
5149 else if (Name[0] == 'a' && Name.back() == '4')
5150 IID = Intrinsic::x86_addcarry_64;
5151 else if (Name[0] == 's' && Name.back() == '2')
5152 IID = Intrinsic::x86_subborrow_32;
5153 else if (Name[0] == 's' && Name.back() == '4')
5154 IID = Intrinsic::x86_subborrow_64;
5155 else
5156 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
5157
5158 // Make a call with 3 operands.
5159 Value *Args[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
5160 CI->getArgOperand(i: 2)};
5161 Value *NewCall = Builder.CreateIntrinsic(ID: IID, Args);
5162
5163 // Extract the second result and store it.
5164 Value *Data = Builder.CreateExtractValue(Agg: NewCall, Idxs: 1);
5165 Builder.CreateAlignedStore(Val: Data, Ptr: CI->getArgOperand(i: 3), Align: Align(1));
5166 // Replace the original call result with the first result of the new call.
5167 Value *CF = Builder.CreateExtractValue(Agg: NewCall, Idxs: 0);
5168
5169 CI->replaceAllUsesWith(V: CF);
5170 Rep = nullptr;
5171 } else if (Name.starts_with(Prefix: "avx512.mask.") &&
5172 upgradeAVX512MaskToSelect(Name, Builder, CI&: *CI, Rep)) {
5173 // Rep will be updated by the call in the condition.
5174 } else if (Name.starts_with(Prefix: "bmi.pdep.")) {
5175 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::pdep);
5176 } else if (Name.starts_with(Prefix: "bmi.pext.")) {
5177 Rep = upgradeX86BinaryIntrinsics(Builder, CI&: *CI, IID: Intrinsic::pext);
5178 } else
5179 reportFatalUsageErrorWithCI(reason: "Unexpected intrinsic", CI);
5180
5181 return Rep;
5182}
5183
5184static Value *upgradeAArch64IntrinsicCall(StringRef Name, CallBase *CI,
5185 Function *F, IRBuilder<> &Builder) {
5186 if (Name.starts_with(Prefix: "neon.bfcvt")) {
5187 if (Name.starts_with(Prefix: "neon.bfcvtn2")) {
5188 SmallVector<int, 32> LoMask(4);
5189 std::iota(first: LoMask.begin(), last: LoMask.end(), value: 0);
5190 SmallVector<int, 32> ConcatMask(8);
5191 std::iota(first: ConcatMask.begin(), last: ConcatMask.end(), value: 0);
5192 Value *Inactive = Builder.CreateShuffleVector(V: CI->getOperand(i_nocapture: 0), Mask: LoMask);
5193 Value *Trunc =
5194 Builder.CreateFPTrunc(V: CI->getOperand(i_nocapture: 1), DestTy: Inactive->getType());
5195 return Builder.CreateShuffleVector(V1: Inactive, V2: Trunc, Mask: ConcatMask);
5196 } else if (Name.starts_with(Prefix: "neon.bfcvtn")) {
5197 SmallVector<int, 32> ConcatMask(8);
5198 std::iota(first: ConcatMask.begin(), last: ConcatMask.end(), value: 0);
5199 Type *V4BF16 =
5200 FixedVectorType::get(ElementType: Type::getBFloatTy(C&: F->getContext()), NumElts: 4);
5201 Value *Trunc = Builder.CreateFPTrunc(V: CI->getOperand(i_nocapture: 0), DestTy: V4BF16);
5202 dbgs() << "Trunc: " << *Trunc << "\n";
5203 return Builder.CreateShuffleVector(
5204 V1: Trunc, V2: ConstantAggregateZero::get(Ty: V4BF16), Mask: ConcatMask);
5205 } else {
5206 return Builder.CreateFPTrunc(V: CI->getOperand(i_nocapture: 0),
5207 DestTy: Type::getBFloatTy(C&: F->getContext()));
5208 }
5209 } else if (Name.starts_with(Prefix: "sve.fcvt")) {
5210 Intrinsic::ID NewID =
5211 StringSwitch<Intrinsic::ID>(Name)
5212 .Case(S: "sve.fcvt.bf16f32", Value: Intrinsic::aarch64_sve_fcvt_bf16f32_v2)
5213 .Case(S: "sve.fcvtnt.bf16f32",
5214 Value: Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2)
5215 .Default(Value: Intrinsic::not_intrinsic);
5216 if (NewID == Intrinsic::not_intrinsic)
5217 llvm_unreachable("Unhandled Intrinsic!");
5218
5219 SmallVector<Value *, 3> Args(CI->args());
5220
5221 // The original intrinsics incorrectly used a predicate based on the
5222 // smallest element type rather than the largest.
5223 Type *BadPredTy = ScalableVectorType::get(ElementType: Builder.getInt1Ty(), MinNumElts: 8);
5224 Type *GoodPredTy = ScalableVectorType::get(ElementType: Builder.getInt1Ty(), MinNumElts: 4);
5225
5226 if (Args[1]->getType() != BadPredTy)
5227 llvm_unreachable("Unexpected predicate type!");
5228
5229 Args[1] = Builder.CreateIntrinsic(ID: Intrinsic::aarch64_sve_convert_to_svbool,
5230 OverloadTypes: BadPredTy, Args: Args[1]);
5231 Args[1] = Builder.CreateIntrinsic(
5232 ID: Intrinsic::aarch64_sve_convert_from_svbool, OverloadTypes: GoodPredTy, Args: Args[1]);
5233
5234 return Builder.CreateIntrinsic(ID: NewID, Args, /*FMFSource=*/nullptr,
5235 Name: CI->getName());
5236 }
5237
5238 if (Name == "neon.vcvtfp2hf")
5239 return Builder.CreateBitCast(
5240 V: Builder.CreateFPTrunc(
5241 V: CI->getOperand(i_nocapture: 0),
5242 DestTy: FixedVectorType::get(ElementType: Type::getHalfTy(C&: F->getContext()), NumElts: 4)),
5243 DestTy: FixedVectorType::get(ElementType: Type::getInt16Ty(C&: F->getContext()), NumElts: 4));
5244 if (Name == "neon.vcvthf2fp")
5245 return Builder.CreateFPExt(
5246 V: Builder.CreateBitCast(
5247 V: CI->getOperand(i_nocapture: 0),
5248 DestTy: FixedVectorType::get(ElementType: Type::getHalfTy(C&: F->getContext()), NumElts: 4)),
5249 DestTy: FixedVectorType::get(ElementType: Type::getFloatTy(C&: F->getContext()), NumElts: 4));
5250
5251 llvm_unreachable("Unhandled Intrinsic!");
5252}
5253
5254static Value *upgradeARMIntrinsicCall(StringRef Name, CallBase *CI, Function *F,
5255 IRBuilder<> &Builder) {
5256 if (Name == "mve.vctp64.old") {
5257 // Replace the old v4i1 vctp64 with a v2i1 vctp and predicate-casts to the
5258 // correct type.
5259 Value *VCTP = Builder.CreateIntrinsic(ID: Intrinsic::arm_mve_vctp64, OverloadTypes: {},
5260 Args: CI->getArgOperand(i: 0),
5261 /*FMFSource=*/nullptr, Name: CI->getName());
5262 Value *C1 = Builder.CreateIntrinsic(
5263 ID: Intrinsic::arm_mve_pred_v2i,
5264 OverloadTypes: {VectorType::get(ElementType: Builder.getInt1Ty(), NumElements: 2, Scalable: false)}, Args: VCTP);
5265 return Builder.CreateIntrinsic(
5266 ID: Intrinsic::arm_mve_pred_i2v,
5267 OverloadTypes: {VectorType::get(ElementType: Builder.getInt1Ty(), NumElements: 4, Scalable: false)}, Args: C1);
5268 } else if (Name == "mve.mull.int.predicated.v2i64.v4i32.v4i1" ||
5269 Name == "mve.vqdmull.predicated.v2i64.v4i32.v4i1" ||
5270 Name == "mve.vldr.gather.base.predicated.v2i64.v2i64.v4i1" ||
5271 Name == "mve.vldr.gather.base.wb.predicated.v2i64.v2i64.v4i1" ||
5272 Name ==
5273 "mve.vldr.gather.offset.predicated.v2i64.p0i64.v2i64.v4i1" ||
5274 Name == "mve.vldr.gather.offset.predicated.v2i64.p0.v2i64.v4i1" ||
5275 Name == "mve.vstr.scatter.base.predicated.v2i64.v2i64.v4i1" ||
5276 Name == "mve.vstr.scatter.base.wb.predicated.v2i64.v2i64.v4i1" ||
5277 Name ==
5278 "mve.vstr.scatter.offset.predicated.p0i64.v2i64.v2i64.v4i1" ||
5279 Name == "mve.vstr.scatter.offset.predicated.p0.v2i64.v2i64.v4i1" ||
5280 Name == "cde.vcx1q.predicated.v2i64.v4i1" ||
5281 Name == "cde.vcx1qa.predicated.v2i64.v4i1" ||
5282 Name == "cde.vcx2q.predicated.v2i64.v4i1" ||
5283 Name == "cde.vcx2qa.predicated.v2i64.v4i1" ||
5284 Name == "cde.vcx3q.predicated.v2i64.v4i1" ||
5285 Name == "cde.vcx3qa.predicated.v2i64.v4i1") {
5286 std::vector<Type *> Tys;
5287 unsigned ID = CI->getIntrinsicID();
5288 Type *V2I1Ty = FixedVectorType::get(ElementType: Builder.getInt1Ty(), NumElts: 2);
5289 switch (ID) {
5290 case Intrinsic::arm_mve_mull_int_predicated:
5291 case Intrinsic::arm_mve_vqdmull_predicated:
5292 case Intrinsic::arm_mve_vldr_gather_base_predicated:
5293 Tys = {CI->getType(), CI->getOperand(i_nocapture: 0)->getType(), V2I1Ty};
5294 break;
5295 case Intrinsic::arm_mve_vldr_gather_base_wb_predicated:
5296 case Intrinsic::arm_mve_vstr_scatter_base_predicated:
5297 case Intrinsic::arm_mve_vstr_scatter_base_wb_predicated:
5298 Tys = {CI->getOperand(i_nocapture: 0)->getType(), CI->getOperand(i_nocapture: 0)->getType(),
5299 V2I1Ty};
5300 break;
5301 case Intrinsic::arm_mve_vldr_gather_offset_predicated:
5302 Tys = {CI->getType(), CI->getOperand(i_nocapture: 0)->getType(),
5303 CI->getOperand(i_nocapture: 1)->getType(), V2I1Ty};
5304 break;
5305 case Intrinsic::arm_mve_vstr_scatter_offset_predicated:
5306 Tys = {CI->getOperand(i_nocapture: 0)->getType(), CI->getOperand(i_nocapture: 1)->getType(),
5307 CI->getOperand(i_nocapture: 2)->getType(), V2I1Ty};
5308 break;
5309 case Intrinsic::arm_cde_vcx1q_predicated:
5310 case Intrinsic::arm_cde_vcx1qa_predicated:
5311 case Intrinsic::arm_cde_vcx2q_predicated:
5312 case Intrinsic::arm_cde_vcx2qa_predicated:
5313 case Intrinsic::arm_cde_vcx3q_predicated:
5314 case Intrinsic::arm_cde_vcx3qa_predicated:
5315 Tys = {CI->getOperand(i_nocapture: 1)->getType(), V2I1Ty};
5316 break;
5317 default:
5318 llvm_unreachable("Unhandled Intrinsic!");
5319 }
5320
5321 std::vector<Value *> Ops;
5322 for (Value *Op : CI->args()) {
5323 Type *Ty = Op->getType();
5324 if (Ty->getScalarSizeInBits() == 1) {
5325 Value *C1 = Builder.CreateIntrinsic(
5326 ID: Intrinsic::arm_mve_pred_v2i,
5327 OverloadTypes: {VectorType::get(ElementType: Builder.getInt1Ty(), NumElements: 4, Scalable: false)}, Args: Op);
5328 Op = Builder.CreateIntrinsic(ID: Intrinsic::arm_mve_pred_i2v, OverloadTypes: {V2I1Ty}, Args: C1);
5329 }
5330 Ops.push_back(x: Op);
5331 }
5332
5333 return Builder.CreateIntrinsic(ID, OverloadTypes: Tys, Args: Ops, /*FMFSource=*/nullptr,
5334 Name: CI->getName());
5335 }
5336 llvm_unreachable("Unknown function for ARM CallBase upgrade.");
5337}
5338
5339// These are expected to have the arguments:
5340// atomic.intrin (ptr, rmw_value, ordering, scope, isVolatile)
5341//
5342// Except for int_amdgcn_ds_fadd_v2bf16 which only has (ptr, rmw_value).
5343//
5344static Value *upgradeAMDGCNIntrinsicCall(StringRef Name, CallBase *CI,
5345 Function *F, IRBuilder<> &Builder) {
5346 // Legacy WMMA iu intrinsics missed the optional clamp operand. Append clamp=0
5347 // for compatibility.
5348 auto UpgradeLegacyWMMAIUIntrinsicCall =
5349 [](Function *F, CallBase *CI, IRBuilder<> &Builder,
5350 ArrayRef<Type *> OverloadTys) -> Value * {
5351 // Prepare arguments, append clamp=0 for compatibility
5352 SmallVector<Value *, 10> Args(CI->args().begin(), CI->args().end());
5353 Args.push_back(Elt: Builder.getFalse());
5354
5355 // Insert the declaration for the right overload types
5356 Function *NewDecl = Intrinsic::getOrInsertDeclaration(
5357 M: F->getParent(), id: F->getIntrinsicID(), OverloadTys);
5358
5359 // Copy operand bundles if any
5360 SmallVector<OperandBundleDef, 1> Bundles;
5361 CI->getOperandBundlesAsDefs(Defs&: Bundles);
5362
5363 // Create the new call and copy calling properties
5364 auto *NewCall = cast<CallInst>(Val: Builder.CreateCall(Callee: NewDecl, Args, OpBundles: Bundles));
5365 NewCall->setTailCallKind(cast<CallInst>(Val: CI)->getTailCallKind());
5366 NewCall->setCallingConv(CI->getCallingConv());
5367 NewCall->setAttributes(CI->getAttributes());
5368 NewCall->copyMetadata(SrcInst: *CI);
5369 return NewCall;
5370 };
5371
5372 if (F->getIntrinsicID() == Intrinsic::amdgcn_wmma_i32_16x16x64_iu8) {
5373 assert(CI->arg_size() == 7 && "Legacy int_amdgcn_wmma_i32_16x16x64_iu8 "
5374 "intrinsic should have 7 arguments");
5375 Type *T1 = CI->getArgOperand(i: 4)->getType();
5376 Type *T2 = CI->getArgOperand(i: 1)->getType();
5377 return UpgradeLegacyWMMAIUIntrinsicCall(F, CI, Builder, {T1, T2});
5378 }
5379 if (F->getIntrinsicID() == Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8) {
5380 assert(CI->arg_size() == 8 && "Legacy int_amdgcn_swmmac_i32_16x16x128_iu8 "
5381 "intrinsic should have 8 arguments");
5382 Type *T1 = CI->getArgOperand(i: 4)->getType();
5383 Type *T2 = CI->getArgOperand(i: 1)->getType();
5384 Type *T3 = CI->getArgOperand(i: 3)->getType();
5385 Type *T4 = CI->getArgOperand(i: 5)->getType();
5386 return UpgradeLegacyWMMAIUIntrinsicCall(F, CI, Builder, {T1, T2, T3, T4});
5387 }
5388
5389 switch (F->getIntrinsicID()) {
5390 default:
5391 break;
5392 case Intrinsic::amdgcn_wmma_f32_16x16x4_f32:
5393 case Intrinsic::amdgcn_wmma_f32_16x16x32_bf16:
5394 case Intrinsic::amdgcn_wmma_f32_16x16x32_f16:
5395 case Intrinsic::amdgcn_wmma_f16_16x16x32_f16:
5396 case Intrinsic::amdgcn_wmma_bf16_16x16x32_bf16:
5397 case Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16: {
5398 // Drop src0 and src1 modifiers.
5399 const Value *Op0 = CI->getArgOperand(i: 0);
5400 const Value *Op2 = CI->getArgOperand(i: 2);
5401 assert(Op0->getType()->isIntegerTy() && Op2->getType()->isIntegerTy());
5402 const ConstantInt *ModA = dyn_cast<ConstantInt>(Val: Op0);
5403 const ConstantInt *ModB = dyn_cast<ConstantInt>(Val: Op2);
5404 if (!ModA->isZero() || !ModB->isZero())
5405 reportFatalUsageError(reason: Name + " matrix A and B modifiers shall be zero");
5406
5407 SmallVector<Value *, 8> Args{CI->getArgOperand(i: 1), CI->getArgOperand(i: 3)};
5408 for (int I = 4, E = CI->arg_size(); I < E; ++I)
5409 Args.push_back(Elt: CI->getArgOperand(i: I));
5410
5411 SmallVector<Type *, 3> Overloads{F->getReturnType(), Args[0]->getType()};
5412 if (F->getIntrinsicID() == Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16)
5413 Overloads.push_back(Elt: Args[3]->getType());
5414 Function *NewDecl = Intrinsic::getOrInsertDeclaration(
5415 M: F->getParent(), id: F->getIntrinsicID(), OverloadTys: Overloads);
5416
5417 SmallVector<OperandBundleDef, 1> Bundles;
5418 CI->getOperandBundlesAsDefs(Defs&: Bundles);
5419
5420 auto *NewCall = cast<CallInst>(Val: Builder.CreateCall(Callee: NewDecl, Args, OpBundles: Bundles));
5421 NewCall->setTailCallKind(cast<CallInst>(Val: CI)->getTailCallKind());
5422 NewCall->setCallingConv(CI->getCallingConv());
5423 NewCall->setAttributes(CI->getAttributes());
5424 NewCall->copyMetadata(SrcInst: *CI);
5425 NewCall->takeName(V: CI);
5426 return NewCall;
5427 }
5428 }
5429
5430 if (Name.starts_with(Prefix: "fcmp.") || Name.starts_with(Prefix: "icmp.")) {
5431 Value *LHS = CI->getArgOperand(i: 0);
5432 Value *RHS = CI->getArgOperand(i: 1);
5433 CmpInst::Predicate Pred = static_cast<CmpInst::Predicate>(
5434 cast<ConstantInt>(Val: CI->getArgOperand(i: 2))->getZExtValue());
5435 Value *Cmp = Builder.CreateCmp(Pred, LHS, RHS);
5436 CallInst *NewCall = Builder.CreateIntrinsicWithoutFolding(
5437 RetTy: CI->getType(), ID: Intrinsic::amdgcn_ballot, Args: Cmp);
5438 NewCall->setTailCallKind(cast<CallInst>(Val: CI)->getTailCallKind());
5439 NewCall->setCallingConv(CI->getCallingConv());
5440 NewCall->copyMetadata(SrcInst: *CI);
5441 NewCall->takeName(V: CI);
5442 return NewCall;
5443 }
5444
5445 if (Name.starts_with(Prefix: "addrspacecast.nonnull")) {
5446 if (CI->getNumOperands() < 2) // Malformed bitcode.
5447 return nullptr;
5448 Value *ASC = Builder.CreateAddrSpaceCast(
5449 V: CI->getArgOperand(i: 0), DestTy: CI->getType(), Name: "", /*IsNonNull=*/true);
5450 ASC->takeName(V: CI);
5451 return ASC;
5452 }
5453
5454 AtomicRMWInst::BinOp RMWOp =
5455 StringSwitch<AtomicRMWInst::BinOp>(Name)
5456 .StartsWith(S: "ds.fadd", Value: AtomicRMWInst::FAdd)
5457 .StartsWith(S: "ds.fmin", Value: AtomicRMWInst::FMin)
5458 .StartsWith(S: "ds.fmax", Value: AtomicRMWInst::FMax)
5459 .StartsWith(S: "atomic.inc.", Value: AtomicRMWInst::UIncWrap)
5460 .StartsWith(S: "atomic.dec.", Value: AtomicRMWInst::UDecWrap)
5461 .StartsWith(S: "global.atomic.fadd", Value: AtomicRMWInst::FAdd)
5462 .StartsWith(S: "flat.atomic.fadd", Value: AtomicRMWInst::FAdd)
5463 .StartsWith(S: "global.atomic.fmin", Value: AtomicRMWInst::FMin)
5464 .StartsWith(S: "flat.atomic.fmin", Value: AtomicRMWInst::FMin)
5465 .StartsWith(S: "global.atomic.fmax", Value: AtomicRMWInst::FMax)
5466 .StartsWith(S: "flat.atomic.fmax", Value: AtomicRMWInst::FMax)
5467 .StartsWith(S: "atomic.cond.sub", Value: AtomicRMWInst::USubCond)
5468 .StartsWith(S: "atomic.csub", Value: AtomicRMWInst::USubSat);
5469
5470 unsigned NumOperands = CI->getNumOperands();
5471 if (NumOperands < 3) // Malformed bitcode.
5472 return nullptr;
5473
5474 Value *Ptr = CI->getArgOperand(i: 0);
5475 PointerType *PtrTy = dyn_cast<PointerType>(Val: Ptr->getType());
5476 if (!PtrTy) // Malformed.
5477 return nullptr;
5478
5479 Value *Val = CI->getArgOperand(i: 1);
5480 if (Val->getType() != CI->getType()) // Malformed.
5481 return nullptr;
5482
5483 ConstantInt *OrderArg = nullptr;
5484 bool IsVolatile = false;
5485
5486 // These should have 5 arguments (plus the callee). A separate version of the
5487 // ds_fadd intrinsic was defined for bf16 which was missing arguments.
5488 if (NumOperands > 3)
5489 OrderArg = dyn_cast<ConstantInt>(Val: CI->getArgOperand(i: 2));
5490
5491 // Ignore scope argument at 3
5492
5493 if (NumOperands > 5) {
5494 ConstantInt *VolatileArg = dyn_cast<ConstantInt>(Val: CI->getArgOperand(i: 4));
5495 IsVolatile = !VolatileArg || !VolatileArg->isZero();
5496 }
5497
5498 AtomicOrdering Order = AtomicOrdering::SequentiallyConsistent;
5499 if (OrderArg && isValidAtomicOrdering(I: OrderArg->getZExtValue()))
5500 Order = static_cast<AtomicOrdering>(OrderArg->getZExtValue());
5501 if (Order == AtomicOrdering::NotAtomic || Order == AtomicOrdering::Unordered)
5502 Order = AtomicOrdering::SequentiallyConsistent;
5503
5504 LLVMContext &Ctx = F->getContext();
5505
5506 // Handle the v2bf16 intrinsic which used <2 x i16> instead of <2 x bfloat>
5507 Type *RetTy = CI->getType();
5508 if (VectorType *VT = dyn_cast<VectorType>(Val: RetTy)) {
5509 if (VT->getElementType()->isIntegerTy(BitWidth: 16)) {
5510 VectorType *AsBF16 =
5511 VectorType::get(ElementType: Type::getBFloatTy(C&: Ctx), EC: VT->getElementCount());
5512 Val = Builder.CreateBitCast(V: Val, DestTy: AsBF16);
5513 }
5514 }
5515
5516 // The scope argument never really worked correctly. Use agent as the most
5517 // conservative option which should still always produce the instruction.
5518 SyncScope::ID SSID = Ctx.getOrInsertSyncScopeID(SSN: "agent");
5519 AtomicRMWInst *RMW =
5520 Builder.CreateAtomicRMW(Op: RMWOp, Ptr, Val, Align: std::nullopt, Ordering: Order, SSID);
5521
5522 unsigned AddrSpace = PtrTy->getAddressSpace();
5523 if (AddrSpace != AMDGPUAS::LOCAL_ADDRESS) {
5524 MDNode *EmptyMD = MDNode::get(Context&: F->getContext(), MDs: {});
5525 RMW->setMetadata(Kind: "amdgpu.no.fine.grained.memory", Node: EmptyMD);
5526 if (RMWOp == AtomicRMWInst::FAdd && RetTy->isFloatTy())
5527 RMW->setMetadata(KindID: LLVMContext::MD_atomic_ignore_denormal_mode, Node: EmptyMD);
5528 }
5529
5530 if (AddrSpace == AMDGPUAS::FLAT_ADDRESS) {
5531 MDBuilder MDB(F->getContext());
5532 MDNode *RangeNotPrivate =
5533 MDB.createRange(Lo: APInt(32, AMDGPUAS::PRIVATE_ADDRESS),
5534 Hi: APInt(32, AMDGPUAS::PRIVATE_ADDRESS + 1));
5535 RMW->setMetadata(KindID: LLVMContext::MD_noalias_addrspace, Node: RangeNotPrivate);
5536 }
5537
5538 if (IsVolatile)
5539 RMW->setVolatile(true);
5540
5541 return Builder.CreateBitCast(V: RMW, DestTy: RetTy);
5542}
5543
5544/// Helper to unwrap intrinsic call MetadataAsValue operands. Return as a
5545/// plain MDNode, as it's the verifier's job to check these are the correct
5546/// types later.
5547static MDNode *unwrapMAVOp(CallBase *CI, unsigned Op) {
5548 if (Op < CI->arg_size()) {
5549 if (MetadataAsValue *MAV =
5550 dyn_cast<MetadataAsValue>(Val: CI->getArgOperand(i: Op))) {
5551 Metadata *MD = MAV->getMetadata();
5552 return dyn_cast_if_present<MDNode>(Val: MD);
5553 }
5554 }
5555 return nullptr;
5556}
5557
5558/// Helper to unwrap Metadata MetadataAsValue operands, such as the Value field.
5559static Metadata *unwrapMAVMetadataOp(CallBase *CI, unsigned Op) {
5560 if (Op < CI->arg_size())
5561 if (MetadataAsValue *MAV = dyn_cast<MetadataAsValue>(Val: CI->getArgOperand(i: Op)))
5562 return MAV->getMetadata();
5563 return nullptr;
5564}
5565
5566/// Convert debug intrinsic calls to non-instruction debug records.
5567/// \p Name - Final part of the intrinsic name, e.g. 'value' in llvm.dbg.value.
5568/// \p CI - The debug intrinsic call.
5569static void upgradeDbgIntrinsicToDbgRecord(StringRef Name, CallBase *CI) {
5570 DbgRecord *DR = nullptr;
5571 if (Name == "label") {
5572 DR = DbgLabelRecord::createUnresolvedDbgLabelRecord(Label: unwrapMAVOp(CI, Op: 0));
5573 } else if (Name == "assign") {
5574 DR = DbgVariableRecord::createUnresolvedDbgVariableRecord(
5575 Type: DbgVariableRecord::LocationType::Assign, Val: unwrapMAVMetadataOp(CI, Op: 0),
5576 Variable: unwrapMAVOp(CI, Op: 1), Expression: unwrapMAVOp(CI, Op: 2), AssignID: unwrapMAVOp(CI, Op: 3),
5577 Address: unwrapMAVMetadataOp(CI, Op: 4),
5578 /*The address is a Value ref, it will be stored as a Metadata */
5579 AddressExpression: unwrapMAVOp(CI, Op: 5));
5580 } else if (Name == "declare") {
5581 DR = DbgVariableRecord::createUnresolvedDbgVariableRecord(
5582 Type: DbgVariableRecord::LocationType::Declare, Val: unwrapMAVMetadataOp(CI, Op: 0),
5583 Variable: unwrapMAVOp(CI, Op: 1), Expression: unwrapMAVOp(CI, Op: 2), AssignID: nullptr, Address: nullptr, AddressExpression: nullptr);
5584 } else if (Name == "addr") {
5585 // Upgrade dbg.addr to dbg.value with DW_OP_deref.
5586 MDNode *ExprNode = unwrapMAVOp(CI, Op: 2);
5587 // Don't try to add something to the expression if it's not an expression.
5588 // Instead, allow the verifier to fail later.
5589 if (DIExpression *Expr = dyn_cast<DIExpression>(Val: ExprNode)) {
5590 ExprNode = DIExpression::append(Expr, Ops: dwarf::DW_OP_deref);
5591 }
5592 DR = DbgVariableRecord::createUnresolvedDbgVariableRecord(
5593 Type: DbgVariableRecord::LocationType::Value, Val: unwrapMAVMetadataOp(CI, Op: 0),
5594 Variable: unwrapMAVOp(CI, Op: 1), Expression: ExprNode, AssignID: nullptr, Address: nullptr, AddressExpression: nullptr);
5595 } else if (Name == "value") {
5596 // An old version of dbg.value had an extra offset argument.
5597 unsigned VarOp = 1;
5598 unsigned ExprOp = 2;
5599 if (CI->arg_size() == 4) {
5600 auto *Offset = dyn_cast_or_null<Constant>(Val: CI->getArgOperand(i: 1));
5601 // Nonzero offset dbg.values get dropped without a replacement.
5602 if (!Offset || !Offset->isNullValue())
5603 return;
5604 VarOp = 2;
5605 ExprOp = 3;
5606 }
5607 DR = DbgVariableRecord::createUnresolvedDbgVariableRecord(
5608 Type: DbgVariableRecord::LocationType::Value, Val: unwrapMAVMetadataOp(CI, Op: 0),
5609 Variable: unwrapMAVOp(CI, Op: VarOp), Expression: unwrapMAVOp(CI, Op: ExprOp), AssignID: nullptr, Address: nullptr,
5610 AddressExpression: nullptr);
5611 }
5612 DR->setDebugLoc(CI->getDebugLoc());
5613 assert(DR && "Unhandled intrinsic kind in upgrade to DbgRecord");
5614 CI->getParent()->insertDbgRecordBefore(DR, Here: CI->getIterator());
5615}
5616
5617static Value *upgradeVectorSplice(CallBase *CI, IRBuilder<> &Builder) {
5618 auto *Offset = dyn_cast<ConstantInt>(Val: CI->getArgOperand(i: 2));
5619 if (!Offset)
5620 reportFatalUsageError(reason: "Invalid llvm.vector.splice offset argument");
5621 int64_t OffsetVal = Offset->getSExtValue();
5622 return Builder.CreateIntrinsic(ID: OffsetVal >= 0
5623 ? Intrinsic::vector_splice_left
5624 : Intrinsic::vector_splice_right,
5625 OverloadTypes: CI->getType(),
5626 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
5627 Builder.getInt32(C: std::abs(i: OffsetVal))});
5628}
5629
5630static Value *upgradeConvertIntrinsicCall(StringRef Name, CallBase *CI,
5631 Function *F, IRBuilder<> &Builder) {
5632 if (Name.starts_with(Prefix: "to.fp16")) {
5633 Value *Cast =
5634 Builder.CreateFPTrunc(V: CI->getArgOperand(i: 0), DestTy: Builder.getHalfTy());
5635 return Builder.CreateBitCast(V: Cast, DestTy: CI->getType());
5636 }
5637
5638 if (Name.starts_with(Prefix: "from.fp16")) {
5639 Value *Cast =
5640 Builder.CreateBitCast(V: CI->getArgOperand(i: 0), DestTy: Builder.getHalfTy());
5641 return Builder.CreateFPExt(V: Cast, DestTy: CI->getType());
5642 }
5643
5644 return nullptr;
5645}
5646
5647static ICmpInst::Predicate getVPIntPredicateFromMD(const Value *Op) {
5648 Metadata *MD = cast<MetadataAsValue>(Val: Op)->getMetadata();
5649 if (!MD || !isa<MDString>(Val: MD))
5650 return ICmpInst::BAD_ICMP_PREDICATE;
5651 return StringSwitch<ICmpInst::Predicate>(cast<MDString>(Val: MD)->getString())
5652 .Case(S: "eq", Value: ICmpInst::ICMP_EQ)
5653 .Case(S: "ne", Value: ICmpInst::ICMP_NE)
5654 .Case(S: "ugt", Value: ICmpInst::ICMP_UGT)
5655 .Case(S: "uge", Value: ICmpInst::ICMP_UGE)
5656 .Case(S: "ult", Value: ICmpInst::ICMP_ULT)
5657 .Case(S: "ule", Value: ICmpInst::ICMP_ULE)
5658 .Case(S: "sgt", Value: ICmpInst::ICMP_SGT)
5659 .Case(S: "sge", Value: ICmpInst::ICMP_SGE)
5660 .Case(S: "slt", Value: ICmpInst::ICMP_SLT)
5661 .Case(S: "sle", Value: ICmpInst::ICMP_SLE)
5662 .Default(Value: ICmpInst::BAD_ICMP_PREDICATE);
5663}
5664
5665static FCmpInst::Predicate getVPFPPredicateFromMD(const Value *Op) {
5666 Metadata *MD = cast<MetadataAsValue>(Val: Op)->getMetadata();
5667 if (!MD || !isa<MDString>(Val: MD))
5668 return FCmpInst::BAD_FCMP_PREDICATE;
5669 return StringSwitch<FCmpInst::Predicate>(cast<MDString>(Val: MD)->getString())
5670 .Case(S: "oeq", Value: FCmpInst::FCMP_OEQ)
5671 .Case(S: "ogt", Value: FCmpInst::FCMP_OGT)
5672 .Case(S: "oge", Value: FCmpInst::FCMP_OGE)
5673 .Case(S: "olt", Value: FCmpInst::FCMP_OLT)
5674 .Case(S: "ole", Value: FCmpInst::FCMP_OLE)
5675 .Case(S: "one", Value: FCmpInst::FCMP_ONE)
5676 .Case(S: "ord", Value: FCmpInst::FCMP_ORD)
5677 .Case(S: "uno", Value: FCmpInst::FCMP_UNO)
5678 .Case(S: "ueq", Value: FCmpInst::FCMP_UEQ)
5679 .Case(S: "ugt", Value: FCmpInst::FCMP_UGT)
5680 .Case(S: "uge", Value: FCmpInst::FCMP_UGE)
5681 .Case(S: "ult", Value: FCmpInst::FCMP_ULT)
5682 .Case(S: "ule", Value: FCmpInst::FCMP_ULE)
5683 .Case(S: "une", Value: FCmpInst::FCMP_UNE)
5684 .Default(Value: FCmpInst::BAD_FCMP_PREDICATE);
5685}
5686
5687static Value *upgradeVPIntrinsicCall(StringRef Name, CallBase *CI,
5688 IRBuilder<> &Builder) {
5689 Value *Rep;
5690 unsigned Opcode = getFunctionalOpcodeForVP(Name);
5691 if (Opcode && Instruction::isUnaryOp(Opcode))
5692 Rep =
5693 Builder.CreateUnOp(Opc: (Instruction::UnaryOps)Opcode, V: CI->getArgOperand(i: 0));
5694 else if (Opcode && Instruction::isBinaryOp(Opcode))
5695 Rep = Builder.CreateBinOp(Opc: (Instruction::BinaryOps)Opcode,
5696 LHS: CI->getArgOperand(i: 0), RHS: CI->getArgOperand(i: 1));
5697 else if (Opcode && Instruction::isCast(Opcode))
5698 Rep = Builder.CreateCast(Op: (Instruction::CastOps)Opcode, V: CI->getArgOperand(i: 0),
5699 DestTy: CI->getType());
5700 else if (Opcode == Instruction::ICmp)
5701 Rep = Builder.CreateICmp(P: getVPIntPredicateFromMD(Op: CI->getArgOperand(i: 2)),
5702 LHS: CI->getArgOperand(i: 0), RHS: CI->getArgOperand(i: 1));
5703 else if (Opcode == Instruction::FCmp)
5704 Rep = Builder.CreateFCmp(P: getVPFPPredicateFromMD(Op: CI->getArgOperand(i: 2)),
5705 LHS: CI->getArgOperand(i: 0), RHS: CI->getArgOperand(i: 1));
5706 else if (Opcode == Instruction::Select)
5707 Rep = Builder.CreateSelect(C: CI->getArgOperand(i: 0), True: CI->getArgOperand(i: 1),
5708 False: CI->getArgOperand(i: 2));
5709 else if (auto IntrinsicID = getFunctionalIntrinsicIDForVP(Name)) {
5710 SmallVector<Value *, 2> Args(drop_end(RangeOrContainer: CI->args(), N: 2));
5711 Rep = Builder.CreateIntrinsic(RetTy: CI->getType(), ID: IntrinsicID, Args, FMFSource: {});
5712 } else
5713 llvm_unreachable("Unexpected vp intrinsic");
5714 Rep->takeName(V: CI);
5715 return Rep;
5716}
5717
5718static bool upgradeIntrinsicCallWithDefaultArgs(CallBase *CI, Function *NewFn,
5719 IRBuilder<> &Builder) {
5720 Intrinsic::ID IID = NewFn->getIntrinsicID();
5721
5722 auto [FirstDefault, Defaults] = Intrinsic::getAllDefaultArgValues(IID);
5723 if (Defaults.empty())
5724 return false;
5725
5726 unsigned OldArgCount = CI->arg_size();
5727 unsigned NewArgCount = NewFn->arg_size();
5728
5729 if (OldArgCount < FirstDefault)
5730 return false;
5731
5732 // More arguments than the new intrinsic accepts, cannot upgrade.
5733 if (OldArgCount > NewArgCount)
5734 return false;
5735
5736 // The call already passes the defaulted arguments explicitly; only the
5737 // callee is still the old, shorter declaration, so retarget it.
5738 if (OldArgCount == NewArgCount) {
5739 if (CI->getFunctionType() != NewFn->getFunctionType())
5740 return false;
5741 CI->setCalledFunction(NewFn);
5742 return true;
5743 }
5744
5745 // OldArgCount < NewArgCount: Fill in each missing trailing default
5746 // argument from the table.
5747 SmallVector<Value *, 8> NewArgs(CI->args());
5748
5749 FunctionType *NewFT = NewFn->getFunctionType();
5750 for (unsigned Idx = OldArgCount; Idx < NewArgCount; ++Idx) {
5751 assert(Idx >= FirstDefault && Idx - FirstDefault < Defaults.size() &&
5752 "missing argument outside the default range");
5753 Type *ParamTy = NewFT->getParamType(i: Idx);
5754
5755 // Only integer types are supported (i1, i8, i16, i32, i64).
5756 if (!ParamTy->isIntegerTy())
5757 return false;
5758 NewArgs.push_back(Elt: ConstantInt::get(Ty: ParamTy, V: Defaults[Idx - FirstDefault]));
5759 }
5760
5761 // Preserve operand bundles by creating the call with them.
5762 SmallVector<OperandBundleDef, 1> OpBundles;
5763 CI->getOperandBundlesAsDefs(Defs&: OpBundles);
5764 CallInst *NewCall = Builder.CreateCall(Callee: NewFn, Args: NewArgs, OpBundles);
5765
5766 NewCall->takeName(V: CI);
5767 NewCall->setCallingConv(CI->getCallingConv());
5768 NewCall->copyMetadata(SrcInst: *CI);
5769 if (auto *OldCI = dyn_cast<CallInst>(Val: CI))
5770 NewCall->setTailCallKind(OldCI->getTailCallKind());
5771
5772 CI->replaceAllUsesWith(V: NewCall);
5773 CI->eraseFromParent();
5774 return true;
5775}
5776
5777/// Upgrade a call to an old intrinsic. All argument and return casting must be
5778/// provided to seamlessly integrate with existing context.
5779void llvm::UpgradeIntrinsicCall(CallBase *CI, Function *NewFn) {
5780 // Note dyn_cast to Function is not quite the same as getCalledFunction, which
5781 // checks the callee's function type matches. It's likely we need to handle
5782 // type changes here.
5783 Function *F = dyn_cast<Function>(Val: CI->getCalledOperand());
5784 if (!F)
5785 return;
5786
5787 LLVMContext &C = CI->getContext();
5788 IRBuilder<> Builder(CI->getIterator());
5789 if (isa<FPMathOperator>(Val: CI))
5790 Builder.setFastMathFlags(CI->getFastMathFlags());
5791
5792 if (!NewFn) {
5793 // Get the Function's name.
5794 StringRef Name = F->getName();
5795 if (!Name.consume_front(Prefix: "llvm."))
5796 llvm_unreachable("intrinsic doesn't start with 'llvm.'");
5797
5798 bool IsX86 = Name.consume_front(Prefix: "x86.");
5799 bool IsNVVM = Name.consume_front(Prefix: "nvvm.");
5800 bool IsAArch64 = Name.consume_front(Prefix: "aarch64.");
5801 bool IsARM = Name.consume_front(Prefix: "arm.");
5802 bool IsAMDGCN = Name.consume_front(Prefix: "amdgcn.");
5803 bool IsDbg = Name.consume_front(Prefix: "dbg.");
5804 bool IsOldSplice =
5805 (Name.consume_front(Prefix: "experimental.vector.splice") ||
5806 Name.consume_front(Prefix: "vector.splice")) &&
5807 !(Name.starts_with(Prefix: ".left") || Name.starts_with(Prefix: ".right"));
5808 Value *Rep = nullptr;
5809
5810 if (!IsX86 && Name == "stackprotectorcheck") {
5811 Rep = nullptr;
5812 } else if (IsNVVM) {
5813 Rep = upgradeNVVMIntrinsicCall(Name, CI, F, Builder);
5814 } else if (IsX86) {
5815 Rep = upgradeX86IntrinsicCall(Name, CI, F, Builder);
5816 } else if (IsAArch64) {
5817 Rep = upgradeAArch64IntrinsicCall(Name, CI, F, Builder);
5818 } else if (IsARM) {
5819 Rep = upgradeARMIntrinsicCall(Name, CI, F, Builder);
5820 } else if (IsAMDGCN) {
5821 Rep = upgradeAMDGCNIntrinsicCall(Name, CI, F, Builder);
5822 } else if (IsDbg) {
5823 upgradeDbgIntrinsicToDbgRecord(Name, CI);
5824 } else if (IsOldSplice) {
5825 Rep = upgradeVectorSplice(CI, Builder);
5826 } else if (Name.consume_front(Prefix: "convert.")) {
5827 Rep = upgradeConvertIntrinsicCall(Name, CI, F, Builder);
5828 } else if (Name == "lifetime.start.i64" || Name == "lifetime.end.i64") {
5829 // Delete calls to invalid @llvm.lifetime.{start,end}.i64 intrinsics.
5830 Rep = nullptr;
5831 } else if (shouldUpgradeVPIntrinsic(Name)) {
5832 Rep = upgradeVPIntrinsicCall(Name, CI, Builder);
5833 } else {
5834 llvm_unreachable("Unknown function for CallBase upgrade.");
5835 }
5836
5837 if (Rep)
5838 CI->replaceAllUsesWith(V: Rep);
5839 CI->eraseFromParent();
5840 return;
5841 }
5842
5843 const auto &DefaultCase = [&]() -> void {
5844 if (F == NewFn)
5845 return;
5846
5847 if (CI->getFunctionType() == NewFn->getFunctionType()) {
5848 // Handle generic mangling change.
5849 assert(
5850 (CI->getCalledFunction()->getName() != NewFn->getName()) &&
5851 "Unknown function for CallBase upgrade and isn't just a name change");
5852 CI->setCalledFunction(NewFn);
5853 return;
5854 }
5855
5856 // This must be an upgrade from a named to a literal struct.
5857 if (auto *OldST = dyn_cast<StructType>(Val: CI->getType())) {
5858 assert(OldST != NewFn->getReturnType() &&
5859 "Return type must have changed");
5860 assert(OldST->getNumElements() ==
5861 cast<StructType>(NewFn->getReturnType())->getNumElements() &&
5862 "Must have same number of elements");
5863
5864 SmallVector<Value *> Args(CI->args());
5865 CallInst *NewCI = Builder.CreateCall(Callee: NewFn, Args);
5866 NewCI->setAttributes(CI->getAttributes());
5867 Value *Res = PoisonValue::get(T: OldST);
5868 for (unsigned Idx = 0; Idx < OldST->getNumElements(); ++Idx) {
5869 Value *Elem = Builder.CreateExtractValue(Agg: NewCI, Idxs: Idx);
5870 Res = Builder.CreateInsertValue(Agg: Res, Val: Elem, Idxs: Idx);
5871 }
5872 CI->replaceAllUsesWith(V: Res);
5873 CI->eraseFromParent();
5874 return;
5875 }
5876
5877 // We're probably about to produce something invalid. Let the verifier catch
5878 // it instead of dying here.
5879 CI->setCalledOperand(
5880 ConstantExpr::getPointerCast(C: NewFn, Ty: CI->getCalledOperand()->getType()));
5881 return;
5882 };
5883 CallInst *NewCall = nullptr;
5884 switch (NewFn->getIntrinsicID()) {
5885 default: {
5886 if (upgradeIntrinsicCallWithDefaultArgs(CI, NewFn, Builder))
5887 return;
5888 DefaultCase();
5889 return;
5890 }
5891 case Intrinsic::arm_neon_vst1:
5892 case Intrinsic::arm_neon_vst2:
5893 case Intrinsic::arm_neon_vst3:
5894 case Intrinsic::arm_neon_vst4:
5895 case Intrinsic::arm_neon_vst2lane:
5896 case Intrinsic::arm_neon_vst3lane:
5897 case Intrinsic::arm_neon_vst4lane: {
5898 SmallVector<Value *, 4> Args(CI->args());
5899 NewCall = Builder.CreateCall(Callee: NewFn, Args);
5900 break;
5901 }
5902 case Intrinsic::aarch64_sve_bfmlalb_lane_v2:
5903 case Intrinsic::aarch64_sve_bfmlalt_lane_v2:
5904 case Intrinsic::aarch64_sve_bfdot_lane_v2: {
5905 LLVMContext &Ctx = F->getParent()->getContext();
5906 SmallVector<Value *, 4> Args(CI->args());
5907 Args[3] = ConstantInt::get(Ty: Type::getInt32Ty(C&: Ctx),
5908 V: cast<ConstantInt>(Val: Args[3])->getZExtValue());
5909 NewCall = Builder.CreateCall(Callee: NewFn, Args);
5910 break;
5911 }
5912 case Intrinsic::aarch64_sve_ld3_sret:
5913 case Intrinsic::aarch64_sve_ld4_sret:
5914 case Intrinsic::aarch64_sve_ld2_sret: {
5915 // Is this a trivial remangle of the name to support ptr address spaces?
5916 if (isa<StructType>(Val: F->getReturnType())) {
5917 DefaultCase();
5918 return;
5919 }
5920
5921 StringRef Name = F->getName();
5922 Name = Name.substr(Start: 5);
5923 unsigned N = StringSwitch<unsigned>(Name)
5924 .StartsWith(S: "aarch64.sve.ld2", Value: 2)
5925 .StartsWith(S: "aarch64.sve.ld3", Value: 3)
5926 .StartsWith(S: "aarch64.sve.ld4", Value: 4)
5927 .Default(Value: 0);
5928 auto *RetTy = cast<ScalableVectorType>(Val: F->getReturnType());
5929 unsigned MinElts = RetTy->getMinNumElements() / N;
5930 SmallVector<Value *, 2> Args(CI->args());
5931 Value *NewLdCall = Builder.CreateCall(Callee: NewFn, Args);
5932 Value *Ret = llvm::PoisonValue::get(T: RetTy);
5933 for (unsigned I = 0; I < N; I++) {
5934 Value *SRet = Builder.CreateExtractValue(Agg: NewLdCall, Idxs: I);
5935 Ret = Builder.CreateInsertVector(DstType: RetTy, SrcVec: Ret, SubVec: SRet, Idx: I * MinElts);
5936 }
5937 NewCall = dyn_cast<CallInst>(Val: Ret);
5938 break;
5939 }
5940
5941 case Intrinsic::coro_end_async:
5942 case Intrinsic::coro_end: {
5943 SmallVector<Value *, 3> Args(CI->args());
5944 if (NewFn->getIntrinsicID() == Intrinsic::coro_end && Args.size() == 2)
5945 Args.push_back(Elt: ConstantTokenNone::get(Context&: CI->getContext()));
5946 NewCall = Builder.CreateCall(Callee: NewFn, Args);
5947
5948 if (!CI->getType()->isVoidTy()) {
5949 if (!CI->use_empty()) {
5950 Function *IsInRamp = Intrinsic::getOrInsertDeclaration(
5951 M: CI->getModule(), id: Intrinsic::coro_is_in_ramp);
5952 Value *InRamp = Builder.CreateCall(Callee: IsInRamp);
5953 CI->replaceAllUsesWith(V: Builder.CreateNot(V: InRamp));
5954 }
5955 CI->eraseFromParent();
5956 return;
5957 }
5958
5959 break;
5960 }
5961
5962 case Intrinsic::vector_extract: {
5963 StringRef Name = F->getName();
5964 Name = Name.substr(Start: 5); // Strip llvm
5965 if (!Name.starts_with(Prefix: "aarch64.sve.tuple.get")) {
5966 DefaultCase();
5967 return;
5968 }
5969 auto *RetTy = cast<ScalableVectorType>(Val: F->getReturnType());
5970 unsigned MinElts = RetTy->getMinNumElements();
5971 uint64_t I = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
5972 Value *NewIdx = ConstantInt::get(Ty: Type::getInt64Ty(C), V: I * MinElts);
5973 NewCall = Builder.CreateCall(Callee: NewFn, Args: {CI->getArgOperand(i: 0), NewIdx});
5974 break;
5975 }
5976
5977 case Intrinsic::vector_insert: {
5978 StringRef Name = F->getName();
5979 Name = Name.substr(Start: 5);
5980 if (!Name.starts_with(Prefix: "aarch64.sve.tuple")) {
5981 DefaultCase();
5982 return;
5983 }
5984 if (Name.starts_with(Prefix: "aarch64.sve.tuple.set")) {
5985 uint64_t I = cast<ConstantInt>(Val: CI->getArgOperand(i: 1))->getZExtValue();
5986 auto *Ty = cast<ScalableVectorType>(Val: CI->getArgOperand(i: 2)->getType());
5987 Value *NewIdx =
5988 ConstantInt::get(Ty: Type::getInt64Ty(C), V: I * Ty->getMinNumElements());
5989 NewCall = Builder.CreateCall(
5990 Callee: NewFn, Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 2), NewIdx});
5991 break;
5992 }
5993 if (Name.starts_with(Prefix: "aarch64.sve.tuple.create")) {
5994 unsigned N = StringSwitch<unsigned>(Name)
5995 .StartsWith(S: "aarch64.sve.tuple.create2", Value: 2)
5996 .StartsWith(S: "aarch64.sve.tuple.create3", Value: 3)
5997 .StartsWith(S: "aarch64.sve.tuple.create4", Value: 4)
5998 .Default(Value: 0);
5999 assert(N > 1 && "Create is expected to be between 2-4");
6000 auto *RetTy = cast<ScalableVectorType>(Val: F->getReturnType());
6001 Value *Ret = llvm::PoisonValue::get(T: RetTy);
6002 unsigned MinElts = RetTy->getMinNumElements() / N;
6003 for (unsigned I = 0; I < N; I++) {
6004 Value *V = CI->getArgOperand(i: I);
6005 Ret = Builder.CreateInsertVector(DstType: RetTy, SrcVec: Ret, SubVec: V, Idx: I * MinElts);
6006 }
6007 NewCall = dyn_cast<CallInst>(Val: Ret);
6008 }
6009 break;
6010 }
6011
6012 case Intrinsic::arm_neon_bfdot:
6013 case Intrinsic::arm_neon_bfmmla:
6014 case Intrinsic::arm_neon_bfmlalb:
6015 case Intrinsic::arm_neon_bfmlalt:
6016 case Intrinsic::aarch64_neon_bfdot:
6017 case Intrinsic::aarch64_neon_bfmmla:
6018 case Intrinsic::aarch64_neon_bfmlalb:
6019 case Intrinsic::aarch64_neon_bfmlalt: {
6020 SmallVector<Value *, 3> Args;
6021 assert(CI->arg_size() == 3 &&
6022 "Mismatch between function args and call args");
6023 size_t OperandWidth =
6024 CI->getArgOperand(i: 1)->getType()->getPrimitiveSizeInBits();
6025 assert((OperandWidth == 64 || OperandWidth == 128) &&
6026 "Unexpected operand width");
6027 Type *NewTy = FixedVectorType::get(ElementType: Type::getBFloatTy(C), NumElts: OperandWidth / 16);
6028 auto Iter = CI->args().begin();
6029 Args.push_back(Elt: *Iter++);
6030 Args.push_back(Elt: Builder.CreateBitCast(V: *Iter++, DestTy: NewTy));
6031 Args.push_back(Elt: Builder.CreateBitCast(V: *Iter++, DestTy: NewTy));
6032 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6033 break;
6034 }
6035
6036 case Intrinsic::bitreverse:
6037 NewCall = Builder.CreateCall(Callee: NewFn, Args: {CI->getArgOperand(i: 0)});
6038 break;
6039
6040 case Intrinsic::ctlz:
6041 case Intrinsic::cttz: {
6042 if (CI->arg_size() != 1) {
6043 DefaultCase();
6044 return;
6045 }
6046
6047 NewCall =
6048 Builder.CreateCall(Callee: NewFn, Args: {CI->getArgOperand(i: 0), Builder.getFalse()});
6049 break;
6050 }
6051
6052 case Intrinsic::objectsize: {
6053 Value *NullIsUnknownSize =
6054 CI->arg_size() == 2 ? Builder.getFalse() : CI->getArgOperand(i: 2);
6055 Value *Dynamic =
6056 CI->arg_size() < 4 ? Builder.getFalse() : CI->getArgOperand(i: 3);
6057 NewCall = Builder.CreateCall(
6058 Callee: NewFn, Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1), NullIsUnknownSize, Dynamic});
6059 break;
6060 }
6061
6062 case Intrinsic::ctpop:
6063 NewCall = Builder.CreateCall(Callee: NewFn, Args: {CI->getArgOperand(i: 0)});
6064 break;
6065 case Intrinsic::dbg_value: {
6066 StringRef Name = F->getName();
6067 Name = Name.substr(Start: 5); // Strip llvm.
6068 // Upgrade `dbg.addr` to `dbg.value` with `DW_OP_deref`.
6069 if (Name.starts_with(Prefix: "dbg.addr")) {
6070 DIExpression *Expr = cast<DIExpression>(
6071 Val: cast<MetadataAsValue>(Val: CI->getArgOperand(i: 2))->getMetadata());
6072 Expr = DIExpression::append(Expr, Ops: dwarf::DW_OP_deref);
6073 NewCall =
6074 Builder.CreateCall(Callee: NewFn, Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
6075 MetadataAsValue::get(Context&: C, MD: Expr)});
6076 break;
6077 }
6078
6079 // Upgrade from the old version that had an extra offset argument.
6080 assert(CI->arg_size() == 4);
6081 // Drop nonzero offsets instead of attempting to upgrade them.
6082 if (auto *Offset = dyn_cast_or_null<Constant>(Val: CI->getArgOperand(i: 1)))
6083 if (Offset->isNullValue()) {
6084 NewCall = Builder.CreateCall(
6085 Callee: NewFn,
6086 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 2), CI->getArgOperand(i: 3)});
6087 break;
6088 }
6089 CI->eraseFromParent();
6090 return;
6091 }
6092
6093 case Intrinsic::ptr_annotation:
6094 // Upgrade from versions that lacked the annotation attribute argument.
6095 if (CI->arg_size() != 4) {
6096 DefaultCase();
6097 return;
6098 }
6099
6100 // Create a new call with an added null annotation attribute argument.
6101 NewCall = Builder.CreateCall(
6102 Callee: NewFn,
6103 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1), CI->getArgOperand(i: 2),
6104 CI->getArgOperand(i: 3), ConstantPointerNull::get(T: Builder.getPtrTy())});
6105 NewCall->takeName(V: CI);
6106 CI->replaceAllUsesWith(V: NewCall);
6107 CI->eraseFromParent();
6108 return;
6109
6110 case Intrinsic::var_annotation:
6111 // Upgrade from versions that lacked the annotation attribute argument.
6112 if (CI->arg_size() != 4) {
6113 DefaultCase();
6114 return;
6115 }
6116 // Create a new call with an added null annotation attribute argument.
6117 NewCall = Builder.CreateCall(
6118 Callee: NewFn,
6119 Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1), CI->getArgOperand(i: 2),
6120 CI->getArgOperand(i: 3), ConstantPointerNull::get(T: Builder.getPtrTy())});
6121 NewCall->takeName(V: CI);
6122 CI->replaceAllUsesWith(V: NewCall);
6123 CI->eraseFromParent();
6124 return;
6125
6126 case Intrinsic::riscv_aes32dsi:
6127 case Intrinsic::riscv_aes32dsmi:
6128 case Intrinsic::riscv_aes32esi:
6129 case Intrinsic::riscv_aes32esmi:
6130 case Intrinsic::riscv_sm4ks:
6131 case Intrinsic::riscv_sm4ed: {
6132 // The last argument to these intrinsics used to be i8 and changed to i32.
6133 // The type overload for sm4ks and sm4ed was removed.
6134 Value *Arg2 = CI->getArgOperand(i: 2);
6135 if (Arg2->getType()->isIntegerTy(BitWidth: 32) && !CI->getType()->isIntegerTy(BitWidth: 64))
6136 return;
6137
6138 Value *Arg0 = CI->getArgOperand(i: 0);
6139 Value *Arg1 = CI->getArgOperand(i: 1);
6140 if (CI->getType()->isIntegerTy(BitWidth: 64)) {
6141 Arg0 = Builder.CreateTrunc(V: Arg0, DestTy: Builder.getInt32Ty());
6142 Arg1 = Builder.CreateTrunc(V: Arg1, DestTy: Builder.getInt32Ty());
6143 }
6144
6145 Arg2 = ConstantInt::get(Ty: Type::getInt32Ty(C),
6146 V: cast<ConstantInt>(Val: Arg2)->getZExtValue());
6147
6148 NewCall = Builder.CreateCall(Callee: NewFn, Args: {Arg0, Arg1, Arg2});
6149 Value *Res = NewCall;
6150 if (Res->getType() != CI->getType())
6151 Res = Builder.CreateIntCast(V: NewCall, DestTy: CI->getType(), /*isSigned*/ true);
6152 NewCall->takeName(V: CI);
6153 CI->replaceAllUsesWith(V: Res);
6154 CI->eraseFromParent();
6155 return;
6156 }
6157 case Intrinsic::nvvm_mapa_shared_cluster: {
6158 // Create a new call with the correct address space.
6159 NewCall =
6160 Builder.CreateCall(Callee: NewFn, Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1)});
6161 Value *Res = NewCall;
6162 Res = Builder.CreateAddrSpaceCast(
6163 V: Res, DestTy: Builder.getPtrTy(AddrSpace: NVPTXAS::ADDRESS_SPACE_SHARED));
6164 NewCall->takeName(V: CI);
6165 CI->replaceAllUsesWith(V: Res);
6166 CI->eraseFromParent();
6167 return;
6168 }
6169 case Intrinsic::nvvm_cp_async_bulk_global_to_shared_cluster: {
6170 SmallVector<Value *, 4> Args(CI->args());
6171 unsigned AS = Args[0]->getType()->getPointerAddressSpace();
6172 if (AS == NVPTXAS::ADDRESS_SPACE_SHARED)
6173 Args[0] = Builder.CreateAddrSpaceCast(
6174 V: Args[0], DestTy: Builder.getPtrTy(AddrSpace: NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6175
6176 // Append the missing trailing flag_valid_pattern (0 = disabled).
6177 Args.push_back(Elt: Builder.getInt32(C: 0));
6178
6179 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6180 NewCall->takeName(V: CI);
6181 CI->replaceAllUsesWith(V: NewCall);
6182 CI->eraseFromParent();
6183 return;
6184 }
6185 case Intrinsic::nvvm_cp_async_bulk_global_to_shared_cta: {
6186 // (dst, mbar, src, size, ch, flag_ch)
6187 // -> (dst, mbar, src, size, i32 0, i32 0, ch, flag_ch, i1 false,
6188 // i32 0 /* flag_valid_pattern=disabled */)
6189 SmallVector<Value *, 10> Args;
6190 for (unsigned I = 0; I < 4; ++I)
6191 Args.push_back(Elt: CI->getArgOperand(i: I));
6192 Args.push_back(Elt: Builder.getInt32(C: 0)); // ignore_bytes_left
6193 Args.push_back(Elt: Builder.getInt32(C: 0)); // ignore_bytes_right
6194 Args.push_back(Elt: CI->getArgOperand(i: 4)); // cache_hint
6195 Args.push_back(Elt: CI->getArgOperand(i: 5)); // flag_ch
6196 Args.push_back(Elt: Builder.getInt1(V: false)); // flag_oob
6197 Args.push_back(Elt: Builder.getInt32(C: 0)); // flag_valid_pattern
6198
6199 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6200 NewCall->takeName(V: CI);
6201 CI->replaceAllUsesWith(V: NewCall);
6202 CI->eraseFromParent();
6203 return;
6204 }
6205 case Intrinsic::nvvm_cp_async_bulk_shared_cta_to_cluster: {
6206 // Create a new call with the correct address space.
6207 SmallVector<Value *, 4> Args(CI->args());
6208 Args[0] = Builder.CreateAddrSpaceCast(
6209 V: Args[0], DestTy: Builder.getPtrTy(AddrSpace: NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6210
6211 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6212 NewCall->takeName(V: CI);
6213 CI->replaceAllUsesWith(V: NewCall);
6214 CI->eraseFromParent();
6215 return;
6216 }
6217 // clang-format off
6218#define G2S_CLUSTER_CASE(ID_SUFFIX, NAME) \
6219 case Intrinsic::nvvm_cp_async_bulk_tensor_g2s_##ID_SUFFIX:
6220 NVVM_TMA_G2S_MODES(G2S_CLUSTER_CASE)
6221#undef G2S_CLUSTER_CASE
6222 {
6223 SmallVector<Value *, 16> Args(CI->args());
6224 unsigned AS = CI->getArgOperand(i: 0)->getType()->getPointerAddressSpace();
6225 if (AS == NVPTXAS::ADDRESS_SPACE_SHARED)
6226 Args[0] = Builder.CreateAddrSpaceCast(
6227 V: Args[0], DestTy: Builder.getPtrTy(AddrSpace: NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6228
6229 // Append the missing trailing arguments with default values (cta_group,
6230 // flag_valid_pattern).
6231 while (Args.size() < NewFn->getFunctionType()->getNumParams())
6232 Args.push_back(Elt: Builder.getInt32(C: 0));
6233
6234 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6235 NewCall->takeName(V: CI);
6236 CI->replaceAllUsesWith(V: NewCall);
6237 CI->eraseFromParent();
6238 return;
6239 }
6240
6241#define G2S_CTA_CASE(ID_SUFFIX, NAME) \
6242 case Intrinsic::nvvm_cp_async_bulk_tensor_g2s_cta_##ID_SUFFIX:
6243 NVVM_TMA_G2S_MODES(G2S_CTA_CASE)
6244#undef G2S_CTA_CASE
6245 {
6246 SmallVector<Value *, 16> Args(CI->args());
6247 // Append the missing trailing flag_valid_pattern argument with default
6248 // value 0.
6249 assert(Args.size() + 1 == NewFn->getFunctionType()->getNumParams() &&
6250 "expected only the trailing flag_valid_pattern to be missing");
6251 Args.push_back(Elt: Builder.getInt32(C: 0));
6252
6253 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6254 NewCall->takeName(V: CI);
6255 CI->replaceAllUsesWith(V: NewCall);
6256 CI->eraseFromParent();
6257 return;
6258 }
6259#undef NVVM_TMA_G2S_MODES
6260 // clang-format on
6261
6262 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_1d:
6263 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_2d:
6264 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_3d:
6265 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_4d:
6266 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_5d:
6267 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_3d:
6268 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_4d:
6269 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_5d: {
6270 StringRef Name = F->getName();
6271 Name.consume_front(Prefix: "llvm.nvvm.cp.async.bulk.tensor.reduce.");
6272 auto RedOp = getNVPTXTMAReductionOp(Name: Name.split(Separator: '.').first);
6273
6274 SmallVector<Value *, 16> Args(CI->args());
6275 Args.insert(I: Args.end() - 1, Elt: Builder.getInt32(C: *RedOp));
6276 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6277 break;
6278 }
6279 case Intrinsic::nvvm_tcgen05_alloc_cg1:
6280 case Intrinsic::nvvm_tcgen05_alloc_cg2:
6281 case Intrinsic::nvvm_tcgen05_dealloc_cg1:
6282 case Intrinsic::nvvm_tcgen05_dealloc_cg2:
6283 NewCall =
6284 Builder.CreateCall(Callee: NewFn, Args: {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
6285 Builder.getFalse()});
6286 break;
6287 case Intrinsic::nvvm_mbarrier_init: {
6288 SmallVector<Value *, 3> Args(CI->args());
6289 // The .shared variant folded into the overloaded form without gaining an
6290 // operand, so only the pre-layout two-argument form needs one appended.
6291 if (Args.size() == 2)
6292 Args.push_back(Elt: Builder.getInt32(C: 0)); // layout = default(0)
6293 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6294 break;
6295 }
6296 case Intrinsic::riscv_sha256sig0:
6297 case Intrinsic::riscv_sha256sig1:
6298 case Intrinsic::riscv_sha256sum0:
6299 case Intrinsic::riscv_sha256sum1:
6300 case Intrinsic::riscv_sm3p0:
6301 case Intrinsic::riscv_sm3p1: {
6302 // The last argument to these intrinsics used to be i8 and changed to i32.
6303 // The type overload for sm4ks and sm4ed was removed.
6304 if (!CI->getType()->isIntegerTy(BitWidth: 64))
6305 return;
6306
6307 Value *Arg =
6308 Builder.CreateTrunc(V: CI->getArgOperand(i: 0), DestTy: Builder.getInt32Ty());
6309
6310 NewCall = Builder.CreateCall(Callee: NewFn, Args: Arg);
6311 Value *Res =
6312 Builder.CreateIntCast(V: NewCall, DestTy: CI->getType(), /*isSigned*/ true);
6313 NewCall->takeName(V: CI);
6314 CI->replaceAllUsesWith(V: Res);
6315 CI->eraseFromParent();
6316 return;
6317 }
6318
6319 case Intrinsic::x86_xop_vfrcz_ss:
6320 case Intrinsic::x86_xop_vfrcz_sd:
6321 NewCall = Builder.CreateCall(Callee: NewFn, Args: {CI->getArgOperand(i: 1)});
6322 break;
6323
6324 case Intrinsic::x86_xop_vpermil2pd:
6325 case Intrinsic::x86_xop_vpermil2ps:
6326 case Intrinsic::x86_xop_vpermil2pd_256:
6327 case Intrinsic::x86_xop_vpermil2ps_256: {
6328 SmallVector<Value *, 4> Args(CI->args());
6329 VectorType *FltIdxTy = cast<VectorType>(Val: Args[2]->getType());
6330 VectorType *IntIdxTy = VectorType::getInteger(VTy: FltIdxTy);
6331 Args[2] = Builder.CreateBitCast(V: Args[2], DestTy: IntIdxTy);
6332 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6333 break;
6334 }
6335
6336 case Intrinsic::x86_sse41_ptestc:
6337 case Intrinsic::x86_sse41_ptestz:
6338 case Intrinsic::x86_sse41_ptestnzc: {
6339 // The arguments for these intrinsics used to be v4f32, and changed
6340 // to v2i64. This is purely a nop, since those are bitwise intrinsics.
6341 // So, the only thing required is a bitcast for both arguments.
6342 // First, check the arguments have the old type.
6343 Value *Arg0 = CI->getArgOperand(i: 0);
6344 if (Arg0->getType() != FixedVectorType::get(ElementType: Type::getFloatTy(C), NumElts: 4))
6345 return;
6346
6347 // Old intrinsic, add bitcasts
6348 Value *Arg1 = CI->getArgOperand(i: 1);
6349
6350 auto *NewVecTy = FixedVectorType::get(ElementType: Type::getInt64Ty(C), NumElts: 2);
6351
6352 Value *BC0 = Builder.CreateBitCast(V: Arg0, DestTy: NewVecTy, Name: "cast");
6353 Value *BC1 = Builder.CreateBitCast(V: Arg1, DestTy: NewVecTy, Name: "cast");
6354
6355 NewCall = Builder.CreateCall(Callee: NewFn, Args: {BC0, BC1});
6356 break;
6357 }
6358
6359 case Intrinsic::x86_rdtscp: {
6360 // This used to take 1 arguments. If we have no arguments, it is already
6361 // upgraded.
6362 if (CI->getNumOperands() == 0)
6363 return;
6364
6365 NewCall = Builder.CreateCall(Callee: NewFn);
6366 // Extract the second result and store it.
6367 Value *Data = Builder.CreateExtractValue(Agg: NewCall, Idxs: 1);
6368 Builder.CreateAlignedStore(Val: Data, Ptr: CI->getArgOperand(i: 0), Align: Align(1));
6369 // Replace the original call result with the first result of the new call.
6370 Value *TSC = Builder.CreateExtractValue(Agg: NewCall, Idxs: 0);
6371
6372 NewCall->takeName(V: CI);
6373 CI->replaceAllUsesWith(V: TSC);
6374 CI->eraseFromParent();
6375 return;
6376 }
6377
6378 case Intrinsic::x86_sse41_insertps:
6379 case Intrinsic::x86_sse41_dppd:
6380 case Intrinsic::x86_sse41_dpps:
6381 case Intrinsic::x86_sse41_mpsadbw:
6382 case Intrinsic::x86_avx_dp_ps_256:
6383 case Intrinsic::x86_avx2_mpsadbw: {
6384 // Need to truncate the last argument from i32 to i8 -- this argument models
6385 // an inherently 8-bit immediate operand to these x86 instructions.
6386 SmallVector<Value *, 4> Args(CI->args());
6387
6388 // Replace the last argument with a trunc.
6389 Args.back() = Builder.CreateTrunc(V: Args.back(), DestTy: Type::getInt8Ty(C), Name: "trunc");
6390 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6391 break;
6392 }
6393
6394 case Intrinsic::x86_avx512_mask_cmp_pd_128:
6395 case Intrinsic::x86_avx512_mask_cmp_pd_256:
6396 case Intrinsic::x86_avx512_mask_cmp_pd_512:
6397 case Intrinsic::x86_avx512_mask_cmp_ps_128:
6398 case Intrinsic::x86_avx512_mask_cmp_ps_256:
6399 case Intrinsic::x86_avx512_mask_cmp_ps_512: {
6400 SmallVector<Value *, 4> Args(CI->args());
6401 unsigned NumElts =
6402 cast<FixedVectorType>(Val: Args[0]->getType())->getNumElements();
6403 Args[3] = getX86MaskVec(Builder, Mask: Args[3], NumElts);
6404
6405 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6406 Value *Res = applyX86MaskOn1BitsVec(Builder, Vec: NewCall, Mask: nullptr);
6407
6408 NewCall->takeName(V: CI);
6409 CI->replaceAllUsesWith(V: Res);
6410 CI->eraseFromParent();
6411 return;
6412 }
6413
6414 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_128:
6415 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_256:
6416 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_512:
6417 case Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128:
6418 case Intrinsic::x86_avx512bf16_cvtneps2bf16_256:
6419 case Intrinsic::x86_avx512bf16_cvtneps2bf16_512: {
6420 SmallVector<Value *, 4> Args(CI->args());
6421 unsigned NumElts = cast<FixedVectorType>(Val: CI->getType())->getNumElements();
6422 if (NewFn->getIntrinsicID() ==
6423 Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128)
6424 Args[1] = Builder.CreateBitCast(
6425 V: Args[1], DestTy: FixedVectorType::get(ElementType: Builder.getBFloatTy(), NumElts));
6426
6427 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6428 Value *Res = Builder.CreateBitCast(
6429 V: NewCall, DestTy: FixedVectorType::get(ElementType: Builder.getInt16Ty(), NumElts));
6430
6431 NewCall->takeName(V: CI);
6432 CI->replaceAllUsesWith(V: Res);
6433 CI->eraseFromParent();
6434 return;
6435 }
6436 case Intrinsic::x86_avx512bf16_dpbf16ps_128:
6437 case Intrinsic::x86_avx512bf16_dpbf16ps_256:
6438 case Intrinsic::x86_avx512bf16_dpbf16ps_512:{
6439 SmallVector<Value *, 4> Args(CI->args());
6440 unsigned NumElts =
6441 cast<FixedVectorType>(Val: CI->getType())->getNumElements() * 2;
6442 Args[1] = Builder.CreateBitCast(
6443 V: Args[1], DestTy: FixedVectorType::get(ElementType: Builder.getBFloatTy(), NumElts));
6444 Args[2] = Builder.CreateBitCast(
6445 V: Args[2], DestTy: FixedVectorType::get(ElementType: Builder.getBFloatTy(), NumElts));
6446
6447 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6448 break;
6449 }
6450
6451 case Intrinsic::thread_pointer: {
6452 NewCall = Builder.CreateCall(Callee: NewFn, Args: {});
6453 break;
6454 }
6455
6456 case Intrinsic::memcpy:
6457 case Intrinsic::memmove:
6458 case Intrinsic::memset: {
6459 // We have to make sure that the call signature is what we're expecting.
6460 // We only want to change the old signatures by removing the alignment arg:
6461 // @llvm.mem[cpy|move]...(i8*, i8*, i[32|i64], i32, i1)
6462 // -> @llvm.mem[cpy|move]...(i8*, i8*, i[32|i64], i1)
6463 // @llvm.memset...(i8*, i8, i[32|64], i32, i1)
6464 // -> @llvm.memset...(i8*, i8, i[32|64], i1)
6465 // Note: i8*'s in the above can be any pointer type
6466 if (CI->arg_size() != 5) {
6467 DefaultCase();
6468 return;
6469 }
6470 // Remove alignment argument (3), and add alignment attributes to the
6471 // dest/src pointers.
6472 Value *Args[4] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
6473 CI->getArgOperand(i: 2), CI->getArgOperand(i: 4)};
6474 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6475 AttributeList OldAttrs = CI->getAttributes();
6476 AttributeList NewAttrs = AttributeList::get(
6477 C, FnAttrs: OldAttrs.getFnAttrs(), RetAttrs: OldAttrs.getRetAttrs(),
6478 ArgAttrs: {OldAttrs.getParamAttrs(ArgNo: 0), OldAttrs.getParamAttrs(ArgNo: 1),
6479 OldAttrs.getParamAttrs(ArgNo: 2), OldAttrs.getParamAttrs(ArgNo: 4)});
6480 NewCall->setAttributes(NewAttrs);
6481 auto *MemCI = cast<MemIntrinsic>(Val: NewCall);
6482 // All mem intrinsics support dest alignment.
6483 const ConstantInt *Align = cast<ConstantInt>(Val: CI->getArgOperand(i: 3));
6484 MemCI->setDestAlignment(Align->getMaybeAlignValue());
6485 // Memcpy/Memmove also support source alignment.
6486 if (auto *MTI = dyn_cast<MemTransferInst>(Val: MemCI))
6487 MTI->setSourceAlignment(Align->getMaybeAlignValue());
6488 break;
6489 }
6490
6491 case Intrinsic::masked_load:
6492 case Intrinsic::masked_gather:
6493 case Intrinsic::masked_store:
6494 case Intrinsic::masked_scatter: {
6495 if (CI->arg_size() != 4) {
6496 DefaultCase();
6497 return;
6498 }
6499
6500 auto GetMaybeAlign = [](Value *Op) {
6501 if (auto *CI = dyn_cast<ConstantInt>(Val: Op)) {
6502 uint64_t Val = CI->getZExtValue();
6503 if (Val == 0)
6504 return MaybeAlign();
6505 if (isPowerOf2_64(Value: Val))
6506 return MaybeAlign(Val);
6507 }
6508 reportFatalUsageError(reason: "Invalid alignment argument");
6509 };
6510 auto GetAlign = [&](Value *Op) {
6511 MaybeAlign Align = GetMaybeAlign(Op);
6512 if (Align)
6513 return *Align;
6514 reportFatalUsageError(reason: "Invalid zero alignment argument");
6515 };
6516
6517 const DataLayout &DL = CI->getDataLayout();
6518 switch (NewFn->getIntrinsicID()) {
6519 case Intrinsic::masked_load:
6520 NewCall = Builder.CreateMaskedLoad(
6521 Ty: CI->getType(), Ptr: CI->getArgOperand(i: 0), Alignment: GetAlign(CI->getArgOperand(i: 1)),
6522 Mask: CI->getArgOperand(i: 2), PassThru: CI->getArgOperand(i: 3));
6523 break;
6524 case Intrinsic::masked_gather:
6525 NewCall = Builder.CreateMaskedGather(
6526 Ty: CI->getType(), Ptrs: CI->getArgOperand(i: 0),
6527 Alignment: DL.getValueOrABITypeAlignment(Alignment: GetMaybeAlign(CI->getArgOperand(i: 1)),
6528 Ty: CI->getType()->getScalarType()),
6529 Mask: CI->getArgOperand(i: 2), PassThru: CI->getArgOperand(i: 3));
6530 break;
6531 case Intrinsic::masked_store:
6532 NewCall = Builder.CreateMaskedStore(
6533 Val: CI->getArgOperand(i: 0), Ptr: CI->getArgOperand(i: 1),
6534 Alignment: GetAlign(CI->getArgOperand(i: 2)), Mask: CI->getArgOperand(i: 3));
6535 break;
6536 case Intrinsic::masked_scatter:
6537 NewCall = Builder.CreateMaskedScatter(
6538 Val: CI->getArgOperand(i: 0), Ptrs: CI->getArgOperand(i: 1),
6539 Alignment: DL.getValueOrABITypeAlignment(
6540 Alignment: GetMaybeAlign(CI->getArgOperand(i: 2)),
6541 Ty: CI->getArgOperand(i: 0)->getType()->getScalarType()),
6542 Mask: CI->getArgOperand(i: 3));
6543 break;
6544 default:
6545 llvm_unreachable("Unexpected intrinsic ID");
6546 }
6547 // Previous metadata is still valid.
6548 NewCall->copyMetadata(SrcInst: *CI);
6549 NewCall->setTailCallKind(cast<CallInst>(Val: CI)->getTailCallKind());
6550 break;
6551 }
6552
6553 case Intrinsic::lifetime_start:
6554 case Intrinsic::lifetime_end: {
6555 if (CI->arg_size() != 2) {
6556 DefaultCase();
6557 return;
6558 }
6559
6560 Value *Ptr = CI->getArgOperand(i: 1);
6561 // Try to strip pointer casts, such that the lifetime works on an alloca.
6562 Ptr = Ptr->stripPointerCasts();
6563 if (isa<AllocaInst>(Val: Ptr)) {
6564 // Don't use NewFn, as we might have looked through an addrspacecast.
6565 if (NewFn->getIntrinsicID() == Intrinsic::lifetime_start)
6566 NewCall = Builder.CreateLifetimeStart(Ptr);
6567 else
6568 NewCall = Builder.CreateLifetimeEnd(Ptr);
6569 break;
6570 }
6571
6572 // Otherwise remove the lifetime marker.
6573 CI->eraseFromParent();
6574 return;
6575 }
6576
6577 case Intrinsic::x86_avx512_vpdpbusd_128:
6578 case Intrinsic::x86_avx512_vpdpbusd_256:
6579 case Intrinsic::x86_avx512_vpdpbusd_512:
6580 case Intrinsic::x86_avx512_vpdpbusds_128:
6581 case Intrinsic::x86_avx512_vpdpbusds_256:
6582 case Intrinsic::x86_avx512_vpdpbusds_512:
6583 case Intrinsic::x86_avx2_vpdpbssd_128:
6584 case Intrinsic::x86_avx2_vpdpbssd_256:
6585 case Intrinsic::x86_avx10_vpdpbssd_512:
6586 case Intrinsic::x86_avx2_vpdpbssds_128:
6587 case Intrinsic::x86_avx2_vpdpbssds_256:
6588 case Intrinsic::x86_avx10_vpdpbssds_512:
6589 case Intrinsic::x86_avx2_vpdpbsud_128:
6590 case Intrinsic::x86_avx2_vpdpbsud_256:
6591 case Intrinsic::x86_avx10_vpdpbsud_512:
6592 case Intrinsic::x86_avx2_vpdpbsuds_128:
6593 case Intrinsic::x86_avx2_vpdpbsuds_256:
6594 case Intrinsic::x86_avx10_vpdpbsuds_512:
6595 case Intrinsic::x86_avx2_vpdpbuud_128:
6596 case Intrinsic::x86_avx2_vpdpbuud_256:
6597 case Intrinsic::x86_avx10_vpdpbuud_512:
6598 case Intrinsic::x86_avx2_vpdpbuuds_128:
6599 case Intrinsic::x86_avx2_vpdpbuuds_256:
6600 case Intrinsic::x86_avx10_vpdpbuuds_512: {
6601 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() / 8;
6602 Value *Args[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
6603 CI->getArgOperand(i: 2)};
6604 Type *NewArgType = VectorType::get(ElementType: Builder.getInt8Ty(), NumElements: NumElts, Scalable: false);
6605 Args[1] = Builder.CreateBitCast(V: Args[1], DestTy: NewArgType);
6606 Args[2] = Builder.CreateBitCast(V: Args[2], DestTy: NewArgType);
6607
6608 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6609 break;
6610 }
6611 case Intrinsic::x86_avx512_vpdpwssd_128:
6612 case Intrinsic::x86_avx512_vpdpwssd_256:
6613 case Intrinsic::x86_avx512_vpdpwssd_512:
6614 case Intrinsic::x86_avx512_vpdpwssds_128:
6615 case Intrinsic::x86_avx512_vpdpwssds_256:
6616 case Intrinsic::x86_avx512_vpdpwssds_512:
6617 case Intrinsic::x86_avx2_vpdpwsud_128:
6618 case Intrinsic::x86_avx2_vpdpwsud_256:
6619 case Intrinsic::x86_avx10_vpdpwsud_512:
6620 case Intrinsic::x86_avx2_vpdpwsuds_128:
6621 case Intrinsic::x86_avx2_vpdpwsuds_256:
6622 case Intrinsic::x86_avx10_vpdpwsuds_512:
6623 case Intrinsic::x86_avx2_vpdpwusd_128:
6624 case Intrinsic::x86_avx2_vpdpwusd_256:
6625 case Intrinsic::x86_avx10_vpdpwusd_512:
6626 case Intrinsic::x86_avx2_vpdpwusds_128:
6627 case Intrinsic::x86_avx2_vpdpwusds_256:
6628 case Intrinsic::x86_avx10_vpdpwusds_512:
6629 case Intrinsic::x86_avx2_vpdpwuud_128:
6630 case Intrinsic::x86_avx2_vpdpwuud_256:
6631 case Intrinsic::x86_avx10_vpdpwuud_512:
6632 case Intrinsic::x86_avx2_vpdpwuuds_128:
6633 case Intrinsic::x86_avx2_vpdpwuuds_256:
6634 case Intrinsic::x86_avx10_vpdpwuuds_512:
6635 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() / 16;
6636 Value *Args[] = {CI->getArgOperand(i: 0), CI->getArgOperand(i: 1),
6637 CI->getArgOperand(i: 2)};
6638 Type *NewArgType = VectorType::get(ElementType: Builder.getInt16Ty(), NumElements: NumElts, Scalable: false);
6639 Args[1] = Builder.CreateBitCast(V: Args[1], DestTy: NewArgType);
6640 Args[2] = Builder.CreateBitCast(V: Args[2], DestTy: NewArgType);
6641
6642 NewCall = Builder.CreateCall(Callee: NewFn, Args);
6643 break;
6644 }
6645 assert(NewCall && "Should have either set this variable or returned through "
6646 "the default case");
6647 NewCall->takeName(V: CI);
6648 CI->replaceAllUsesWith(V: NewCall);
6649 CI->eraseFromParent();
6650}
6651
6652void llvm::UpgradeCallsToIntrinsic(Function *F) {
6653 assert(F && "Illegal attempt to upgrade a non-existent intrinsic.");
6654
6655 // Check if this function should be upgraded and get the replacement function
6656 // if there is one.
6657 Function *NewFn;
6658 if (UpgradeIntrinsicFunction(F, NewFn)) {
6659 // Replace all users of the old function with the new function or new
6660 // instructions. This is not a range loop because the call is deleted.
6661 for (User *U : make_early_inc_range(Range: F->users()))
6662 if (CallBase *CB = dyn_cast<CallBase>(Val: U))
6663 UpgradeIntrinsicCall(CI: CB, NewFn);
6664
6665 // Remove old function, no longer used, from the module.
6666 if (F != NewFn)
6667 F->eraseFromParent();
6668 }
6669}
6670
6671MDNode *llvm::UpgradeTBAANode(MDNode &MD) {
6672 const unsigned NumOperands = MD.getNumOperands();
6673 if (NumOperands == 0)
6674 return &MD; // Invalid, punt to a verifier error.
6675
6676 // Check if the tag uses struct-path aware TBAA format.
6677 if (isa<MDNode>(Val: MD.getOperand(I: 0)) && NumOperands >= 3)
6678 return &MD;
6679
6680 auto &Context = MD.getContext();
6681 if (NumOperands == 3) {
6682 Metadata *Elts[] = {MD.getOperand(I: 0), MD.getOperand(I: 1)};
6683 MDNode *ScalarType = MDNode::get(Context, MDs: Elts);
6684 // Create a MDNode <ScalarType, ScalarType, offset 0, const>
6685 Metadata *Elts2[] = {ScalarType, ScalarType,
6686 ConstantAsMetadata::get(
6687 C: Constant::getNullValue(Ty: Type::getInt64Ty(C&: Context))),
6688 MD.getOperand(I: 2)};
6689 return MDNode::get(Context, MDs: Elts2);
6690 }
6691 // Create a MDNode <MD, MD, offset 0>
6692 Metadata *Elts[] = {&MD, &MD, ConstantAsMetadata::get(C: Constant::getNullValue(
6693 Ty: Type::getInt64Ty(C&: Context)))};
6694 return MDNode::get(Context, MDs: Elts);
6695}
6696
6697MDNode *llvm::UpgradeTBAAStructNode(MDNode &MD) {
6698 // !tbaa.struct is a list of (offset, size, tag) triples. Upgrade any
6699 // old-style scalar field tag to struct-path form via UpgradeTBAANode.
6700 unsigned NumOperands = MD.getNumOperands();
6701 if (NumOperands == 0 || NumOperands % 3 != 0)
6702 return &MD; // Malformed; leave it for the verifier to reject.
6703
6704 SmallVector<Metadata *, 12> Elts(MD.op_begin(), MD.op_end());
6705 bool Changed = false;
6706 for (unsigned I = 2; I < NumOperands; I += 3) {
6707 auto *Tag = dyn_cast_or_null<MDNode>(Val: Elts[I]);
6708 if (!Tag)
6709 continue;
6710 MDNode *Upgraded = UpgradeTBAANode(MD&: *Tag);
6711 if (Upgraded == Tag)
6712 continue;
6713 Elts[I] = Upgraded;
6714 Changed = true;
6715 }
6716 return Changed ? MDNode::get(Context&: MD.getContext(), MDs: Elts) : &MD;
6717}
6718
6719Instruction *llvm::UpgradeBitCastInst(unsigned Opc, Value *V, Type *DestTy,
6720 Instruction *&Temp) {
6721 if (Opc != Instruction::BitCast)
6722 return nullptr;
6723
6724 Temp = nullptr;
6725 Type *SrcTy = V->getType();
6726 if (SrcTy->isPtrOrPtrVectorTy() && DestTy->isPtrOrPtrVectorTy() &&
6727 SrcTy->getPointerAddressSpace() != DestTy->getPointerAddressSpace()) {
6728 LLVMContext &Context = V->getContext();
6729
6730 // We have no information about target data layout, so we assume that
6731 // the maximum pointer size is 64bit.
6732 Type *MidTy = Type::getInt64Ty(C&: Context);
6733 Temp = CastInst::Create(Instruction::PtrToInt, S: V, Ty: MidTy);
6734
6735 return CastInst::Create(Instruction::IntToPtr, S: Temp, Ty: DestTy);
6736 }
6737
6738 return nullptr;
6739}
6740
6741Constant *llvm::UpgradeBitCastExpr(unsigned Opc, Constant *C, Type *DestTy) {
6742 if (Opc != Instruction::BitCast)
6743 return nullptr;
6744
6745 Type *SrcTy = C->getType();
6746 if (SrcTy->isPtrOrPtrVectorTy() && DestTy->isPtrOrPtrVectorTy() &&
6747 SrcTy->getPointerAddressSpace() != DestTy->getPointerAddressSpace()) {
6748 LLVMContext &Context = C->getContext();
6749
6750 // We have no information about target data layout, so we assume that
6751 // the maximum pointer size is 64bit.
6752 Type *MidTy = Type::getInt64Ty(C&: Context);
6753
6754 return ConstantExpr::getIntToPtr(C: ConstantExpr::getPtrToInt(C, Ty: MidTy),
6755 Ty: DestTy);
6756 }
6757
6758 return nullptr;
6759}
6760
6761static std::optional<StringRef> getModuleFlagNameSafely(const MDNode &Flag) {
6762 if (Flag.getNumOperands() < 3)
6763 return std::nullopt;
6764 if (MDString *Name = dyn_cast_or_null<MDString>(Val: Flag.getOperand(I: 1)))
6765 return Name->getString();
6766 return std::nullopt;
6767}
6768
6769/// Check the debug info version number, if it is out-dated, drop the debug
6770/// info. Return true if module is modified.
6771bool llvm::UpgradeDebugInfo(Module &M) {
6772 if (DisableAutoUpgradeDebugInfo)
6773 return false;
6774
6775 llvm::TimeTraceScope timeScope("Upgrade debug info");
6776 // We need to get metadata before the module is verified (i.e., getModuleFlag
6777 // makes assumptions that we haven't verified yet). Carefully extract the flag
6778 // from the metadata.
6779 unsigned Version = 0;
6780 if (NamedMDNode *ModFlags = M.getModuleFlagsMetadata()) {
6781 auto OpIt = find_if(Range: ModFlags->operands(), P: [](const MDNode *Flag) {
6782 if (auto Name = getModuleFlagNameSafely(Flag: *Flag))
6783 return *Name == "Debug Info Version";
6784 return false;
6785 });
6786 if (OpIt != ModFlags->op_end()) {
6787 const MDOperand &ValOp = (*OpIt)->getOperand(I: 2);
6788 if (auto *CI = mdconst::dyn_extract_or_null<ConstantInt>(MD: ValOp))
6789 Version = CI->getZExtValue();
6790 }
6791 }
6792
6793 if (Version == DEBUG_METADATA_VERSION) {
6794 bool BrokenDebugInfo = false;
6795 if (verifyModule(M, OS: &llvm::errs(), BrokenDebugInfo: &BrokenDebugInfo))
6796 report_fatal_error(reason: "Broken module found, compilation aborted!");
6797 if (!BrokenDebugInfo)
6798 // Everything is ok.
6799 return false;
6800 else {
6801 // Diagnose malformed debug info.
6802 DiagnosticInfoIgnoringInvalidDebugMetadata Diag(M);
6803 M.getContext().diagnose(DI: Diag);
6804 }
6805 }
6806 bool Modified = StripDebugInfo(M);
6807 if (Modified && Version != DEBUG_METADATA_VERSION) {
6808 // Diagnose a version mismatch.
6809 DiagnosticInfoDebugMetadataVersion DiagVersion(M, Version);
6810 M.getContext().diagnose(DI: DiagVersion);
6811 }
6812 return Modified;
6813}
6814
6815static void upgradeNVVMFnVectorAttr(const StringRef Attr, const char DimC,
6816 GlobalValue *GV, const Metadata *V) {
6817 Function *F = cast<Function>(Val: GV);
6818
6819 constexpr StringLiteral DefaultValue = "1";
6820 StringRef Vect3[3] = {DefaultValue, DefaultValue, DefaultValue};
6821 unsigned Length = 0;
6822
6823 if (F->hasFnAttribute(Kind: Attr)) {
6824 // We expect the existing attribute to have the form "x[,y[,z]]". Here we
6825 // parse these elements placing them into Vect3
6826 StringRef S = F->getFnAttribute(Kind: Attr).getValueAsString();
6827 for (; Length < 3 && !S.empty(); Length++) {
6828 auto [Part, Rest] = S.split(Separator: ',');
6829 Vect3[Length] = Part.trim();
6830 S = Rest;
6831 }
6832 }
6833
6834 const unsigned Dim = DimC - 'x';
6835 assert(Dim < 3 && "Unexpected dim char");
6836
6837 const uint64_t VInt = mdconst::extract<ConstantInt>(MD&: V)->getZExtValue();
6838
6839 // local variable required for StringRef in Vect3 to point to.
6840 const std::string VStr = llvm::utostr(X: VInt);
6841 Vect3[Dim] = VStr;
6842 Length = std::max(a: Length, b: Dim + 1);
6843
6844 const std::string NewAttr = llvm::join(R: ArrayRef(Vect3, Length), Separator: ",");
6845 F->addFnAttr(Kind: Attr, Val: NewAttr);
6846}
6847
6848static inline bool isXYZ(StringRef S) {
6849 return S == "x" || S == "y" || S == "z";
6850}
6851
6852bool static upgradeSingleNVVMAnnotation(GlobalValue *GV, StringRef K,
6853 const Metadata *V) {
6854 if (K == "kernel") {
6855 if (!mdconst::extract<ConstantInt>(MD&: V)->isZero())
6856 cast<Function>(Val: GV)->setCallingConv(CallingConv::PTX_Kernel);
6857 return true;
6858 }
6859 if (K == "align") {
6860 // V is a bitfeild specifying two 16-bit values. The alignment value is
6861 // specfied in low 16-bits, The index is specified in the high bits. For the
6862 // index, 0 indicates the return value while higher values correspond to
6863 // each parameter (idx = param + 1).
6864 const uint64_t AlignIdxValuePair =
6865 mdconst::extract<ConstantInt>(MD&: V)->getZExtValue();
6866 const unsigned Idx = (AlignIdxValuePair >> 16);
6867 const Align StackAlign = Align(AlignIdxValuePair & 0xFFFF);
6868 cast<Function>(Val: GV)->addAttributeAtIndex(
6869 i: Idx, Attr: Attribute::getWithStackAlignment(Context&: GV->getContext(), Alignment: StackAlign));
6870 return true;
6871 }
6872 if (K == "maxclusterrank" || K == "cluster_max_blocks") {
6873 const auto CV = mdconst::extract<ConstantInt>(MD&: V)->getZExtValue();
6874 cast<Function>(Val: GV)->addFnAttr(Kind: NVVMAttr::MaxClusterRank, Val: llvm::utostr(X: CV));
6875 return true;
6876 }
6877 if (K == "minctasm") {
6878 const auto CV = mdconst::extract<ConstantInt>(MD&: V)->getZExtValue();
6879 cast<Function>(Val: GV)->addFnAttr(Kind: NVVMAttr::MinCTASm, Val: llvm::utostr(X: CV));
6880 return true;
6881 }
6882 if (K == "maxnreg") {
6883 const auto CV = mdconst::extract<ConstantInt>(MD&: V)->getZExtValue();
6884 cast<Function>(Val: GV)->addFnAttr(Kind: NVVMAttr::MaxNReg, Val: llvm::utostr(X: CV));
6885 return true;
6886 }
6887 if (K.consume_front(Prefix: "maxntid") && isXYZ(S: K)) {
6888 upgradeNVVMFnVectorAttr(Attr: NVVMAttr::MaxNTID, DimC: K[0], GV, V);
6889 return true;
6890 }
6891 if (K.consume_front(Prefix: "reqntid") && isXYZ(S: K)) {
6892 upgradeNVVMFnVectorAttr(Attr: NVVMAttr::ReqNTID, DimC: K[0], GV, V);
6893 return true;
6894 }
6895 if (K.consume_front(Prefix: "cluster_dim_") && isXYZ(S: K)) {
6896 upgradeNVVMFnVectorAttr(Attr: NVVMAttr::ClusterDim, DimC: K[0], GV, V);
6897 return true;
6898 }
6899 if (K == "grid_constant") {
6900 const auto Attr = Attribute::get(Context&: GV->getContext(), Kind: NVVMAttr::GridConstant);
6901 for (const auto &Op : cast<MDNode>(Val: V)->operands()) {
6902 // For some reason, the index is 1-based in the metadata. Good thing we're
6903 // able to auto-upgrade it!
6904 const auto Index = mdconst::extract<ConstantInt>(MD: Op)->getZExtValue() - 1;
6905 cast<Function>(Val: GV)->addParamAttr(ArgNo: Index, Attr);
6906 }
6907 return true;
6908 }
6909
6910 return false;
6911}
6912
6913void llvm::UpgradeNVVMAnnotations(Module &M) {
6914 NamedMDNode *NamedMD = M.getNamedMetadata(Name: "nvvm.annotations");
6915 if (!NamedMD)
6916 return;
6917
6918 SmallVector<MDNode *, 8> NewNodes;
6919 SmallPtrSet<const MDNode *, 8> SeenNodes;
6920 for (MDNode *MD : NamedMD->operands()) {
6921 if (!SeenNodes.insert(Ptr: MD).second)
6922 continue;
6923
6924 auto *GV = mdconst::dyn_extract_or_null<GlobalValue>(MD: MD->getOperand(I: 0));
6925 if (!GV)
6926 continue;
6927
6928 assert((MD->getNumOperands() % 2) == 1 && "Invalid number of operands");
6929
6930 SmallVector<Metadata *, 8> NewOperands{MD->getOperand(I: 0)};
6931 // Each nvvm.annotations metadata entry will be of the following form:
6932 // !{ ptr @gv, !"key1", value1, !"key2", value2, ... }
6933 // start index = 1, to skip the global variable key
6934 // increment = 2, to skip the value for each property-value pairs
6935 for (unsigned j = 1, je = MD->getNumOperands(); j < je; j += 2) {
6936 MDString *K = cast<MDString>(Val: MD->getOperand(I: j));
6937 const MDOperand &V = MD->getOperand(I: j + 1);
6938 bool Upgraded = upgradeSingleNVVMAnnotation(GV, K: K->getString(), V);
6939 if (!Upgraded)
6940 NewOperands.append(IL: {K, V});
6941 }
6942
6943 if (NewOperands.size() > 1)
6944 NewNodes.push_back(Elt: MDNode::get(Context&: M.getContext(), MDs: NewOperands));
6945 }
6946
6947 NamedMD->clearOperands();
6948 for (MDNode *N : NewNodes)
6949 NamedMD->addOperand(M: N);
6950}
6951
6952/// This checks for objc retain release marker which should be upgraded. It
6953/// returns true if module is modified.
6954static bool upgradeRetainReleaseMarker(Module &M) {
6955 bool Changed = false;
6956 const char *MarkerKey = "clang.arc.retainAutoreleasedReturnValueMarker";
6957 NamedMDNode *ModRetainReleaseMarker = M.getNamedMetadata(Name: MarkerKey);
6958 if (ModRetainReleaseMarker) {
6959 MDNode *Op = ModRetainReleaseMarker->getOperand(i: 0);
6960 if (Op) {
6961 MDString *ID = dyn_cast_or_null<MDString>(Val: Op->getOperand(I: 0));
6962 if (ID) {
6963 SmallVector<StringRef, 4> ValueComp;
6964 ID->getString().split(A&: ValueComp, Separator: "#");
6965 if (ValueComp.size() == 2) {
6966 std::string NewValue = ValueComp[0].str() + ";" + ValueComp[1].str();
6967 ID = MDString::get(Context&: M.getContext(), Str: NewValue);
6968 }
6969 M.addModuleFlag(Behavior: Module::Error, Key: MarkerKey, Val: ID);
6970 M.eraseNamedMetadata(NMD: ModRetainReleaseMarker);
6971 Changed = true;
6972 }
6973 }
6974 }
6975 return Changed;
6976}
6977
6978void llvm::UpgradeARCRuntime(Module &M) {
6979 // This lambda converts normal function calls to ARC runtime functions to
6980 // intrinsic calls.
6981 auto UpgradeToIntrinsic = [&](const char *OldFunc,
6982 llvm::Intrinsic::ID IntrinsicFunc) {
6983 Function *Fn = M.getFunction(Name: OldFunc);
6984
6985 if (!Fn)
6986 return;
6987
6988 Function *NewFn =
6989 llvm::Intrinsic::getOrInsertDeclaration(M: &M, id: IntrinsicFunc);
6990
6991 for (User *U : make_early_inc_range(Range: Fn->users())) {
6992 CallInst *CI = dyn_cast<CallInst>(Val: U);
6993 if (!CI || CI->getCalledFunction() != Fn)
6994 continue;
6995
6996 IRBuilder<> Builder(CI->getIterator());
6997 FunctionType *NewFuncTy = NewFn->getFunctionType();
6998 SmallVector<Value *, 2> Args;
6999
7000 // Don't upgrade the intrinsic if it's not valid to bitcast the return
7001 // value to the return type of the old function.
7002 if (NewFuncTy->getReturnType() != CI->getType() &&
7003 !CastInst::castIsValid(op: Instruction::BitCast, S: CI,
7004 DstTy: NewFuncTy->getReturnType()))
7005 continue;
7006
7007 bool InvalidCast = false;
7008
7009 for (unsigned I = 0, E = CI->arg_size(); I != E; ++I) {
7010 Value *Arg = CI->getArgOperand(i: I);
7011
7012 // Bitcast argument to the parameter type of the new function if it's
7013 // not a variadic argument.
7014 if (I < NewFuncTy->getNumParams()) {
7015 // Don't upgrade the intrinsic if it's not valid to bitcast the argument
7016 // to the parameter type of the new function.
7017 if (!CastInst::castIsValid(op: Instruction::BitCast, S: Arg,
7018 DstTy: NewFuncTy->getParamType(i: I))) {
7019 InvalidCast = true;
7020 break;
7021 }
7022 Arg = Builder.CreateBitCast(V: Arg, DestTy: NewFuncTy->getParamType(i: I));
7023 }
7024 Args.push_back(Elt: Arg);
7025 }
7026
7027 if (InvalidCast)
7028 continue;
7029
7030 // Create a call instruction that calls the new function.
7031 CallInst *NewCall = Builder.CreateCall(FTy: NewFuncTy, Callee: NewFn, Args);
7032 NewCall->setTailCallKind(cast<CallInst>(Val: CI)->getTailCallKind());
7033 NewCall->takeName(V: CI);
7034
7035 // Bitcast the return value back to the type of the old call.
7036 Value *NewRetVal = Builder.CreateBitCast(V: NewCall, DestTy: CI->getType());
7037
7038 if (!CI->use_empty())
7039 CI->replaceAllUsesWith(V: NewRetVal);
7040 CI->eraseFromParent();
7041 }
7042
7043 if (Fn->use_empty())
7044 Fn->eraseFromParent();
7045 };
7046
7047 // Unconditionally convert a call to "clang.arc.use" to a call to
7048 // "llvm.objc.clang.arc.use".
7049 UpgradeToIntrinsic("clang.arc.use", llvm::Intrinsic::objc_clang_arc_use);
7050
7051 // Upgrade the retain release marker. If there is no need to upgrade
7052 // the marker, that means either the module is already new enough to contain
7053 // new intrinsics or it is not ARC. There is no need to upgrade runtime call.
7054 if (!upgradeRetainReleaseMarker(M))
7055 return;
7056
7057 std::pair<const char *, llvm::Intrinsic::ID> RuntimeFuncs[] = {
7058 {"objc_autorelease", llvm::Intrinsic::objc_autorelease},
7059 {"objc_autoreleasePoolPop", llvm::Intrinsic::objc_autoreleasePoolPop},
7060 {"objc_autoreleasePoolPush", llvm::Intrinsic::objc_autoreleasePoolPush},
7061 {"objc_autoreleaseReturnValue",
7062 llvm::Intrinsic::objc_autoreleaseReturnValue},
7063 {"objc_copyWeak", llvm::Intrinsic::objc_copyWeak},
7064 {"objc_destroyWeak", llvm::Intrinsic::objc_destroyWeak},
7065 {"objc_initWeak", llvm::Intrinsic::objc_initWeak},
7066 {"objc_loadWeak", llvm::Intrinsic::objc_loadWeak},
7067 {"objc_loadWeakRetained", llvm::Intrinsic::objc_loadWeakRetained},
7068 {"objc_moveWeak", llvm::Intrinsic::objc_moveWeak},
7069 {"objc_release", llvm::Intrinsic::objc_release},
7070 {"objc_retain", llvm::Intrinsic::objc_retain},
7071 {"objc_retainAutorelease", llvm::Intrinsic::objc_retainAutorelease},
7072 {"objc_retainAutoreleaseReturnValue",
7073 llvm::Intrinsic::objc_retainAutoreleaseReturnValue},
7074 {"objc_retainAutoreleasedReturnValue",
7075 llvm::Intrinsic::objc_retainAutoreleasedReturnValue},
7076 {"objc_retainBlock", llvm::Intrinsic::objc_retainBlock},
7077 {"objc_storeStrong", llvm::Intrinsic::objc_storeStrong},
7078 {"objc_storeWeak", llvm::Intrinsic::objc_storeWeak},
7079 {"objc_unsafeClaimAutoreleasedReturnValue",
7080 llvm::Intrinsic::objc_unsafeClaimAutoreleasedReturnValue},
7081 {"objc_retainedObject", llvm::Intrinsic::objc_retainedObject},
7082 {"objc_unretainedObject", llvm::Intrinsic::objc_unretainedObject},
7083 {"objc_unretainedPointer", llvm::Intrinsic::objc_unretainedPointer},
7084 {"objc_retain_autorelease", llvm::Intrinsic::objc_retain_autorelease},
7085 {"objc_sync_enter", llvm::Intrinsic::objc_sync_enter},
7086 {"objc_sync_exit", llvm::Intrinsic::objc_sync_exit},
7087 {"objc_arc_annotation_topdown_bbstart",
7088 llvm::Intrinsic::objc_arc_annotation_topdown_bbstart},
7089 {"objc_arc_annotation_topdown_bbend",
7090 llvm::Intrinsic::objc_arc_annotation_topdown_bbend},
7091 {"objc_arc_annotation_bottomup_bbstart",
7092 llvm::Intrinsic::objc_arc_annotation_bottomup_bbstart},
7093 {"objc_arc_annotation_bottomup_bbend",
7094 llvm::Intrinsic::objc_arc_annotation_bottomup_bbend}};
7095
7096 for (auto &I : RuntimeFuncs)
7097 UpgradeToIntrinsic(I.first, I.second);
7098}
7099
7100// Upgrade the way signing of pointers to init/fini functions is described.
7101//
7102// Originally, the `@llvm.global_(ctors|dtors)` arrays contained `ptrauth`
7103// constants, if signing was requested. After the upgrade, these arrays contain
7104// plain function pointers and the desired signing schema is described via a
7105// pair of module flags.
7106//
7107// Note that the upgrade is only performed if all elements of *both* arrays
7108// agree on a common signing schema.
7109static bool upgradePtrauthInitFiniArrays(Module &M) {
7110 // As we cannot always decide whether the particular module should have
7111 // ptrauth-init-fini flags, we have to treat absent flags as having zero
7112 // values for compatibility reasons. Thus, upgradePtrauthInitFiniArrays
7113 // returns as soon as it spots any non-signed init/fini pointer: either we
7114 // should request non-signed pointers (safe to omit both flags) or there is
7115 // no common schema (and thus we do not modify anything).
7116 //
7117 // UseAddressDisc's value either represents "not decided yet" state (nullopt)
7118 // or whether we should request address diversity in addition to the basic
7119 // constant diversity. There is no value representing "decided not to sign"
7120 // for the reasons explained above.
7121 std::optional<bool> UseAddressDisc;
7122
7123 // Do not attempt upgrading if the new module flags already exist.
7124 if (const NamedMDNode *ModFlags = M.getModuleFlagsMetadata()) {
7125 for (const MDNode *Flag : ModFlags->operands()) {
7126 std::optional<StringRef> Name = getModuleFlagNameSafely(Flag: *Flag);
7127 if (Name && (*Name == "ptrauth-init-fini" ||
7128 *Name == "ptrauth-init-fini-address-discrimination"))
7129 return false;
7130 }
7131 }
7132
7133 auto UpgradeSinglePointer = [&UseAddressDisc](Constant *CV) -> Constant * {
7134 constexpr unsigned ExpectedConstDisc = 0xD9D4;
7135 constexpr unsigned ExpectedAddressMarker = 1;
7136
7137 auto *CPA = dyn_cast<ConstantPtrAuth>(Val: CV);
7138 if (!CPA || !CPA->getDiscriminator()->equalsInt(V: ExpectedConstDisc))
7139 return nullptr; // Nothing to upgrade or unknown pattern found.
7140
7141 bool HasAddressDisc;
7142 if (!CPA->hasAddressDiscriminator())
7143 HasAddressDisc = false;
7144 else if (CPA->hasSpecialAddressDiscriminator(Value: ExpectedAddressMarker))
7145 HasAddressDisc = true;
7146 else
7147 return nullptr; // Unknown pattern.
7148
7149 if (UseAddressDisc && *UseAddressDisc != HasAddressDisc)
7150 return nullptr; // Disagreement with the decided mode.
7151
7152 UseAddressDisc = HasAddressDisc;
7153 return CPA->getPointer();
7154 };
7155
7156 // Do not apply any changes until we know the upgrade is non-ambiguous.
7157 using PendingUpgrade = std::pair<GlobalVariable *, Constant *>;
7158 SmallVector<PendingUpgrade, 2> GlobalArraysToUpgrade;
7159
7160 for (const char *Name : {"llvm.global_ctors", "llvm.global_dtors"}) {
7161 auto *GV = dyn_cast_if_present<GlobalVariable>(Val: M.getNamedValue(Name));
7162 if (!GV || !GV->hasInitializer())
7163 continue; // Skip, but it is okay to upgrade the other variable.
7164
7165 auto *OldStructorsArray = dyn_cast<ConstantArray>(Val: GV->getInitializer());
7166 if (!OldStructorsArray || OldStructorsArray->getNumOperands() == 0)
7167 return false;
7168
7169 std::vector<Constant *> NewStructors;
7170 NewStructors.reserve(n: OldStructorsArray->getNumOperands());
7171
7172 for (Use &U : OldStructorsArray->operands()) {
7173 ConstantStruct *Structor = dyn_cast<ConstantStruct>(Val: U.get());
7174 if (!Structor || Structor->getNumOperands() != 3)
7175 return false;
7176
7177 Constant *Prio = Structor->getOperand(i_nocapture: 0);
7178 Constant *Func = Structor->getOperand(i_nocapture: 1);
7179 Constant *Arg = Structor->getOperand(i_nocapture: 2);
7180
7181 Func = UpgradeSinglePointer(Func);
7182 if (!Func)
7183 return false;
7184
7185 NewStructors.push_back(
7186 x: ConstantStruct::get(T: Structor->getType(), V: {Prio, Func, Arg}));
7187 }
7188
7189 Constant *NewInit =
7190 ConstantArray::get(T: OldStructorsArray->getType(), V: NewStructors);
7191 GlobalArraysToUpgrade.emplace_back(Args&: GV, Args&: NewInit);
7192 }
7193
7194 if (GlobalArraysToUpgrade.empty())
7195 return false;
7196 assert(UseAddressDisc.has_value());
7197
7198 for (auto [GV, NewInit] : GlobalArraysToUpgrade)
7199 GV->setInitializer(NewInit);
7200
7201 M.addModuleFlag(Behavior: Module::Error, Key: "ptrauth-init-fini", Val: 1);
7202 M.addModuleFlag(Behavior: Module::Error, Key: "ptrauth-init-fini-address-discrimination",
7203 Val: *UseAddressDisc);
7204
7205 return true;
7206}
7207
7208bool llvm::UpgradeModuleFlags(Module &M) {
7209 bool Changed = false;
7210 Changed |= upgradePtrauthInitFiniArrays(M);
7211
7212 NamedMDNode *ModFlags = M.getModuleFlagsMetadata();
7213 if (!ModFlags)
7214 return Changed;
7215
7216 bool HasObjCFlag = false, HasClassProperties = false;
7217 bool HasSwiftVersionFlag = false;
7218 uint8_t SwiftMajorVersion, SwiftMinorVersion;
7219 uint32_t SwiftABIVersion;
7220 auto Int8Ty = Type::getInt8Ty(C&: M.getContext());
7221 auto Int32Ty = Type::getInt32Ty(C&: M.getContext());
7222
7223 for (unsigned I = 0, E = ModFlags->getNumOperands(); I != E; ++I) {
7224 MDNode *Op = ModFlags->getOperand(i: I);
7225 if (Op->getNumOperands() != 3)
7226 continue;
7227 MDString *ID = dyn_cast_or_null<MDString>(Val: Op->getOperand(I: 1));
7228 if (!ID)
7229 continue;
7230 auto SetBehavior = [&](Module::ModFlagBehavior B) {
7231 Metadata *Ops[3] = {ConstantAsMetadata::get(C: ConstantInt::get(
7232 Ty: Type::getInt32Ty(C&: M.getContext()), V: B)),
7233 MDString::get(Context&: M.getContext(), Str: ID->getString()),
7234 Op->getOperand(I: 2)};
7235 ModFlags->setOperand(I, New: MDNode::get(Context&: M.getContext(), MDs: Ops));
7236 Changed = true;
7237 };
7238
7239 if (ID->getString() == "Objective-C Image Info Version")
7240 HasObjCFlag = true;
7241 if (ID->getString() == "Objective-C Class Properties")
7242 HasClassProperties = true;
7243 // Upgrade PIC from Error/Max to Min.
7244 if (ID->getString() == "PIC Level") {
7245 if (auto *Behavior =
7246 mdconst::dyn_extract_or_null<ConstantInt>(MD: Op->getOperand(I: 0))) {
7247 uint64_t V = Behavior->getLimitedValue();
7248 if (V == Module::Error || V == Module::Max)
7249 SetBehavior(Module::Min);
7250 }
7251 }
7252 // Upgrade "PIE Level" from Error to Max.
7253 if (ID->getString() == "PIE Level")
7254 if (auto *Behavior =
7255 mdconst::dyn_extract_or_null<ConstantInt>(MD: Op->getOperand(I: 0)))
7256 if (Behavior->getLimitedValue() == Module::Error)
7257 SetBehavior(Module::Max);
7258
7259 // Upgrade branch protection and return address signing module flags. The
7260 // module flag behavior for these fields were Error and now they are Min.
7261 // The one exception is "sign-return-address-harden".
7262 if (ID->getString() == "branch-target-enforcement" ||
7263 (ID->getString().starts_with(Prefix: "sign-return-address") &&
7264 ID->getString() != "sign-return-address-harden")) {
7265 if (auto *Behavior =
7266 mdconst::dyn_extract_or_null<ConstantInt>(MD: Op->getOperand(I: 0))) {
7267 if (Behavior->getLimitedValue() == Module::Error) {
7268 Type *Int32Ty = Type::getInt32Ty(C&: M.getContext());
7269 Metadata *Ops[3] = {
7270 ConstantAsMetadata::get(C: ConstantInt::get(Ty: Int32Ty, V: Module::Min)),
7271 Op->getOperand(I: 1), Op->getOperand(I: 2)};
7272 ModFlags->setOperand(I, New: MDNode::get(Context&: M.getContext(), MDs: Ops));
7273 Changed = true;
7274 }
7275 }
7276 }
7277
7278 // Upgrade Objective-C Image Info Section. Removed the whitespce in the
7279 // section name so that llvm-lto will not complain about mismatching
7280 // module flags that is functionally the same.
7281 if (ID->getString() == "Objective-C Image Info Section") {
7282 if (auto *Value = dyn_cast_or_null<MDString>(Val: Op->getOperand(I: 2))) {
7283 SmallVector<StringRef, 4> ValueComp;
7284 Value->getString().split(A&: ValueComp, Separator: " ");
7285 if (ValueComp.size() != 1) {
7286 std::string NewValue;
7287 for (auto &S : ValueComp)
7288 NewValue += S.str();
7289 Metadata *Ops[3] = {Op->getOperand(I: 0), Op->getOperand(I: 1),
7290 MDString::get(Context&: M.getContext(), Str: NewValue)};
7291 ModFlags->setOperand(I, New: MDNode::get(Context&: M.getContext(), MDs: Ops));
7292 Changed = true;
7293 }
7294 }
7295 }
7296
7297 // IRUpgrader turns a i32 type "Objective-C Garbage Collection" into i8 value.
7298 // If the higher bits are set, it adds new module flag for swift info.
7299 if (ID->getString() == "Objective-C Garbage Collection") {
7300 auto Md = dyn_cast<ConstantAsMetadata>(Val: Op->getOperand(I: 2));
7301 if (Md) {
7302 assert(Md->getValue() && "Expected non-empty metadata");
7303 auto Type = Md->getValue()->getType();
7304 if (Type == Int8Ty)
7305 continue;
7306 unsigned Val = Md->getValue()->getUniqueInteger().getZExtValue();
7307 if ((Val & 0xff) != Val) {
7308 HasSwiftVersionFlag = true;
7309 SwiftABIVersion = (Val & 0xff00) >> 8;
7310 SwiftMajorVersion = (Val & 0xff000000) >> 24;
7311 SwiftMinorVersion = (Val & 0xff0000) >> 16;
7312 }
7313 Metadata *Ops[3] = {
7314 ConstantAsMetadata::get(C: ConstantInt::get(Ty: Int32Ty,V: Module::Error)),
7315 Op->getOperand(I: 1),
7316 ConstantAsMetadata::get(C: ConstantInt::get(Ty: Int8Ty,V: Val & 0xff))};
7317 ModFlags->setOperand(I, New: MDNode::get(Context&: M.getContext(), MDs: Ops));
7318 Changed = true;
7319 }
7320 }
7321
7322 if (ID->getString() == "amdgpu_code_object_version") {
7323 Metadata *Ops[3] = {
7324 Op->getOperand(I: 0),
7325 MDString::get(Context&: M.getContext(), Str: "amdhsa_code_object_version"),
7326 Op->getOperand(I: 2)};
7327 ModFlags->setOperand(I, New: MDNode::get(Context&: M.getContext(), MDs: Ops));
7328 Changed = true;
7329 }
7330
7331 // clang/PowerPC used to use "float-abi" to describe the long double format;
7332 // it has been renamed to "long-double-type", with its values changed to the
7333 // corresponding IR floating-point type names.
7334 if (M.getTargetTriple().isPPC() && ID->getString() == "float-abi") {
7335 StringRef Format;
7336 if (auto *S = dyn_cast_or_null<MDString>(Val: Op->getOperand(I: 2)))
7337 Format = S->getString();
7338
7339 // The "float-abi" key is now reserved for the target-independent
7340 // soft/hard ABI flag, so leave a valid value alone. Map any other value
7341 // (including unrecognized ones, which were never valid) to the default.
7342 if (!FloatABI::parseABIType(S: Format)) {
7343 LongDoubleFormat NewFormat =
7344 StringSwitch<LongDoubleFormat>(Format)
7345 .Case(S: "ieeequad", Value: LongDoubleFormat::IEEEquad)
7346 .Case(S: "ieeedouble", Value: LongDoubleFormat::IEEEdouble)
7347 .Default(Value: LongDoubleFormat::PPCDoubleDouble);
7348 Metadata *Ops[3] = {
7349 Op->getOperand(I: 0),
7350 MDString::get(Context&: M.getContext(), Str: "long-double-type"),
7351 MDString::get(Context&: M.getContext(), Str: getLongDoubleFormatName(Format: NewFormat))};
7352 ModFlags->setOperand(I, New: MDNode::get(Context&: M.getContext(), MDs: Ops));
7353 Changed = true;
7354 }
7355 }
7356 }
7357
7358 // "Objective-C Class Properties" is recently added for Objective-C. We
7359 // upgrade ObjC bitcodes to contain a "Objective-C Class Properties" module
7360 // flag of value 0, so we can correclty downgrade this flag when trying to
7361 // link an ObjC bitcode without this module flag with an ObjC bitcode with
7362 // this module flag.
7363 if (HasObjCFlag && !HasClassProperties) {
7364 M.addModuleFlag(Behavior: llvm::Module::Override, Key: "Objective-C Class Properties",
7365 Val: (uint32_t)0);
7366 Changed = true;
7367 }
7368
7369 if (HasSwiftVersionFlag) {
7370 M.addModuleFlag(Behavior: Module::Error, Key: "Swift ABI Version",
7371 Val: SwiftABIVersion);
7372 M.addModuleFlag(Behavior: Module::Error, Key: "Swift Major Version",
7373 Val: ConstantInt::get(Ty: Int8Ty, V: SwiftMajorVersion));
7374 M.addModuleFlag(Behavior: Module::Error, Key: "Swift Minor Version",
7375 Val: ConstantInt::get(Ty: Int8Ty, V: SwiftMinorVersion));
7376 Changed = true;
7377 }
7378
7379 return Changed;
7380}
7381
7382bool llvm::UpgradeCFIFunctionsMetadata(Module &M) {
7383 NamedMDNode *CFIConsts = M.getNamedMetadata(Name: "cfi.functions");
7384 // If this metadata has operands, we expect all of them to be either from
7385 // before or from after the format change handled here, so we can bail out
7386 // fast if the first (if any) operands is of the new format.
7387 auto MatchesVersion = [](const MDNode *Op) {
7388 return Op->getNumOperands() >= 3 &&
7389 isa<ConstantAsMetadata>(Val: Op->getOperand(I: 2)) &&
7390 cast<ConstantAsMetadata>(Val: Op->getOperand(I: 2))
7391 ->getType()
7392 ->isIntegerTy(BitWidth: 64);
7393 };
7394
7395 if (!CFIConsts || !CFIConsts->getNumOperands() ||
7396 MatchesVersion(CFIConsts->getOperand(i: 0)))
7397 return false;
7398
7399 bool Changed = false;
7400 for (unsigned I = 0, E = CFIConsts->getNumOperands(); I != E; ++I) {
7401 MDNode *Op = CFIConsts->getOperand(i: I);
7402 assert(!MatchesVersion(Op) && "Unexpected mix of CFIConstant formats");
7403 assert(Op->getNumOperands() >= 2 &&
7404 "Expected at least 2 operands - name and linkage type");
7405 MDString *NameMD = dyn_cast<MDString>(Val: Op->getOperand(I: 0));
7406 StringRef Name = NameMD->getString();
7407 GlobalValue::GUID GUID = GlobalValue::getGUIDAssumingExternalLinkage(
7408 GlobalName: GlobalValue::dropLLVMManglingEscape(Name));
7409
7410 SmallVector<Metadata *, 4> Elts;
7411 Elts.push_back(Elt: Op->getOperand(I: 0));
7412 Elts.push_back(Elt: Op->getOperand(I: 1));
7413 Elts.push_back(Elt: ConstantAsMetadata::get(
7414 C: ConstantInt::get(Ty: Type::getInt64Ty(C&: M.getContext()), V: GUID)));
7415
7416 for (unsigned J = 2, EJ = Op->getNumOperands(); J != EJ; ++J)
7417 Elts.push_back(Elt: Op->getOperand(I: J));
7418
7419 CFIConsts->setOperand(I, New: MDNode::get(Context&: M.getContext(), MDs: Elts));
7420 Changed = true;
7421 }
7422
7423 return Changed;
7424}
7425
7426void llvm::UpgradeSectionAttributes(Module &M) {
7427 auto TrimSpaces = [](StringRef Section) -> std::string {
7428 SmallVector<StringRef, 5> Components;
7429 Section.split(A&: Components, Separator: ',');
7430
7431 SmallString<32> Buffer;
7432 raw_svector_ostream OS(Buffer);
7433
7434 for (auto Component : Components)
7435 OS << ',' << Component.trim();
7436
7437 return std::string(OS.str().substr(Start: 1));
7438 };
7439
7440 for (auto &GV : M.globals()) {
7441 if (!GV.hasSection())
7442 continue;
7443
7444 StringRef Section = GV.getSection();
7445
7446 if (!Section.starts_with(Prefix: "__DATA, __objc_catlist"))
7447 continue;
7448
7449 // __DATA, __objc_catlist, regular, no_dead_strip
7450 // __DATA,__objc_catlist,regular,no_dead_strip
7451 GV.setSection(TrimSpaces(Section));
7452 }
7453}
7454
7455namespace {
7456// Prior to LLVM 10.0, the strictfp attribute could be used on individual
7457// callsites within a function that did not also have the strictfp attribute.
7458// Since 10.0, if strict FP semantics are needed within a function, the
7459// function must have the strictfp attribute and all calls within the function
7460// must also have the strictfp attribute. This latter restriction is
7461// necessary to prevent unwanted libcall simplification when a function is
7462// being cloned (such as for inlining).
7463//
7464// The "dangling" strictfp attribute usage was only used to prevent constant
7465// folding and other libcall simplification. The nobuiltin attribute on the
7466// callsite has the same effect.
7467struct StrictFPUpgradeVisitor : public InstVisitor<StrictFPUpgradeVisitor> {
7468 StrictFPUpgradeVisitor() = default;
7469
7470 void visitCallBase(CallBase &Call) {
7471 if (!Call.isStrictFP())
7472 return;
7473 if (isa<ConstrainedFPIntrinsic>(Val: &Call))
7474 return;
7475 // If we get here, the caller doesn't have the strictfp attribute
7476 // but this callsite does. Replace the strictfp attribute with nobuiltin.
7477 Call.removeFnAttr(Kind: Attribute::StrictFP);
7478 Call.addFnAttr(Kind: Attribute::NoBuiltin);
7479 }
7480};
7481
7482/// Replace "amdgpu-unsafe-fp-atomics" metadata with atomicrmw metadata
7483struct AMDGPUUnsafeFPAtomicsUpgradeVisitor
7484 : public InstVisitor<AMDGPUUnsafeFPAtomicsUpgradeVisitor> {
7485 AMDGPUUnsafeFPAtomicsUpgradeVisitor() = default;
7486
7487 void visitAtomicRMWInst(AtomicRMWInst &RMW) {
7488 if (!RMW.isFloatingPointOperation())
7489 return;
7490
7491 MDNode *Empty = MDNode::get(Context&: RMW.getContext(), MDs: {});
7492 RMW.setMetadata(Kind: "amdgpu.no.fine.grained.host.memory", Node: Empty);
7493 RMW.setMetadata(Kind: "amdgpu.no.remote.memory.access", Node: Empty);
7494 RMW.setMetadata(KindID: LLVMContext::MD_atomic_ignore_denormal_mode, Node: Empty);
7495 }
7496};
7497} // namespace
7498
7499void llvm::UpgradeFunctionAttributes(Function &F) {
7500 // If a function definition doesn't have the strictfp attribute,
7501 // convert any callsite strictfp attributes to nobuiltin.
7502 if (!F.isDeclaration() && !F.hasFnAttribute(Kind: Attribute::StrictFP)) {
7503 StrictFPUpgradeVisitor SFPV;
7504 SFPV.visit(F);
7505 }
7506
7507 // Remove all incompatibile attributes from function.
7508 F.removeRetAttrs(Attrs: AttributeFuncs::typeIncompatible(
7509 Ty: F.getReturnType(), AS: F.getAttributes().getRetAttrs()));
7510 for (auto &Arg : F.args())
7511 Arg.removeAttrs(
7512 AM: AttributeFuncs::typeIncompatible(Ty: Arg.getType(), AS: Arg.getAttributes()));
7513
7514 bool AddingAttrs = false, RemovingAttrs = false;
7515 AttrBuilder AttrsToAdd(F.getContext());
7516 AttributeMask AttrsToRemove;
7517
7518 // Older versions of LLVM treated an "implicit-section-name" attribute
7519 // similarly to directly setting the section on a Function.
7520 if (Attribute A = F.getFnAttribute(Kind: "implicit-section-name");
7521 A.isValid() && A.isStringAttribute()) {
7522 F.setSection(A.getValueAsString());
7523 AttrsToRemove.addAttribute(A: "implicit-section-name");
7524 RemovingAttrs = true;
7525 }
7526
7527 if (Attribute A = F.getFnAttribute(Kind: "nooutline");
7528 A.isValid() && A.isStringAttribute()) {
7529 AttrsToRemove.addAttribute(A: "nooutline");
7530 AttrsToAdd.addAttribute(Val: Attribute::NoOutline);
7531 AddingAttrs = RemovingAttrs = true;
7532 }
7533
7534 if (Attribute A = F.getFnAttribute(Kind: "uniform-work-group-size");
7535 A.isValid() && A.isStringAttribute() && !A.getValueAsString().empty()) {
7536 AttrsToRemove.addAttribute(A: "uniform-work-group-size");
7537 RemovingAttrs = true;
7538 if (A.getValueAsString() == "true") {
7539 AttrsToAdd.addAttribute(A: "uniform-work-group-size");
7540 AddingAttrs = true;
7541 }
7542 }
7543
7544 if (!F.empty()) {
7545 // For some reason this is called twice, and the first time is before any
7546 // instructions are loaded into the body.
7547
7548 if (Attribute A = F.getFnAttribute(Kind: "amdgpu-unsafe-fp-atomics");
7549 A.isValid()) {
7550
7551 if (A.getValueAsBool()) {
7552 AMDGPUUnsafeFPAtomicsUpgradeVisitor Visitor;
7553 Visitor.visit(F);
7554 }
7555
7556 // We will leave behind dead attribute uses on external declarations, but
7557 // clang never added these to declarations anyway.
7558 AttrsToRemove.addAttribute(A: "amdgpu-unsafe-fp-atomics");
7559 RemovingAttrs = true;
7560 }
7561 }
7562
7563 DenormalMode DenormalFPMath = DenormalMode::getIEEE();
7564 DenormalMode DenormalFPMathF32 = DenormalMode::getInvalid();
7565
7566 bool HandleDenormalMode = false;
7567
7568 if (Attribute Attr = F.getFnAttribute(Kind: "denormal-fp-math"); Attr.isValid()) {
7569 DenormalMode ParsedMode = parseDenormalFPAttribute(Str: Attr.getValueAsString());
7570 if (ParsedMode.isValid()) {
7571 DenormalFPMath = ParsedMode;
7572 AttrsToRemove.addAttribute(A: "denormal-fp-math");
7573 AddingAttrs = RemovingAttrs = true;
7574 HandleDenormalMode = true;
7575 }
7576 }
7577
7578 if (Attribute Attr = F.getFnAttribute(Kind: "denormal-fp-math-f32");
7579 Attr.isValid()) {
7580 DenormalMode ParsedMode = parseDenormalFPAttribute(Str: Attr.getValueAsString());
7581 if (ParsedMode.isValid()) {
7582 DenormalFPMathF32 = ParsedMode;
7583 AttrsToRemove.addAttribute(A: "denormal-fp-math-f32");
7584 AddingAttrs = RemovingAttrs = true;
7585 HandleDenormalMode = true;
7586 }
7587 }
7588
7589 if (HandleDenormalMode)
7590 AttrsToAdd.addDenormalFPEnvAttr(
7591 Mode: DenormalFPEnv(DenormalFPMath, DenormalFPMathF32));
7592
7593 if (RemovingAttrs)
7594 F.removeFnAttrs(Attrs: AttrsToRemove);
7595
7596 if (AddingAttrs)
7597 F.addFnAttrs(Attrs: AttrsToAdd);
7598}
7599
7600// Check if the function attribute is not present and set it.
7601static void setFunctionAttrIfNotSet(Function &F, StringRef FnAttrName,
7602 StringRef Value) {
7603 if (!F.hasFnAttribute(Kind: FnAttrName)) {
7604 F.addFnAttr(Kind: FnAttrName, Val: Value);
7605 LLVM_DEBUG(dbgs() << "Set attribute: " << FnAttrName << "=\"" << Value
7606 << "\", function: " << F.getName() << "\n");
7607 }
7608}
7609
7610// Check if the function attribute is not present and set it if needed.
7611// If the attribute is "false" then removes it.
7612// If the attribute is "true" resets it to a valueless attribute.
7613static void ConvertFunctionAttr(Function &F, bool Set, StringRef FnAttrName) {
7614 if (!F.hasFnAttribute(Kind: FnAttrName)) {
7615 if (Set) {
7616 F.addFnAttr(Kind: FnAttrName);
7617 LLVM_DEBUG(dbgs() << "Added attribute: " << FnAttrName
7618 << ", function: " << F.getName() << "\n");
7619 }
7620 } else {
7621 auto A = F.getFnAttribute(Kind: FnAttrName);
7622 if ("false" == A.getValueAsString()) {
7623 F.removeFnAttr(Kind: FnAttrName);
7624 LLVM_DEBUG(dbgs() << "Removed attribute: " << FnAttrName
7625 << "=\"false\", function: " << F.getName() << "\n");
7626 } else if ("true" == A.getValueAsString()) {
7627 F.removeFnAttr(Kind: FnAttrName);
7628 F.addFnAttr(Kind: FnAttrName);
7629 LLVM_DEBUG(dbgs() << "Converted attribute: " << FnAttrName
7630 << "=\"true\", function: " << F.getName() << "\n");
7631 }
7632 }
7633}
7634
7635static void ConvertModuleFlag(Module &M, Module::ModFlagBehavior Behavior,
7636 StringRef Key, uint32_t Val) {
7637 M.setModuleFlag(Behavior, Key, Val);
7638 LLVM_DEBUG(dbgs() << "Converted module flag: " << "{" << Behavior << ", "
7639 << Key << ", " << Val << "}\n");
7640}
7641
7642void llvm::copyModuleAttrToFunctions(Module &M) {
7643 Triple T(M.getTargetTriple());
7644 if (!T.isThumb() && !T.isARM() && !T.isAArch64())
7645 return;
7646
7647 uint64_t BTEValue = 0;
7648 uint64_t BPPLRValue = 0;
7649 uint64_t GCSValue = 0;
7650 uint64_t SRAValue = 0;
7651 uint64_t SRAALLValue = 0;
7652 uint64_t SRABKeyValue = 0;
7653
7654 NamedMDNode *ModFlags = M.getModuleFlagsMetadata();
7655 if (ModFlags) {
7656 for (unsigned I = 0, E = ModFlags->getNumOperands(); I != E; ++I) {
7657 MDNode *Op = ModFlags->getOperand(i: I);
7658 if (Op->getNumOperands() != 3)
7659 continue;
7660
7661 MDString *ID = dyn_cast_or_null<MDString>(Val: Op->getOperand(I: 1));
7662 auto *CI = mdconst::dyn_extract<ConstantInt>(MD: Op->getOperand(I: 2));
7663 if (!ID || !CI)
7664 continue;
7665
7666 StringRef IDStr = ID->getString();
7667 uint64_t *ValPtr = IDStr == "branch-target-enforcement" ? &BTEValue
7668 : IDStr == "branch-protection-pauth-lr" ? &BPPLRValue
7669 : IDStr == "guarded-control-stack" ? &GCSValue
7670 : IDStr == "sign-return-address" ? &SRAValue
7671 : IDStr == "sign-return-address-all" ? &SRAALLValue
7672 : IDStr == "sign-return-address-with-bkey"
7673 ? &SRABKeyValue
7674 : nullptr;
7675 if (!ValPtr)
7676 continue;
7677
7678 *ValPtr = CI->getZExtValue();
7679 if (*ValPtr == 2)
7680 return;
7681
7682 LLVM_DEBUG(dbgs() << "Found module flag: " << IDStr << "(" << *ValPtr
7683 << ")\n");
7684 }
7685 }
7686
7687 bool BTE = BTEValue == 1;
7688 bool BPPLR = BPPLRValue == 1;
7689 bool GCS = GCSValue == 1;
7690 bool SRA = SRAValue == 1;
7691
7692 StringRef SignTypeValue = "non-leaf";
7693 if (SRA && SRAALLValue == 1)
7694 SignTypeValue = "all";
7695
7696 StringRef SignKeyValue = "a_key";
7697 if (SRA && SRABKeyValue == 1)
7698 SignKeyValue = "b_key";
7699
7700 for (Function &F : M.getFunctionList()) {
7701 if (F.isDeclaration())
7702 continue;
7703
7704 if (SRA) {
7705 setFunctionAttrIfNotSet(F, FnAttrName: "sign-return-address", Value: SignTypeValue);
7706 setFunctionAttrIfNotSet(F, FnAttrName: "sign-return-address-key", Value: SignKeyValue);
7707 } else {
7708 if (auto A = F.getFnAttribute(Kind: "sign-return-address");
7709 A.isValid() && "none" == A.getValueAsString()) {
7710 F.removeFnAttr(Kind: "sign-return-address");
7711 F.removeFnAttr(Kind: "sign-return-address-key");
7712 }
7713 }
7714 ConvertFunctionAttr(F, Set: BTE, FnAttrName: "branch-target-enforcement");
7715 ConvertFunctionAttr(F, Set: BPPLR, FnAttrName: "branch-protection-pauth-lr");
7716 ConvertFunctionAttr(F, Set: GCS, FnAttrName: "guarded-control-stack");
7717 }
7718
7719 if (BTE)
7720 ConvertModuleFlag(M, Behavior: llvm::Module::Min, Key: "branch-target-enforcement", Val: 2);
7721 if (BPPLR)
7722 ConvertModuleFlag(M, Behavior: llvm::Module::Min, Key: "branch-protection-pauth-lr", Val: 2);
7723 if (GCS)
7724 ConvertModuleFlag(M, Behavior: llvm::Module::Min, Key: "guarded-control-stack", Val: 2);
7725 if (SRA) {
7726 ConvertModuleFlag(M, Behavior: llvm::Module::Min, Key: "sign-return-address", Val: 2);
7727 if (SRAALLValue == 1)
7728 ConvertModuleFlag(M, Behavior: llvm::Module::Min, Key: "sign-return-address-all", Val: 2);
7729 if (SRABKeyValue == 1) {
7730 ConvertModuleFlag(M, Behavior: llvm::Module::Min, Key: "sign-return-address-with-bkey",
7731 Val: 2);
7732 }
7733 }
7734}
7735
7736/// Return the replacement tags if \p T still uses a removed two-operand form.
7737static const BooleanLoopTags *getOldBooleanLoopTags(const MDTuple *T) {
7738 if (T->getNumOperands() != 2 || !mdconst::hasa<ConstantInt>(MD: T->getOperand(I: 1)))
7739 return nullptr;
7740 auto *Tag = dyn_cast_or_null<MDString>(Val: T->getOperand(I: 0));
7741 return Tag ? findBooleanLoopTags(Name: Tag->getString()) : nullptr;
7742}
7743
7744/// Build the single-operand node that replaces a boolean operand: nonzero
7745/// selects the enable tag, zero the disable tag.
7746static Metadata *makeBooleanLoopNode(LLVMContext &C,
7747 const BooleanLoopTags &Tags,
7748 const MDOperand &Op) {
7749 bool Enable = !mdconst::extract<ConstantInt>(MD: Op)->isZero();
7750 return MDTuple::get(Context&: C,
7751 MDs: {MDString::get(Context&: C, Str: Enable ? Tags.Enable : Tags.Disable)});
7752}
7753
7754static bool isOldLoopArgument(Metadata *MD) {
7755 auto *T = dyn_cast_or_null<MDTuple>(Val: MD);
7756 if (!T)
7757 return false;
7758 if (T->getNumOperands() < 1)
7759 return false;
7760 auto *S = dyn_cast_or_null<MDString>(Val: T->getOperand(I: 0));
7761 if (!S)
7762 return false;
7763 if (S->getString().starts_with(Prefix: "llvm.vectorizer."))
7764 return true;
7765 return getOldBooleanLoopTags(T) != nullptr;
7766}
7767
7768static MDString *upgradeLoopTag(LLVMContext &C, StringRef OldTag) {
7769 StringRef OldPrefix = "llvm.vectorizer.";
7770 assert(OldTag.starts_with(OldPrefix) && "Expected old prefix");
7771
7772 if (OldTag == "llvm.vectorizer.unroll")
7773 return MDString::get(Context&: C, Str: "llvm.loop.interleave.count");
7774
7775 return MDString::get(
7776 Context&: C, Str: (Twine("llvm.loop.vectorize.") + OldTag.drop_front(N: OldPrefix.size()))
7777 .str());
7778}
7779
7780static Metadata *upgradeLoopArgument(Metadata *MD) {
7781 auto *T = dyn_cast_or_null<MDTuple>(Val: MD);
7782 if (!T)
7783 return MD;
7784 if (T->getNumOperands() < 1)
7785 return MD;
7786 auto *OldTag = dyn_cast_or_null<MDString>(Val: T->getOperand(I: 0));
7787 if (!OldTag)
7788 return MD;
7789
7790 LLVMContext &C = T->getContext();
7791
7792 /// Rewrite a removed two-operand boolean form to the single-operand pair.
7793 if (const BooleanLoopTags *Tags = getOldBooleanLoopTags(T))
7794 return makeBooleanLoopNode(C, Tags: *Tags, Op: T->getOperand(I: 1));
7795
7796 if (!OldTag->getString().starts_with(Prefix: "llvm.vectorizer."))
7797 return MD;
7798
7799 // This has an old tag. Upgrade it.
7800 MDString *NewTag = upgradeLoopTag(C, OldTag: OldTag->getString());
7801
7802 // The legacy !{!"llvm.vectorizer.enable", i1 X} maps onto the single-operand
7803 // vectorize.enable/disable pair, not a two-operand enable node.
7804 if (T->getNumOperands() == 2 && mdconst::hasa<ConstantInt>(MD: T->getOperand(I: 1)))
7805 if (const BooleanLoopTags *Tags = findBooleanLoopTags(Name: NewTag->getString()))
7806 return makeBooleanLoopNode(C, Tags: *Tags, Op: T->getOperand(I: 1));
7807
7808 SmallVector<Metadata *, 8> Ops;
7809 Ops.reserve(N: T->getNumOperands());
7810 Ops.push_back(Elt: NewTag);
7811 for (unsigned I = 1, E = T->getNumOperands(); I != E; ++I)
7812 Ops.push_back(Elt: T->getOperand(I));
7813
7814 return MDTuple::get(Context&: C, MDs: Ops);
7815}
7816
7817MDNode *llvm::upgradeInstructionLoopAttachment(MDNode &N) {
7818 auto *T = dyn_cast<MDTuple>(Val: &N);
7819 if (!T)
7820 return &N;
7821
7822 if (none_of(Range: T->operands(), P: isOldLoopArgument))
7823 return &N;
7824
7825 // Fix the removed two-operand boolean nodes in place: the Verifier rejects
7826 // any MDNode carrying those tags with more than one operand, so a leftover
7827 // reference (from the distinct loop-ID) would still trigger a diagnostic.
7828 // In-place mutation is safe on distinct MDNodes.
7829 if (T->isDistinct()) {
7830 for (unsigned I = 0, E = T->getNumOperands(); I < E; ++I) {
7831 auto *OpT = dyn_cast_or_null<MDTuple>(Val: T->getOperand(I));
7832 if (OpT && getOldBooleanLoopTags(T: OpT))
7833 T->replaceOperandWith(I, New: upgradeLoopArgument(MD: OpT));
7834 }
7835 if (none_of(Range: T->operands(), P: isOldLoopArgument))
7836 return &N;
7837 }
7838
7839 // Remaining old arguments (e.g. llvm.vectorizer.*) are handled via a wrapper
7840 // attachment; the original distinct loop-ID is kept as the first operand.
7841 SmallVector<Metadata *, 8> Ops;
7842 Ops.reserve(N: T->getNumOperands());
7843 for (Metadata *MD : T->operands())
7844 Ops.push_back(Elt: upgradeLoopArgument(MD));
7845
7846 return MDTuple::get(Context&: T->getContext(), MDs: Ops);
7847}
7848
7849std::string llvm::UpgradeDataLayoutString(StringRef DL, StringRef TT) {
7850 Triple T(TT);
7851 // The only data layout upgrades needed for pre-GCN, SPIR or SPIRV are setting
7852 // the address space of globals to 1. This does not apply to SPIRV Logical.
7853 if ((T.isSPIR() || (T.isSPIRV() && !T.isSPIRVLogical())) &&
7854 !DL.contains(Other: "-G") && !DL.starts_with(Prefix: "G")) {
7855 return DL.empty() ? std::string("G1") : (DL + "-G1").str();
7856 }
7857
7858 if (T.isLoongArch64() || T.isRISCV64()) {
7859 // Make i32 a native type for 64-bit LoongArch and RISC-V.
7860 auto I = DL.find(Str: "-n64-");
7861 if (I != StringRef::npos)
7862 return (DL.take_front(N: I) + "-n32:64-" + DL.drop_front(N: I + 5)).str();
7863 return DL.str();
7864 }
7865
7866 // AMDGPU data layout upgrades.
7867 std::string Res = DL.str();
7868 if (T.isAMDGPU()) {
7869 // Define address spaces for constants.
7870 if (!DL.contains(Other: "-G") && !DL.starts_with(Prefix: "G"))
7871 Res.append(s: Res.empty() ? "G1" : "-G1");
7872
7873 // AMDGCN data layout upgrades.
7874 if (T.isAMDGCN()) {
7875
7876 // Add missing non-integral declarations.
7877 // This goes before adding new address spaces to prevent incoherent string
7878 // values.
7879 if (!DL.contains(Other: "-ni") && !DL.starts_with(Prefix: "ni"))
7880 Res.append(s: "-ni:7:8:9");
7881 // Update ni:7 to ni:7:8:9.
7882 if (DL.ends_with(Suffix: "ni:7"))
7883 Res.append(s: ":8:9");
7884 if (DL.ends_with(Suffix: "ni:7:8"))
7885 Res.append(s: ":9");
7886
7887 // Add sizing for address spaces 7 and 8 (fat raw buffers and buffer
7888 // resources) An empty data layout has already been upgraded to G1 by now.
7889 if (!DL.contains(Other: "-p7") && !DL.starts_with(Prefix: "p7"))
7890 Res.append(s: "-p7:160:256:256:32");
7891 if (!DL.contains(Other: "-p8") && !DL.starts_with(Prefix: "p8"))
7892 Res.append(s: "-p8:128:128:128:48");
7893 constexpr StringRef OldP8("-p8:128:128-");
7894 if (DL.contains(Other: OldP8))
7895 Res.replace(pos: Res.find(svt: OldP8), n1: OldP8.size(), s: "-p8:128:128:128:48-");
7896 if (!DL.contains(Other: "-p9") && !DL.starts_with(Prefix: "p9"))
7897 Res.append(s: "-p9:192:256:256:32");
7898
7899 // Add sizing for address space 10 through 15.
7900 // AS 10-14 are reserved and defaulted to 32:32
7901 // AS 15 is in use and is 32:32.
7902 for (StringRef AS : {"p10", "p11", "p12", "p13", "p14", "p15"}) {
7903 if (!DL.contains(Other: ("-" + AS).str()) && !DL.starts_with(Prefix: AS))
7904 Res.append(str: ("-" + AS + ":32:32").str());
7905 }
7906 }
7907
7908 // Upgrade the ELF mangling mode.
7909 if (!DL.contains(Other: "m:e"))
7910 Res = Res.empty() ? "m:e" : "m:e-" + Res;
7911
7912 return Res;
7913 }
7914
7915 if (T.isSystemZ() && !DL.empty()) {
7916 // Make sure the stack alignment is present.
7917 if (!DL.contains(Other: "-S64"))
7918 return "E-S64" + DL.drop_front(N: 1).str();
7919 return DL.str();
7920 }
7921
7922 auto AddPtr32Ptr64AddrSpaces = [&DL, &Res]() {
7923 // If the datalayout matches the expected format, add pointer size address
7924 // spaces to the datalayout.
7925 StringRef AddrSpaces{"-p270:32:32-p271:32:32-p272:64:64"};
7926 if (!DL.contains(Other: AddrSpaces)) {
7927 SmallVector<StringRef, 4> Groups;
7928 Regex R("^([Ee]-m:[a-z](-p:32:32)?)(-.*)$");
7929 if (R.match(String: Res, Matches: &Groups))
7930 Res = (Groups[1] + AddrSpaces + Groups[3]).str();
7931 }
7932 };
7933
7934 // AArch64 data layout upgrades.
7935 if (T.isAArch64()) {
7936 // Add "-Fn32"
7937 if (!DL.empty() && !DL.contains(Other: "-Fn32"))
7938 Res.append(s: "-Fn32");
7939 AddPtr32Ptr64AddrSpaces();
7940 return Res;
7941 }
7942
7943 if (T.isSPARC() || (T.isMIPS64() && !DL.contains(Other: "m:m")) || T.isPPC64() ||
7944 T.isWasm()) {
7945 // Mips64 with o32 ABI did not add "-i128:128".
7946 // Add "-i128:128"
7947 std::string I64 = "-i64:64";
7948 std::string I128 = "-i128:128";
7949 if (!StringRef(Res).contains(Other: I128)) {
7950 size_t Pos = Res.find(str: I64);
7951 if (Pos != size_t(-1))
7952 Res.insert(pos1: Pos + I64.size(), str: I128);
7953 }
7954 }
7955
7956 if (T.isPPC() && T.isOSAIX() && !DL.contains(Other: "f64:32:64") && !DL.empty()) {
7957 size_t Pos = Res.find(s: "-S128");
7958 if (Pos == StringRef::npos)
7959 Pos = Res.size();
7960 Res.insert(pos: Pos, s: "-f64:32:64");
7961 }
7962
7963 // ARM data layout upgrades.
7964 // Add -Fi8 if a -F has not already been specified.
7965 if (T.isARM() && !DL.empty() && !DL.contains(Other: "Fi") && !DL.contains(Other: "Fn")) {
7966 const StringRef p3232 = "p:32:32";
7967 size_t Pos = Res.find(svt: p3232);
7968 if (Pos != StringRef::npos)
7969 Res.insert(pos: Pos + p3232.size(), s: "-Fi8");
7970 }
7971
7972 if (!T.isX86())
7973 return Res;
7974
7975 AddPtr32Ptr64AddrSpaces();
7976
7977 // i128 values need to be 16-byte-aligned. LLVM already called into libgcc
7978 // for i128 operations prior to this being reflected in the data layout, and
7979 // clang mostly produced LLVM IR that already aligned i128 to 16 byte
7980 // boundaries, so although this is a breaking change, the upgrade is expected
7981 // to fix more IR than it breaks.
7982 // Intel MCU is an exception and uses 4-byte-alignment.
7983 if (!T.isOSIAMCU()) {
7984 std::string I128 = "-i128:128";
7985 if (StringRef Ref = Res; !Ref.contains(Other: I128)) {
7986 SmallVector<StringRef, 4> Groups;
7987 Regex R("^(e(-[mpi][^-]*)*)((-[^mpi][^-]*)*)$");
7988 if (R.match(String: Res, Matches: &Groups))
7989 Res = (Groups[1] + I128 + Groups[3]).str();
7990 }
7991 }
7992
7993 // For 32-bit MSVC targets, raise the alignment of f80 values to 16 bytes.
7994 // Raising the alignment is safe because Clang did not produce f80 values in
7995 // the MSVC environment before this upgrade was added.
7996 if (T.isWindowsMSVCEnvironment() && !T.isArch64Bit()) {
7997 StringRef Ref = Res;
7998 auto I = Ref.find(Str: "-f80:32-");
7999 if (I != StringRef::npos)
8000 Res = (Ref.take_front(N: I) + "-f80:128-" + Ref.drop_front(N: I + 8)).str();
8001 }
8002
8003 return Res;
8004}
8005
8006void llvm::UpgradeAttributes(AttrBuilder &B) {
8007 StringRef FramePointer;
8008 Attribute A = B.getAttribute(Kind: "no-frame-pointer-elim");
8009 if (A.isValid()) {
8010 // The value can be "true" or "false".
8011 FramePointer = A.getValueAsString() == "true" ? "all" : "none";
8012 B.removeAttribute(A: "no-frame-pointer-elim");
8013 }
8014 if (B.contains(A: "no-frame-pointer-elim-non-leaf")) {
8015 // The value is ignored. "no-frame-pointer-elim"="true" takes priority.
8016 if (FramePointer != "all")
8017 FramePointer = "non-leaf";
8018 B.removeAttribute(A: "no-frame-pointer-elim-non-leaf");
8019 }
8020 if (!FramePointer.empty())
8021 B.addAttribute(A: "frame-pointer", V: FramePointer);
8022
8023 A = B.getAttribute(Kind: "null-pointer-is-valid");
8024 if (A.isValid()) {
8025 // The value can be "true" or "false".
8026 bool NullPointerIsValid = A.getValueAsString() == "true";
8027 B.removeAttribute(A: "null-pointer-is-valid");
8028 if (NullPointerIsValid)
8029 B.addAttribute(Val: Attribute::NullPointerIsValid);
8030 }
8031
8032 A = B.getAttribute(Kind: "uniform-work-group-size");
8033 if (A.isValid()) {
8034 StringRef Val = A.getValueAsString();
8035 if (!Val.empty()) {
8036 bool IsTrue = Val == "true";
8037 B.removeAttribute(A: "uniform-work-group-size");
8038 if (IsTrue)
8039 B.addAttribute(A: "uniform-work-group-size");
8040 }
8041 }
8042}
8043
8044void llvm::UpgradeOperandBundles(std::vector<OperandBundleDef> &Bundles) {
8045 // clang.arc.attachedcall bundles are now required to have an operand.
8046 // If they don't, it's okay to drop them entirely: when there is an operand,
8047 // the "attachedcall" is meaningful and required, but without an operand,
8048 // it's just a marker NOP. Dropping it merely prevents an optimization.
8049 erase_if(C&: Bundles, P: [&](OperandBundleDef &OBD) {
8050 return OBD.getTag() == "clang.arc.attachedcall" &&
8051 OBD.inputs().empty();
8052 });
8053}
8054