1//===- AMDGPUTargetTransformInfo.cpp - AMDGPU specific TTI pass -----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// \file
10// This file implements a TargetTransformInfo analysis pass specific to the
11// AMDGPU target machine. It uses the target's detailed information to provide
12// more precise answers to certain TTI queries, while letting the target
13// independent and default TTI implementations handle the rest.
14//
15//===----------------------------------------------------------------------===//
16
17#include "AMDGPUTargetTransformInfo.h"
18#include "AMDGPUSubtarget.h"
19#include "AMDGPUTargetMachine.h"
20#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
21#include "SIModeRegisterDefaults.h"
22#include "llvm/ADT/SmallBitVector.h"
23#include "llvm/Analysis/InlineCost.h"
24#include "llvm/Analysis/LoopInfo.h"
25#include "llvm/Analysis/ValueTracking.h"
26#include "llvm/CodeGen/Analysis.h"
27#include "llvm/IR/Function.h"
28#include "llvm/IR/IRBuilder.h"
29#include "llvm/IR/IntrinsicsAMDGPU.h"
30#include "llvm/IR/PatternMatch.h"
31#include "llvm/Support/KnownBits.h"
32#include <optional>
33
34using namespace llvm;
35
36#define DEBUG_TYPE "AMDGPUtti"
37
38static cl::opt<unsigned> UnrollThresholdPrivate(
39 "amdgpu-unroll-threshold-private",
40 cl::desc("Unroll threshold for AMDGPU if private memory used in a loop"),
41 cl::init(Val: 2700), cl::Hidden);
42
43static cl::opt<unsigned> UnrollThresholdLocal(
44 "amdgpu-unroll-threshold-local",
45 cl::desc("Unroll threshold for AMDGPU if local memory used in a loop"),
46 cl::init(Val: 1000), cl::Hidden);
47
48static cl::opt<unsigned> UnrollThresholdIf(
49 "amdgpu-unroll-threshold-if",
50 cl::desc("Unroll threshold increment for AMDGPU for each if statement inside loop"),
51 cl::init(Val: 200), cl::Hidden);
52
53static cl::opt<bool> UnrollRuntimeLocal(
54 "amdgpu-unroll-runtime-local",
55 cl::desc("Allow runtime unroll for AMDGPU if local memory used in a loop"),
56 cl::init(Val: true), cl::Hidden);
57
58static cl::opt<unsigned> UnrollMaxBlockToAnalyze(
59 "amdgpu-unroll-max-block-to-analyze",
60 cl::desc("Inner loop block size threshold to analyze in unroll for AMDGPU"),
61 cl::init(Val: 32), cl::Hidden);
62
63static cl::opt<unsigned> ArgAllocaCost("amdgpu-inline-arg-alloca-cost",
64 cl::Hidden, cl::init(Val: 4000),
65 cl::desc("Cost of alloca argument"));
66
67// If the amount of scratch memory to eliminate exceeds our ability to allocate
68// it into registers we gain nothing by aggressively inlining functions for that
69// heuristic.
70static cl::opt<unsigned>
71 ArgAllocaCutoff("amdgpu-inline-arg-alloca-cutoff", cl::Hidden,
72 cl::init(Val: 256),
73 cl::desc("Maximum alloca size to use for inline cost"));
74
75// Inliner constraint to achieve reasonable compilation time.
76static cl::opt<size_t> InlineMaxBB(
77 "amdgpu-inline-max-bb", cl::Hidden, cl::init(Val: 1100),
78 cl::desc("Maximum number of BBs allowed in a function after inlining"
79 " (compile time constraint)"));
80
81// This default unroll factor is based on microbenchmarks on gfx1030.
82static cl::opt<unsigned> MemcpyLoopUnroll(
83 "amdgpu-memcpy-loop-unroll",
84 cl::desc("Unroll factor (affecting 4x32-bit operations) to use for memory "
85 "operations when lowering statically-sized memcpy, memmove, or"
86 "memset as a loop"),
87 cl::init(Val: 16), cl::Hidden);
88
89static bool dependsOnLocalPhi(const Loop *L, const Value *Cond,
90 unsigned Depth = 0) {
91 const Instruction *I = dyn_cast<Instruction>(Val: Cond);
92 if (!I)
93 return false;
94
95 if (!L->contains(Inst: I))
96 return false;
97 for (const Value *V : I->operand_values()) {
98 if (const PHINode *PHI = dyn_cast<PHINode>(Val: V)) {
99 if (llvm::none_of(Range: L->getSubLoops(), P: [PHI](const Loop* SubLoop) {
100 return SubLoop->contains(Inst: PHI); }))
101 return true;
102 } else if (Depth < 10 && dependsOnLocalPhi(L, Cond: V, Depth: Depth+1))
103 return true;
104 }
105 return false;
106}
107
108AMDGPUTTIImpl::AMDGPUTTIImpl(const AMDGPUTargetMachine *TM, const Function &F)
109 : BaseT(TM, F.getDataLayout()),
110 TargetTriple(F.getParent()->getTargetTriple()),
111 ST(static_cast<const GCNSubtarget *>(TM->getSubtargetImpl(F))),
112 TLI(ST->getTargetLowering()) {}
113
114void AMDGPUTTIImpl::getUnrollingPreferences(
115 Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP,
116 OptimizationRemarkEmitter *ORE) const {
117 const Function &F = *L->getHeader()->getParent();
118 UP.Threshold =
119 F.getFnAttributeAsParsedInteger(Kind: "amdgpu-unroll-threshold", Default: 300);
120 UP.PartialThreshold =
121 F.getFnAttributeAsParsedInteger(Kind: "amdgpu-partial-unroll-threshold", Default: 150);
122 UP.MaxCount = std::numeric_limits<unsigned>::max();
123 UP.Partial = true;
124
125 // Conditional branch in a loop back edge needs 3 additional exec
126 // manipulations in average.
127 UP.BEInsns += 3;
128
129 // We want to run unroll even for the loops which have been vectorized.
130 UP.UnrollVectorizedLoop = true;
131
132 // Enable runtime unrolling for loops whose trip count is not known at
133 // compile time.
134 UP.Runtime = true;
135
136 // Maximum alloca size than can fit registers. Reserve 16 registers.
137 const unsigned MaxAlloca = (256 - 16) * 4;
138 unsigned ThresholdPrivate = UnrollThresholdPrivate;
139 unsigned ThresholdLocal = UnrollThresholdLocal;
140
141 // If this loop has the amdgpu.loop.unroll.threshold metadata we will use the
142 // provided threshold value as the default for Threshold
143 if (MDNode *LoopUnrollThreshold =
144 findOptionMDForLoop(TheLoop: L, Name: "amdgpu.loop.unroll.threshold")) {
145 if (LoopUnrollThreshold->getNumOperands() == 2) {
146 ConstantInt *MetaThresholdValue = mdconst::extract_or_null<ConstantInt>(
147 MD: LoopUnrollThreshold->getOperand(I: 1));
148 if (MetaThresholdValue) {
149 // We will also use the supplied value for PartialThreshold for now.
150 // We may introduce additional metadata if it becomes necessary in the
151 // future.
152 UP.Threshold = MetaThresholdValue->getSExtValue();
153 UP.PartialThreshold = UP.Threshold;
154 ThresholdPrivate = std::min(a: ThresholdPrivate, b: UP.Threshold);
155 ThresholdLocal = std::min(a: ThresholdLocal, b: UP.Threshold);
156 }
157 }
158 }
159
160 unsigned MaxBoost = std::max(a: ThresholdPrivate, b: ThresholdLocal);
161 for (const BasicBlock *BB : L->getBlocks()) {
162 const DataLayout &DL = BB->getDataLayout();
163 unsigned LocalGEPsSeen = 0;
164
165 if (llvm::any_of(Range: L->getSubLoops(), P: [BB](const Loop* SubLoop) {
166 return SubLoop->contains(BB); }))
167 continue; // Block belongs to an inner loop.
168
169 for (const Instruction &I : *BB) {
170 // Unroll a loop which contains an "if" statement whose condition
171 // defined by a PHI belonging to the loop. This may help to eliminate
172 // if region and potentially even PHI itself, saving on both divergence
173 // and registers used for the PHI.
174 // Add a small bonus for each of such "if" statements.
175 if (const CondBrInst *Br = dyn_cast<CondBrInst>(Val: &I)) {
176 if (UP.Threshold < MaxBoost) {
177 BasicBlock *Succ0 = Br->getSuccessor(i: 0);
178 BasicBlock *Succ1 = Br->getSuccessor(i: 1);
179 if ((L->contains(BB: Succ0) && L->isLoopExiting(BB: Succ0)) ||
180 (L->contains(BB: Succ1) && L->isLoopExiting(BB: Succ1)))
181 continue;
182 if (dependsOnLocalPhi(L, Cond: Br->getCondition())) {
183 UP.Threshold += UnrollThresholdIf;
184 LLVM_DEBUG(dbgs() << "Set unroll threshold " << UP.Threshold
185 << " for loop:\n"
186 << *L << " due to " << *Br << '\n');
187 if (UP.Threshold >= MaxBoost)
188 return;
189 }
190 }
191 continue;
192 }
193
194 const GetElementPtrInst *GEP = dyn_cast<GetElementPtrInst>(Val: &I);
195 if (!GEP)
196 continue;
197
198 unsigned AS = GEP->getAddressSpace();
199 unsigned Threshold = 0;
200 if (AS == AMDGPUAS::PRIVATE_ADDRESS)
201 Threshold = ThresholdPrivate;
202 else if (AS == AMDGPUAS::LOCAL_ADDRESS || AS == AMDGPUAS::REGION_ADDRESS)
203 Threshold = ThresholdLocal;
204 else
205 continue;
206
207 if (UP.Threshold >= Threshold)
208 continue;
209
210 if (AS == AMDGPUAS::PRIVATE_ADDRESS) {
211 const Value *Ptr = GEP->getPointerOperand();
212 const AllocaInst *Alloca =
213 dyn_cast<AllocaInst>(Val: getUnderlyingObject(V: Ptr));
214 if (!Alloca || !Alloca->isStaticAlloca())
215 continue;
216 auto AllocaSize = Alloca->getAllocationSize(DL);
217 if (!AllocaSize || AllocaSize->getFixedValue() > MaxAlloca)
218 continue;
219 } else if (AS == AMDGPUAS::LOCAL_ADDRESS ||
220 AS == AMDGPUAS::REGION_ADDRESS) {
221 LocalGEPsSeen++;
222 // Inhibit unroll for local memory if we have seen addressing not to
223 // a variable, most likely we will be unable to combine it.
224 // Do not unroll too deep inner loops for local memory to give a chance
225 // to unroll an outer loop for a more important reason.
226 if (LocalGEPsSeen > 1 || L->getLoopDepth() > 2 ||
227 (!isa<GlobalVariable>(Val: GEP->getPointerOperand()) &&
228 !isa<Argument>(Val: GEP->getPointerOperand())))
229 continue;
230 LLVM_DEBUG(dbgs() << "Allow unroll runtime for loop:\n"
231 << *L << " due to LDS use.\n");
232 UP.Runtime = UnrollRuntimeLocal;
233 }
234
235 // Check if GEP depends on a value defined by this loop itself.
236 bool HasLoopDef = false;
237 for (const Value *Op : GEP->operands()) {
238 const Instruction *Inst = dyn_cast<Instruction>(Val: Op);
239 if (!Inst || L->isLoopInvariant(V: Op))
240 continue;
241
242 if (llvm::any_of(Range: L->getSubLoops(), P: [Inst](const Loop* SubLoop) {
243 return SubLoop->contains(Inst); }))
244 continue;
245 HasLoopDef = true;
246 break;
247 }
248 if (!HasLoopDef)
249 continue;
250
251 // We want to do whatever we can to limit the number of alloca
252 // instructions that make it through to the code generator. allocas
253 // require us to use indirect addressing, which is slow and prone to
254 // compiler bugs. If this loop does an address calculation on an
255 // alloca ptr, then we want to use a higher than normal loop unroll
256 // threshold. This will give SROA a better chance to eliminate these
257 // allocas.
258 //
259 // We also want to have more unrolling for local memory to let ds
260 // instructions with different offsets combine.
261 //
262 // Don't use the maximum allowed value here as it will make some
263 // programs way too big.
264 UP.Threshold = Threshold;
265 LLVM_DEBUG(dbgs() << "Set unroll threshold " << Threshold
266 << " for loop:\n"
267 << *L << " due to " << *GEP << '\n');
268 if (UP.Threshold >= MaxBoost)
269 return;
270 }
271
272 // If we got a GEP in a small BB from inner loop then increase max trip
273 // count to analyze for better estimation cost in unroll
274 if (L->isInnermost() && BB->size() < UnrollMaxBlockToAnalyze)
275 UP.MaxIterationsCountToAnalyze = 32;
276 }
277}
278
279void AMDGPUTTIImpl::getPeelingPreferences(Loop *L, ScalarEvolution &SE,
280 TTI::PeelingPreferences &PP) const {
281 BaseT::getPeelingPreferences(L, SE, PP);
282}
283
284uint64_t AMDGPUTTIImpl::getMaxMemIntrinsicInlineSizeThreshold() const {
285 return 1024;
286}
287
288GCNTTIImpl::GCNTTIImpl(const AMDGPUTargetMachine *TM, const Function &F)
289 : BaseT(TM, F.getDataLayout()),
290 ST(static_cast<const GCNSubtarget *>(TM->getSubtargetImpl(F))),
291 TLI(ST->getTargetLowering()), CommonTTI(TM, F),
292 IsGraphics(AMDGPU::isGraphics(CC: F.getCallingConv())) {
293 SIModeRegisterDefaults Mode(F, *ST);
294 HasFP32Denormals = Mode.FP32Denormals != DenormalMode::getPreserveSign();
295}
296
297bool GCNTTIImpl::hasBranchDivergence(const Function *F) const {
298 return !F || !ST->isSingleLaneExecution(Kernel: *F);
299}
300
301unsigned GCNTTIImpl::getNumberOfRegisters(unsigned RCID) const {
302 // NB: RCID is not an RCID. In fact it is 0 or 1 for scalar or vector
303 // registers. See getRegisterClassForType for the implementation.
304 // In this case vector registers are not vector in terms of
305 // VGPRs, but those which can hold multiple values.
306
307 // This is really the number of registers to fill when vectorizing /
308 // interleaving loops, so we lie to avoid trying to use all registers.
309 return 4;
310}
311
312TypeSize
313GCNTTIImpl::getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const {
314 switch (K) {
315 case TargetTransformInfo::RGK_Scalar:
316 return TypeSize::getFixed(ExactSize: 32);
317 case TargetTransformInfo::RGK_FixedWidthVector:
318 return TypeSize::getFixed(
319 ExactSize: (ST->hasAnyPackedFP64Ops() || ST->hasAnyPackedU64Ops()) ? 128
320 : ST->hasAnyPackedFP32Ops() ? 64
321 : 32);
322 case TargetTransformInfo::RGK_ScalableVector:
323 return TypeSize::getScalable(MinimumSize: 0);
324 }
325 llvm_unreachable("Unsupported register kind");
326}
327
328unsigned GCNTTIImpl::getMinVectorRegisterBitWidth() const {
329 return 32;
330}
331
332unsigned GCNTTIImpl::getMaximumVF(unsigned ElemWidth, unsigned Opcode) const {
333 if (Opcode == Instruction::Load || Opcode == Instruction::Store)
334 return 32 * 4 / ElemWidth;
335 // For a given width return the max 0number of elements that can be combined
336 // into a wider bit value:
337 return (ElemWidth == 8 && ST->has16BitInsts()) ? 4
338 : (ElemWidth == 16 && ST->has16BitInsts()) ? 2
339 : (ElemWidth == 32 && ST->hasAnyPackedFP32Ops()) ? 2
340 : (ElemWidth == 64 &&
341 (ST->hasAnyPackedFP64Ops() || ST->hasAnyPackedU64Ops()))
342 ? 2
343 : 1;
344}
345
346bool GCNTTIImpl::preferSLPInstCountCheck() const {
347 // The integer inst-count heuristic causes regressions on gfx94x and gfx950
348 // because 2-element vector trees that pass the scalar/vector instruction
349 // count comparison still widen scalar moves (e.g. v_mov_b32 to v_mov_b64)
350 // after codegen, increasing register pressure and throughput cost without
351 // reducing the total instruction count.
352 return !ST->hasGFX940Insts() && !ST->hasGFX950Insts();
353}
354
355unsigned GCNTTIImpl::getLoadVectorFactor(unsigned VF, unsigned LoadSize,
356 unsigned ChainSizeInBytes,
357 VectorType *VecTy) const {
358 unsigned VecRegBitWidth = VF * LoadSize;
359 if (VecRegBitWidth > 128 && VecTy->getScalarSizeInBits() < 32)
360 // TODO: Support element-size less than 32bit?
361 return 128 / LoadSize;
362
363 return VF;
364}
365
366unsigned GCNTTIImpl::getStoreVectorFactor(unsigned VF, unsigned StoreSize,
367 unsigned ChainSizeInBytes,
368 VectorType *VecTy) const {
369 unsigned VecRegBitWidth = VF * StoreSize;
370 if (VecRegBitWidth > 128)
371 return 128 / StoreSize;
372
373 return VF;
374}
375
376unsigned GCNTTIImpl::getLoadStoreVecRegBitWidth(unsigned AddrSpace) const {
377 if (AddrSpace == AMDGPUAS::GLOBAL_ADDRESS ||
378 AddrSpace == AMDGPUAS::CONSTANT_ADDRESS ||
379 AddrSpace == AMDGPUAS::CONSTANT_ADDRESS_32BIT ||
380 AddrSpace == AMDGPUAS::BUFFER_FAT_POINTER ||
381 AddrSpace == AMDGPUAS::BUFFER_RESOURCE ||
382 AddrSpace == AMDGPUAS::BUFFER_STRIDED_POINTER) {
383 return 512;
384 }
385
386 if (AddrSpace == AMDGPUAS::PRIVATE_ADDRESS)
387 return 8 * ST->getMaxPrivateElementSize();
388
389 // Common to flat, global, local and region. Assume for unknown addrspace.
390 return 128;
391}
392
393bool GCNTTIImpl::isLegalToVectorizeMemChain(unsigned ChainSizeInBytes,
394 Align Alignment,
395 unsigned AddrSpace) const {
396 // We allow vectorization of flat stores, even though we may need to decompose
397 // them later if they may access private memory. We don't have enough context
398 // here, and legalization can handle it.
399 if (AddrSpace == AMDGPUAS::PRIVATE_ADDRESS) {
400 return (Alignment >= 4 || ST->hasUnalignedScratchAccessEnabled()) &&
401 ChainSizeInBytes <= ST->getMaxPrivateElementSize();
402 }
403 return true;
404}
405
406bool GCNTTIImpl::isLegalToVectorizeLoadChain(unsigned ChainSizeInBytes,
407 Align Alignment,
408 unsigned AddrSpace) const {
409 return isLegalToVectorizeMemChain(ChainSizeInBytes, Alignment, AddrSpace);
410}
411
412bool GCNTTIImpl::isLegalToVectorizeStoreChain(unsigned ChainSizeInBytes,
413 Align Alignment,
414 unsigned AddrSpace) const {
415 return isLegalToVectorizeMemChain(ChainSizeInBytes, Alignment, AddrSpace);
416}
417
418uint64_t GCNTTIImpl::getMaxMemIntrinsicInlineSizeThreshold() const {
419 return 1024;
420}
421
422Type *GCNTTIImpl::getMemcpyLoopLoweringType(
423 LLVMContext &Context, Value *Length, unsigned SrcAddrSpace,
424 unsigned DestAddrSpace, Align SrcAlign, Align DestAlign,
425 std::optional<uint32_t> AtomicElementSize) const {
426
427 if (AtomicElementSize)
428 return Type::getIntNTy(C&: Context, N: *AtomicElementSize * 8);
429
430 // 16-byte accesses achieve the highest copy throughput.
431 // If the operation has a fixed known length that is large enough, it is
432 // worthwhile to return an even wider type and let legalization lower it into
433 // multiple accesses, effectively unrolling the memcpy loop.
434 // We also rely on legalization to decompose into smaller accesses for
435 // subtargets and address spaces where it is necessary.
436 //
437 // Don't unroll if Length is not a constant, since unrolling leads to worse
438 // performance for length values that are smaller or slightly larger than the
439 // total size of the type returned here. Mitigating that would require a more
440 // complex lowering for variable-length memcpy and memmove.
441 unsigned I32EltsInVector = 4;
442 if (MemcpyLoopUnroll > 0 && isa<ConstantInt>(Val: Length))
443 return FixedVectorType::get(ElementType: Type::getInt32Ty(C&: Context),
444 NumElts: MemcpyLoopUnroll * I32EltsInVector);
445
446 return FixedVectorType::get(ElementType: Type::getInt32Ty(C&: Context), NumElts: I32EltsInVector);
447}
448
449void GCNTTIImpl::getMemcpyLoopResidualLoweringType(
450 SmallVectorImpl<Type *> &OpsOut, LLVMContext &Context,
451 unsigned RemainingBytes, unsigned SrcAddrSpace, unsigned DestAddrSpace,
452 Align SrcAlign, Align DestAlign,
453 std::optional<uint32_t> AtomicCpySize) const {
454
455 if (AtomicCpySize)
456 BaseT::getMemcpyLoopResidualLoweringType(
457 OpsOut, Context, RemainingBytes, SrcAddrSpace, DestAddrSpace, SrcAlign,
458 DestAlign, AtomicCpySize);
459
460 Type *I32x4Ty = FixedVectorType::get(ElementType: Type::getInt32Ty(C&: Context), NumElts: 4);
461 while (RemainingBytes >= 16) {
462 OpsOut.push_back(Elt: I32x4Ty);
463 RemainingBytes -= 16;
464 }
465
466 Type *I64Ty = Type::getInt64Ty(C&: Context);
467 while (RemainingBytes >= 8) {
468 OpsOut.push_back(Elt: I64Ty);
469 RemainingBytes -= 8;
470 }
471
472 Type *I32Ty = Type::getInt32Ty(C&: Context);
473 while (RemainingBytes >= 4) {
474 OpsOut.push_back(Elt: I32Ty);
475 RemainingBytes -= 4;
476 }
477
478 Type *I16Ty = Type::getInt16Ty(C&: Context);
479 while (RemainingBytes >= 2) {
480 OpsOut.push_back(Elt: I16Ty);
481 RemainingBytes -= 2;
482 }
483
484 Type *I8Ty = Type::getInt8Ty(C&: Context);
485 while (RemainingBytes) {
486 OpsOut.push_back(Elt: I8Ty);
487 --RemainingBytes;
488 }
489}
490
491unsigned GCNTTIImpl::getMaxInterleaveFactor(ElementCount VF,
492 bool HasUnorderedReductions) const {
493 // Disable unrolling if the loop is not vectorized.
494 // TODO: Enable this again.
495 if (VF.isScalar())
496 return 1;
497
498 return 8;
499}
500
501bool GCNTTIImpl::getTgtMemIntrinsic(IntrinsicInst *Inst,
502 MemIntrinsicInfo &Info) const {
503 switch (Inst->getIntrinsicID()) {
504 case Intrinsic::amdgcn_ds_ordered_add:
505 case Intrinsic::amdgcn_ds_ordered_swap: {
506 auto *Ordering = dyn_cast<ConstantInt>(Val: Inst->getArgOperand(i: 2));
507 auto *Volatile = dyn_cast<ConstantInt>(Val: Inst->getArgOperand(i: 4));
508 if (!Ordering || !Volatile)
509 return false; // Invalid.
510
511 unsigned OrderingVal = Ordering->getZExtValue();
512 if (OrderingVal > static_cast<unsigned>(AtomicOrdering::SequentiallyConsistent))
513 return false;
514
515 Info.PtrVal = Inst->getArgOperand(i: 0);
516 Info.Ordering = static_cast<AtomicOrdering>(OrderingVal);
517 Info.ReadMem = true;
518 Info.WriteMem = true;
519 Info.IsVolatile = !Volatile->isZero();
520 return true;
521 }
522 default:
523 return false;
524 }
525}
526
527/// \returns true if \p FMul and its single fadd/fsub user \p FAddSub are
528/// expected to fuse during instruction selection. \p Ty is the type the fused
529/// operation runs on.
530static bool canFuseFMulWithFAddSub(const SITargetLowering &TLI, Type *Ty,
531 const Instruction *FMul,
532 const Instruction *FAddSub) {
533 assert((FAddSub->getOpcode() == Instruction::FAdd ||
534 FAddSub->getOpcode() == Instruction::FSub) &&
535 "Expected an fadd or an fsub");
536
537 // The mad forms fuse exactly without fast-math flags but flush denormals.
538 // An fma forms only when it is not slower than the separate operations.
539 const Function &F = *FAddSub->getFunction();
540 const bool HasFMAD = TLI.isFMADLegal(F, Ty);
541 const bool HasFMA = TLI.isFMAFasterThanFMulAndFAdd(F, Ty);
542 if (!HasFMAD && !HasFMA)
543 return false;
544
545 // Without a mad the pair fuses only when both carry contract.
546 return HasFMAD || (FAddSub->hasAllowContract() && FMul->hasAllowContract());
547}
548
549/// An fma holds one multiply, so only one fmul operand fuses with \p FAddSub.
550static const Instruction *getFusedFMul(const SITargetLowering &TLI, Type *Ty,
551 const Instruction *FAddSub) {
552 for (const Value *Op : FAddSub->operands()) {
553 const auto *FMul = dyn_cast<Instruction>(Val: Op);
554 if (FMul && FMul->getOpcode() == Instruction::FMul && FMul->hasOneUse() &&
555 canFuseFMulWithFAddSub(TLI, Ty, FMul, FAddSub))
556 return FMul;
557 }
558 return nullptr;
559}
560
561static bool isFusedFMul(const SITargetLowering &TLI, Type *Ty,
562 const Instruction *FMul, const Instruction *FAddSub) {
563 const Instruction *Fused = getFusedFMul(TLI, Ty, FAddSub);
564 if (Fused == FMul)
565 return true;
566 // (a * b + c * d) + e becomes fma(a, b, fma(c, d, e)) if the outer fadd has
567 // reassoc.
568 if (!Fused || FAddSub->getOpcode() != Instruction::FAdd ||
569 !FAddSub->hasOneUse())
570 return false;
571 const auto *Outer = dyn_cast<BinaryOperator>(Val: *FAddSub->user_begin());
572 return Outer && Outer->getOpcode() == Instruction::FAdd &&
573 Outer->hasAllowReassoc() &&
574 canFuseFMulWithFAddSub(TLI, Ty, FMul, FAddSub: Outer);
575}
576
577InstructionCost GCNTTIImpl::getArithmeticInstrCost(
578 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
579 TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
580 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
581
582 // Legalize the type.
583 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
584 int ISD = TLI->InstructionOpcodeToISD(Opcode);
585
586 // Because we don't have any legal vector operations, but the legal types, we
587 // need to account for split vectors.
588 unsigned NElts = LT.second.isVector() ?
589 LT.second.getVectorNumElements() : 1;
590
591 MVT::SimpleValueType SLT = LT.second.getScalarType().SimpleTy;
592
593 switch (ISD) {
594 case ISD::SHL:
595 case ISD::SRL:
596 case ISD::SRA:
597 if (SLT == MVT::i64)
598 return get64BitInstrCost(CostKind) * LT.first * NElts;
599
600 if (ST->has16BitInsts() && SLT == MVT::i16)
601 NElts = (NElts + 1) / 2;
602
603 // i32
604 return getFullRateInstrCost() * LT.first * NElts;
605 case ISD::ADD:
606 case ISD::SUB:
607 if (SLT == MVT::i64 && ST->hasAnyPackedU64Ops())
608 NElts = (NElts + 1) / 2;
609 [[fallthrough]];
610 case ISD::AND:
611 case ISD::OR:
612 case ISD::XOR:
613 if (SLT == MVT::i64) {
614 // and, or and xor are typically split into 2 VALU instructions.
615 return 2 * getFullRateInstrCost() * LT.first * NElts;
616 }
617
618 if (ST->has16BitInsts() && SLT == MVT::i16)
619 NElts = (NElts + 1) / 2;
620
621 return LT.first * NElts * getFullRateInstrCost();
622 case ISD::MUL: {
623 const int QuarterRateCost = getQuarterRateInstrCost(CostKind);
624 if (SLT == MVT::i64) {
625 const int FullRateCost = getFullRateInstrCost();
626 return (4 * QuarterRateCost + (2 * 2) * FullRateCost) * LT.first * NElts;
627 }
628
629 if (ST->has16BitInsts() && SLT == MVT::i16)
630 NElts = (NElts + 1) / 2;
631
632 // i32
633 return QuarterRateCost * NElts * LT.first;
634 }
635 case ISD::FMUL:
636 // Check possible fuse {fadd|fsub}(a,fmul(b,c)) and return zero cost for
637 // fmul(b,c) supposing the fadd|fsub will get estimated cost for the whole
638 // fused operation.
639 if (CtxI && CtxI->hasOneUse()) {
640 const auto *FAddSub = dyn_cast<BinaryOperator>(Val: *CtxI->user_begin());
641 if (FAddSub &&
642 (FAddSub->getOpcode() == Instruction::FAdd ||
643 FAddSub->getOpcode() == Instruction::FSub) &&
644 isFusedFMul(TLI: *TLI, Ty, FMul: CtxI, FAddSub))
645 return TargetTransformInfo::TCC_Free;
646 }
647 [[fallthrough]];
648 case ISD::FADD:
649 case ISD::FSUB:
650 if (ST->hasAnyPackedFP32Ops() && SLT == MVT::f32)
651 NElts = (NElts + 1) / 2;
652 if (ST->hasBF16PackedInsts() && SLT == MVT::bf16)
653 NElts = (NElts + 1) / 2;
654 if (SLT == MVT::f64) {
655 if (ST->hasAnyPackedFP64Ops())
656 NElts = (NElts + 1) / 2;
657 return LT.first * NElts * get64BitInstrCost(CostKind);
658 }
659
660 if (ST->has16BitInsts() && SLT == MVT::f16)
661 NElts = (NElts + 1) / 2;
662
663 if (SLT == MVT::f32 || SLT == MVT::f16 || SLT == MVT::bf16)
664 return LT.first * NElts * getFullRateInstrCost();
665 break;
666 case ISD::FDIV:
667 case ISD::FREM:
668 // FIXME: frem should be handled separately. The fdiv in it is most of it,
669 // but the current lowering is also not entirely correct.
670 if (SLT == MVT::f64) {
671 int Cost = 7 * get64BitInstrCost(CostKind) +
672 getQuarterRateInstrCost(CostKind) +
673 3 * getHalfRateInstrCost(CostKind);
674 // Add cost of workaround.
675 if (!ST->hasUsableDivScaleConditionOutput())
676 Cost += 3 * getFullRateInstrCost();
677
678 return LT.first * Cost * NElts;
679 }
680
681 if (!Args.empty() && match(V: Args[0], P: PatternMatch::m_FPOne())) {
682 // TODO: This is more complicated, unsafe flags etc.
683 if ((SLT == MVT::f32 && !HasFP32Denormals) ||
684 (SLT == MVT::f16 && ST->has16BitInsts())) {
685 return LT.first * getTransInstrCost(CostKind) * NElts;
686 }
687 }
688
689 if (SLT == MVT::f16 && ST->has16BitInsts()) {
690 // 2 x v_cvt_f32_f16
691 // f32 rcp
692 // f32 fmul
693 // v_cvt_f16_f32
694 // f16 div_fixup
695 int Cost = 4 * getFullRateInstrCost() + 2 * getTransInstrCost(CostKind);
696 return LT.first * Cost * NElts;
697 }
698
699 if (SLT == MVT::f32 && (CtxI && CtxI->hasApproxFunc())) {
700 // Fast unsafe fdiv lowering:
701 // f32 rcp
702 // f32 fmul
703 int Cost = getTransInstrCost(CostKind) + getFullRateInstrCost();
704 return LT.first * Cost * NElts;
705 }
706
707 if (SLT == MVT::f32 || SLT == MVT::f16) {
708 // 4 more v_cvt_* insts without f16 insts support
709 int Cost = (SLT == MVT::f16 ? 14 : 10) * getFullRateInstrCost() +
710 1 * getTransInstrCost(CostKind);
711
712 if (!HasFP32Denormals) {
713 // FP mode switches.
714 Cost += 2 * getFullRateInstrCost();
715 }
716
717 return LT.first * NElts * Cost;
718 }
719 break;
720 case ISD::FNEG:
721 // Use the backend' estimation. If fneg is not free each element will cost
722 // one additional instruction.
723 return TLI->isFNegFree(VT: SLT) ? 0 : NElts;
724 default:
725 break;
726 }
727
728 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Opd1Info: Op1Info, Opd2Info: Op2Info,
729 Args, CtxI);
730}
731
732// Return true if there's a potential benefit from using v2f16/v2i16
733// instructions for an intrinsic, even if it requires nontrivial legalization.
734static bool intrinsicHasPackedVectorBenefit(Intrinsic::ID ID) {
735 switch (ID) {
736 case Intrinsic::fma:
737 case Intrinsic::fmuladd:
738 case Intrinsic::copysign:
739 case Intrinsic::minimumnum:
740 case Intrinsic::maximumnum:
741 case Intrinsic::canonicalize:
742 // There's a small benefit to using vector ops in the legalized code.
743 case Intrinsic::round:
744 case Intrinsic::uadd_sat:
745 case Intrinsic::usub_sat:
746 case Intrinsic::sadd_sat:
747 case Intrinsic::ssub_sat:
748 case Intrinsic::abs:
749 return true;
750 default:
751 return false;
752 }
753}
754
755InstructionCost
756GCNTTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
757 TTI::TargetCostKind CostKind) const {
758 switch (ICA.getID()) {
759 case Intrinsic::fabs:
760 // Free source modifier in the common case.
761 return 0;
762 case Intrinsic::amdgcn_workitem_id_x:
763 case Intrinsic::amdgcn_workitem_id_y:
764 case Intrinsic::amdgcn_workitem_id_z:
765 // TODO: If hasPackedTID, or if the calling context is not an entry point
766 // there may be a bit instruction.
767 return 0;
768 case Intrinsic::amdgcn_workgroup_id_x:
769 case Intrinsic::amdgcn_workgroup_id_y:
770 case Intrinsic::amdgcn_workgroup_id_z:
771 case Intrinsic::amdgcn_lds_kernel_id:
772 case Intrinsic::amdgcn_dispatch_ptr:
773 case Intrinsic::amdgcn_dispatch_id:
774 case Intrinsic::amdgcn_implicitarg_ptr:
775 case Intrinsic::amdgcn_queue_ptr:
776 // Read from an argument register.
777 return 0;
778 default:
779 break;
780 }
781
782 Type *RetTy = ICA.getReturnType();
783
784 Intrinsic::ID IID = ICA.getID();
785 switch (IID) {
786 case Intrinsic::exp:
787 case Intrinsic::exp2:
788 case Intrinsic::exp10: {
789 // Legalize the type.
790 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: RetTy);
791 MVT::SimpleValueType SLT = LT.second.getScalarType().SimpleTy;
792 unsigned NElts =
793 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
794
795 if (SLT == MVT::f64) {
796 unsigned NumOps = 20;
797 if (IID == Intrinsic::exp)
798 ++NumOps;
799 else if (IID == Intrinsic::exp10)
800 NumOps += 3;
801
802 return LT.first * NElts * NumOps * get64BitInstrCost(CostKind);
803 }
804
805 if (SLT == MVT::f32) {
806 unsigned NumFullRateOps = 0;
807 // v_exp_f32 (transcendental).
808 unsigned NumTransOps = 1;
809
810 if (!ICA.getFlags().approxFunc() && IID != Intrinsic::exp2) {
811 // Non-AFN exp/exp10: range reduction + v_exp_f32 + ldexp +
812 // overflow/underflow checks (lowerFEXP). Denorm is also handled.
813 // FMA preamble: ~13 full-rate ops; non-FMA: ~17.
814 NumFullRateOps = ST->hasFastFMAF32() ? 13 : 17;
815 } else {
816 if (IID == Intrinsic::exp) {
817 // lowerFEXPUnsafe: fmul (base conversion) + v_exp_f32.
818 NumFullRateOps = 1;
819 } else if (IID == Intrinsic::exp10) {
820 // lowerFEXP10Unsafe: 3 fmul + 2 v_exp_f32 (double-exp2).
821 NumFullRateOps = 3;
822 NumTransOps = 2;
823 }
824 // Denorm scaling adds setcc + select + fadd + select + fmul.
825 if (HasFP32Denormals)
826 NumFullRateOps += 5;
827 }
828
829 InstructionCost Cost = NumFullRateOps * getFullRateInstrCost() +
830 NumTransOps * getTransInstrCost(CostKind);
831 return LT.first * NElts * Cost;
832 }
833
834 break;
835 }
836 case Intrinsic::log:
837 case Intrinsic::log2:
838 case Intrinsic::log10: {
839 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: RetTy);
840 MVT::SimpleValueType SLT = LT.second.getScalarType().SimpleTy;
841 unsigned NElts =
842 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
843
844 if (SLT == MVT::f32) {
845 unsigned NumFullRateOps = 0;
846
847 if (IID == Intrinsic::log2) {
848 // LowerFLOG2: just v_log_f32.
849 } else if (ICA.getFlags().approxFunc()) {
850 // LowerFLOGUnsafe: v_log_f32 + fmul (base conversion).
851 NumFullRateOps = 1;
852 } else {
853 // LowerFLOGCommon non-AFN: v_log_f32 + extended-precision
854 // multiply + finite check.
855 NumFullRateOps = ST->hasFastFMAF32() ? 8 : 11;
856 }
857
858 if (HasFP32Denormals)
859 NumFullRateOps += 5;
860
861 InstructionCost Cost =
862 NumFullRateOps * getFullRateInstrCost() + getTransInstrCost(CostKind);
863 return LT.first * NElts * Cost;
864 }
865
866 break;
867 }
868 case Intrinsic::sin:
869 case Intrinsic::cos: {
870 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: RetTy);
871 MVT::SimpleValueType SLT = LT.second.getScalarType().SimpleTy;
872 unsigned NElts =
873 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
874
875 if (SLT == MVT::f32) {
876 // LowerTrig: fmul(1/2pi) + v_sin/v_cos.
877 unsigned NumFullRateOps = ST->hasTrigReducedRange() ? 2 : 1;
878
879 InstructionCost Cost =
880 NumFullRateOps * getFullRateInstrCost() + getTransInstrCost(CostKind);
881 return LT.first * NElts * Cost;
882 }
883
884 break;
885 }
886 case Intrinsic::sqrt: {
887 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: RetTy);
888 MVT::SimpleValueType SLT = LT.second.getScalarType().SimpleTy;
889 unsigned NElts =
890 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
891
892 if (SLT == MVT::f32) {
893 unsigned NumFullRateOps = 0;
894
895 if (!ICA.getFlags().approxFunc()) {
896 // lowerFSQRTF32 non-AFN: v_sqrt_f32 + refinement + scale fixup.
897 NumFullRateOps = HasFP32Denormals ? 17 : 16;
898 }
899
900 InstructionCost Cost =
901 NumFullRateOps * getFullRateInstrCost() + getTransInstrCost(CostKind);
902 return LT.first * NElts * Cost;
903 }
904
905 break;
906 }
907 default:
908 break;
909 }
910
911 if (!intrinsicHasPackedVectorBenefit(ID: ICA.getID()))
912 return BaseT::getIntrinsicInstrCost(ICA, CostKind);
913
914 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: RetTy);
915 MVT::SimpleValueType SLT = LT.second.getScalarType().SimpleTy;
916 unsigned NElts = LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
917
918 if ((ST->hasVOP3PInsts() &&
919 (SLT == MVT::f16 || SLT == MVT::i16 ||
920 (SLT == MVT::bf16 && ST->hasBF16PackedInsts()))) ||
921 (ST->hasAnyPackedFP64Ops() && SLT == MVT::f64) ||
922 (ST->hasAnyPackedU64Ops() && SLT == MVT::i64)) {
923 NElts = (NElts + 1) / 2;
924 } else if (SLT == MVT::f32) {
925 bool HasPk2FP32Op = ST->hasAnyPackedFP32Ops() &&
926 IID != Intrinsic::minimumnum &&
927 IID != Intrinsic::maximumnum;
928 NElts = HasPk2FP32Op ? (NElts + 1) / 2 : NElts;
929 }
930
931 // TODO: Get more refined intrinsic costs?
932 unsigned InstRate = getQuarterRateInstrCost(CostKind);
933
934 switch (ICA.getID()) {
935 case Intrinsic::fma:
936 case Intrinsic::fmuladd:
937 if (SLT == MVT::f64) {
938 InstRate = get64BitInstrCost(CostKind);
939 break;
940 }
941
942 if ((SLT == MVT::f32 && ST->hasFastFMAF32()) || SLT == MVT::f16)
943 InstRate = getFullRateInstrCost();
944 else {
945 InstRate = ST->hasFastFMAF32() ? getHalfRateInstrCost(CostKind)
946 : getQuarterRateInstrCost(CostKind);
947 }
948 break;
949 case Intrinsic::copysign:
950 return NElts * getFullRateInstrCost();
951 case Intrinsic::minimumnum:
952 case Intrinsic::maximumnum: {
953 // Instruction + 2 canonicalizes. For cases that need type promotion, we the
954 // promotion takes the place of the canonicalize.
955 unsigned NumOps = 3;
956 if (const IntrinsicInst *II = ICA.getInst()) {
957 // Directly legal with ieee=0
958 // TODO: Not directly legal with strictfp
959 if (fpenvIEEEMode(I: *II) == KnownIEEEMode::Off)
960 NumOps = 1;
961 }
962
963 unsigned BaseRate =
964 SLT == MVT::f64 ? get64BitInstrCost(CostKind) : getFullRateInstrCost();
965 InstRate = BaseRate * NumOps;
966 break;
967 }
968 case Intrinsic::canonicalize: {
969 InstRate =
970 SLT == MVT::f64 ? get64BitInstrCost(CostKind) : getFullRateInstrCost();
971 break;
972 }
973 case Intrinsic::uadd_sat:
974 case Intrinsic::usub_sat:
975 case Intrinsic::sadd_sat:
976 case Intrinsic::ssub_sat: {
977 if (SLT == MVT::i16 || SLT == MVT::i32)
978 InstRate = getFullRateInstrCost();
979
980 static const auto ValidSatTys = {MVT::v2i16, MVT::v4i16};
981 if (any_of(Range: ValidSatTys, P: equal_to(Arg&: LT.second)))
982 NElts = 1;
983 break;
984 }
985 case Intrinsic::abs:
986 // Expansion takes 2 instructions for VALU
987 if (SLT == MVT::i16 || SLT == MVT::i32)
988 InstRate = 2 * getFullRateInstrCost();
989 break;
990 default:
991 break;
992 }
993
994 return LT.first * NElts * InstRate;
995}
996
997InstructionCost GCNTTIImpl::getCFInstrCost(unsigned Opcode,
998 TTI::TargetCostKind CostKind,
999 const Instruction *I) const {
1000 assert((I == nullptr || I->getOpcode() == Opcode) &&
1001 "Opcode should reflect passed instruction.");
1002 const bool SCost =
1003 (CostKind == TTI::TCK_CodeSize || CostKind == TTI::TCK_SizeAndLatency);
1004 const int CBrCost = SCost ? 5 : 7;
1005 switch (Opcode) {
1006 case Instruction::UncondBr:
1007 // Branch instruction takes about 4 slots on gfx900.
1008 return SCost ? 1 : 4;
1009 case Instruction::CondBr:
1010 // Suppose conditional branch takes additional 3 exec manipulations
1011 // instructions in average.
1012 return CBrCost;
1013 case Instruction::Switch: {
1014 const auto *SI = dyn_cast_or_null<SwitchInst>(Val: I);
1015 // Each case (including default) takes 1 cmp + 1 cbr instructions in
1016 // average.
1017 return (SI ? (SI->getNumCases() + 1) : 4) * (CBrCost + 1);
1018 }
1019 case Instruction::Ret:
1020 return SCost ? 1 : 10;
1021 }
1022 return BaseT::getCFInstrCost(Opcode, CostKind, I);
1023}
1024
1025InstructionCost GCNTTIImpl::getCmpSelInstrCost(
1026 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
1027 TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
1028 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
1029 // For size and latency cost kinds, return a low cost independent of vector
1030 // width to enable SimplifyCFG's speculativelyExecuteBB optimization.
1031 if (CostKind == TTI::TCK_SizeAndLatency)
1032 return 1;
1033
1034 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
1035 Op1Info, Op2Info, I);
1036}
1037
1038// Measured packing cost of i1 for gfx9-12 is 4.0 to 4.8, up to 5.4 with
1039// true16; unpacking is 2.6 to 2.9.
1040static constexpr unsigned MaskPackCostPerElt = 4;
1041static constexpr unsigned MaskUnpackCostPerElt = 3;
1042
1043static std::optional<unsigned> getNumberOfPackedMaskElts(Type *Ty) {
1044 auto *FVT = dyn_cast<FixedVectorType>(Val: Ty);
1045 if (FVT && FVT->getElementType()->isIntegerTy(BitWidth: 1) && FVT->getNumElements() > 1)
1046 return FVT->getNumElements();
1047 return std::nullopt;
1048}
1049
1050InstructionCost GCNTTIImpl::getCastInstrCost(unsigned Opcode, Type *Dst,
1051 Type *Src,
1052 TTI::CastContextHint CCH,
1053 TTI::TargetCostKind CostKind,
1054 const Instruction *I) const {
1055 // A bitcast between a vector of i1 and an integer packs or unpacks a mask.
1056 if (Opcode == Instruction::BitCast) {
1057 if (std::optional<unsigned> Elts = getNumberOfPackedMaskElts(Ty: Src);
1058 Elts && Dst->isIntegerTy(BitWidth: *Elts))
1059 return InstructionCost(MaskPackCostPerElt) * *Elts *
1060 getFullRateInstrCost();
1061 if (std::optional<unsigned> Elts = getNumberOfPackedMaskElts(Ty: Dst);
1062 Elts && Src->isIntegerTy(BitWidth: *Elts))
1063 return InstructionCost(MaskUnpackCostPerElt) * *Elts *
1064 getFullRateInstrCost();
1065 }
1066
1067 // Each f32 lane is rounded and the halves are packed in pairs. A packed
1068 // conversion rounds a pair at once and true16 writes either half.
1069 if (auto *SrcVTy = dyn_cast<FixedVectorType>(Val: Src);
1070 SrcVTy && Opcode == Instruction::FPTrunc && ST->has16BitInsts() &&
1071 SrcVTy->getElementType()->isFloatTy() &&
1072 Dst->getScalarType()->isHalfTy()) {
1073 const unsigned NElts = SrcVTy->getNumElements();
1074 const unsigned Ops = ST->hasCvtPkF16F32Inst() ? divideCeil(Numerator: NElts, Denominator: 2)
1075 : ST->useRealTrue16Insts() ? NElts
1076 : NElts + NElts / 2;
1077 return InstructionCost(Ops) * getFullRateInstrCost();
1078 }
1079
1080 const int ISD = TLI->InstructionOpcodeToISD(Opcode);
1081 switch (ISD) {
1082 case ISD::SINT_TO_FP:
1083 case ISD::UINT_TO_FP:
1084 case ISD::FP_TO_SINT:
1085 case ISD::FP_TO_UINT:
1086 break;
1087 default:
1088 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1089 }
1090
1091 const bool IsIntToFP = ISD == ISD::SINT_TO_FP || ISD == ISD::UINT_TO_FP;
1092 Type *FPTy = (IsIntToFP ? Dst : Src)->getScalarType();
1093 if (!FPTy->isHalfTy() && !FPTy->isBFloatTy() && !FPTy->isFloatTy() &&
1094 !FPTy->isDoubleTy())
1095 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1096
1097 unsigned NElts = 1;
1098 if (auto *VT = dyn_cast<FixedVectorType>(Val: Src))
1099 NElts = VT->getNumElements();
1100
1101 const unsigned SrcBits = Src->getScalarSizeInBits();
1102 const unsigned DstBits = Dst->getScalarSizeInBits();
1103 const bool IsSigned = ISD == ISD::SINT_TO_FP || ISD == ISD::FP_TO_SINT;
1104 const unsigned IntBits = IsIntToFP ? SrcBits : DstBits;
1105 const bool UsesInt64 = IntBits > 32 && IntBits <= 64;
1106
1107 auto Scale = [&](unsigned FullRateOps,
1108 unsigned FP64Ops = 0) -> InstructionCost {
1109 return NElts * (InstructionCost(FullRateOps) * getFullRateInstrCost() +
1110 InstructionCost(FP64Ops) * get64BitInstrCost(CostKind));
1111 };
1112
1113 if (IsIntToFP) {
1114 // A scalar load of a whole number of bytes that is not a power of two is
1115 // split and its high part load extends the source. A constant address
1116 // space or invariant global load aligned to 4 bytes may be widened instead
1117 // and then still needs the extension. So does a buffer fat pointer load.
1118 const auto *Load =
1119 I && Src->isIntegerTy() && I->getOperand(i: 0)->getType() == Src
1120 ? dyn_cast<LoadInst>(Val: I->getOperand(i: 0))
1121 : nullptr;
1122 const unsigned AS = Load ? Load->getPointerAddressSpace() : 0;
1123 const bool LoadMayWiden =
1124 Load && Load->getAlign() >= Align(4) &&
1125 (AS == AMDGPUAS::CONSTANT_ADDRESS ||
1126 AS == AMDGPUAS::CONSTANT_ADDRESS_32BIT ||
1127 (AS == AMDGPUAS::GLOBAL_ADDRESS &&
1128 Load->hasMetadata(KindID: LLVMContext::MD_invariant_load)));
1129 const bool LoadExtends = Load && Load->isSimple() && Load->hasOneUse() &&
1130 !LoadMayWiden &&
1131 AS != AMDGPUAS::BUFFER_FAT_POINTER &&
1132 AS != AMDGPUAS::BUFFER_STRIDED_POINTER &&
1133 SrcBits % 8 == 0 && !isPowerOf2_32(Value: SrcBits);
1134 const unsigned ExtOps =
1135 UsesInt64 && SrcBits < 64 && !LoadExtends ? (IsSigned ? 2 : 1) : 0;
1136 if (FPTy->isBFloatTy()) {
1137 if (SrcBits < 8 || (SrcBits > 32 && !UsesInt64))
1138 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1139
1140 // Each integer is converted to f32 first.
1141 InstructionCost FloatCost =
1142 Scale(UsesInt64 ? ExtOps + (IsSigned ? 12 : 8) : 1);
1143 if (SrcBits < 32)
1144 FloatCost = getCastInstrCost(
1145 Opcode, Dst: Dst->getWithNewType(EltTy: Type::getFloatTy(C&: Dst->getContext())),
1146 Src, CCH, CostKind, I);
1147
1148 // Native rounding can convert a pair. With 16 bit instructions the
1149 // expansion extracts the low significand bit, adds the rounding bias,
1150 // preserves NaNs and shifts the result. Without gfx9 instructions the two
1151 // additions cannot use v_add3_u32.
1152 InstructionCost RoundCost =
1153 ST->hasBF16ConversionInsts()
1154 ? InstructionCost(divideCeil(Numerator: NElts, Denominator: 2)) * getFullRateInstrCost()
1155 : Scale(!ST->has16BitInsts() ? 1
1156 : ST->hasGFX9Insts() ? 6
1157 : 7);
1158 return FloatCost + RoundCost;
1159 }
1160
1161 // No instruction converts from a 64 bit integer.
1162 if (UsesInt64) {
1163 if (FPTy->isDoubleTy()) {
1164 // Two conversions, ldexp and add, all using the FP64 rate.
1165 return Scale(ExtOps, 4);
1166 }
1167 if (FPTy->isFloatTy())
1168 return Scale(ExtOps + (IsSigned ? 12 : 8));
1169 return Scale(ExtOps + (IsSigned ? 13 : 9));
1170 }
1171
1172 // A narrow vector source is converted lane by lane.
1173 if (SrcBits >= 8 && SrcBits < 32 && isa<FixedVectorType>(Val: Src)) {
1174 if (FPTy->isDoubleTy())
1175 return Scale(1 + (SrcBits < 16 && IsSigned && ST->has16BitInsts()), 1);
1176 if (FPTy->isHalfTy()) {
1177 const InstructionCost PairCost =
1178 InstructionCost(NElts / 2) * getFullRateInstrCost();
1179 // Lanes wider than 16 bits, and every lane without 16 bit instructions,
1180 // are converted to f32 first. With the packed conversion each pair is
1181 // then converted at once. Otherwise each lane is converted, and without
1182 // real true16 each pair of halves is packed.
1183 if (!ST->has16BitInsts() || SrcBits > 16) {
1184 auto *FloatTy =
1185 FixedVectorType::get(ElementType: Type::getFloatTy(C&: Dst->getContext()), NumElts: NElts);
1186 const InstructionCost FloatCost =
1187 getCastInstrCost(Opcode, Dst: FloatTy, Src, CCH, CostKind);
1188 if (ST->has16BitInsts() && ST->hasCvtPkF16F32Inst())
1189 return FloatCost + InstructionCost(divideCeil(Numerator: NElts, Denominator: 2)) *
1190 getFullRateInstrCost();
1191 const unsigned PackOps =
1192 !ST->has16BitInsts() ? 2 : !ST->useRealTrue16Insts();
1193 return FloatCost + Scale(1) + PackOps * PairCost;
1194 }
1195 // Each lane is converted from a 16 bit subword. Lanes narrower than 16
1196 // bits are extended lane by lane, except bytes with SDWA. Signed bytes
1197 // are sign extended a pair at a time. Real true16 reads the high
1198 // subword in place, so only signed lanes narrower than 16 bits pay per
1199 // pair. Otherwise wider lanes shift the high subword of each pair down
1200 // without SDWA, and each pair of halves is packed.
1201 const unsigned PerElt =
1202 1 + (SrcBits == 8 ? !ST->hasSDWA() : SrcBits < 16);
1203 const bool RealTrue16 = ST->useRealTrue16Insts();
1204 const unsigned PerPair =
1205 !RealTrue16 + (SrcBits == 8 || RealTrue16 ? SrcBits < 16 && IsSigned
1206 : !ST->hasSDWA());
1207 return Scale(PerElt) + PerPair * PairCost;
1208 }
1209 // An unsigned byte is converted straight out of its register.
1210 if (SrcBits == 8 && !IsSigned)
1211 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1212 if (SrcBits < 16 && !IsSigned)
1213 return Scale(2);
1214 unsigned PerElt = ST->hasSDWA() && SrcBits <= 16 ? 1 : 2;
1215 if (SrcBits < 16 && ST->has16BitInsts())
1216 ++PerElt;
1217 return Scale(PerElt);
1218 }
1219
1220 // A narrow scalar source is extended before the conversion. An unsigned
1221 // byte is converted straight out of its register, and a signed one with
1222 // SDWA. A source wider than a byte but narrower than 16 bits is masked
1223 // or sign extended first. Without 16 bit instructions a half result is
1224 // rounded from f32.
1225 if (SrcBits >= 8 && SrcBits < 32) {
1226 if (LoadExtends) {
1227 if (FPTy->isDoubleTy())
1228 return Scale(0, 1);
1229 return Scale(FPTy->isHalfTy() ? 2 : 1);
1230 }
1231 const bool Narrow = SrcBits > 8 && SrcBits < 16;
1232 const bool SignExtend16 = IsSigned && Narrow && ST->has16BitInsts();
1233 if (FPTy->isDoubleTy())
1234 return Scale(1 + 2 * SignExtend16, 1);
1235 if (SrcBits > 16)
1236 return Scale(FPTy->isHalfTy() ? 3 : 2);
1237 if (FPTy->isHalfTy() && ST->has16BitInsts())
1238 return Scale(Narrow ? (IsSigned ? 3 : 2)
1239 : SrcBits == 8 && !ST->hasSDWA() ? 2
1240 : 1);
1241 const InstructionCost FloatCost =
1242 SignExtend16 ? Scale(ST->hasSDWA() ? 3 : 4)
1243 : Narrow ? Scale(2)
1244 : SrcBits == 8 && !IsSigned ? Scale(1)
1245 : Scale(ST->hasSDWA() ? 1 : 2);
1246 return FPTy->isHalfTy() ? FloatCost + Scale(1) : FloatCost;
1247 }
1248
1249 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1250 }
1251
1252 // No instruction converts to a 64 bit integer.
1253 if (UsesInt64) {
1254 const bool IsSigned64 = IsSigned || (DstBits < 64 && NElts == 1);
1255 if (FPTy->isDoubleTy()) {
1256 // With native trunc/floor there are six FP64 operations. The expanded
1257 // rounding sequence has seven, plus integer operations and constants.
1258 return ST->haveRoundOpsF64() ? Scale(1, 6) : Scale(22, 7);
1259 }
1260 // The f32 expansion has one fma, which is quarter rate without fast FMA.
1261 const InstructionCost SlowFMACost =
1262 ST->hasFastFMAF32() ? 0
1263 : NElts * (getQuarterRateInstrCost(CostKind) -
1264 getFullRateInstrCost());
1265 if (FPTy->isFloatTy())
1266 return Scale(IsSigned64 ? 13 : 6) + SlowFMACost;
1267 if (FPTy->isBFloatTy()) {
1268 // Unlike half, bf16 does not fit in i32. Extend to f32 and use the full
1269 // i64 expansion.
1270 return Scale(1 + (IsSigned64 ? 13 : 6)) + SlowFMACost;
1271 }
1272 return Scale(3);
1273 }
1274
1275 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1276}
1277
1278InstructionCost
1279GCNTTIImpl::getArithmeticReductionCost(unsigned Opcode, VectorType *Ty,
1280 std::optional<FastMathFlags> FMF,
1281 TTI::TargetCostKind CostKind) const {
1282 if (TTI::requiresOrderedReduction(FMF))
1283 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
1284
1285 // An add or xor reduction over a vector of i1 becomes a bit count over the
1286 // packed mask; the generic model prices a shuffle tree and misses that.
1287 if (Opcode == Instruction::Add || Opcode == Instruction::Xor) {
1288 if (std::optional<unsigned> Elts = getNumberOfPackedMaskElts(Ty))
1289 return InstructionCost(MaskPackCostPerElt) * *Elts *
1290 getFullRateInstrCost();
1291 }
1292
1293 EVT OrigTy = TLI->getValueType(DL, Ty);
1294
1295 // Computes cost on targets that have packed math instructions(which support
1296 // 16-bit types only).
1297 if (!ST->hasVOP3PInsts() || OrigTy.getScalarSizeInBits() != 16)
1298 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
1299
1300 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1301 return LT.first * getFullRateInstrCost();
1302}
1303
1304InstructionCost
1305GCNTTIImpl::getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty,
1306 FastMathFlags FMF,
1307 TTI::TargetCostKind CostKind) const {
1308 EVT OrigTy = TLI->getValueType(DL, Ty);
1309
1310 // Computes cost on targets that have packed math instructions(which support
1311 // 16-bit types only).
1312 if (!ST->hasVOP3PInsts() || OrigTy.getScalarSizeInBits() != 16)
1313 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
1314
1315 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1316 return LT.first * getHalfRateInstrCost(CostKind);
1317}
1318
1319InstructionCost GCNTTIImpl::getVectorInstrCost(
1320 unsigned Opcode, Type *ValTy, TTI::TargetCostKind CostKind, unsigned Index,
1321 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
1322 switch (Opcode) {
1323 case Instruction::ExtractElement:
1324 case Instruction::InsertElement: {
1325 unsigned EltSize
1326 = DL.getTypeSizeInBits(Ty: cast<VectorType>(Val: ValTy)->getElementType());
1327 // Dynamic indexing isn't free and is best avoided.
1328 if (Index == ~0u)
1329 return 2;
1330 if (EltSize < 32) {
1331 if (EltSize == 16 && Index == 0 && ST->has16BitInsts())
1332 return 0;
1333 // Inserts of booleans are free.
1334 // TODO: Extracts are free too.
1335 if (EltSize == 1 && Opcode == Instruction::InsertElement)
1336 return TargetTransformInfo::TCC_Free;
1337 // Extract element sequences of consecutive i8 values that match a
1338 // register size are free most likely. It is not possible to know
1339 // if this extract is part of a consecutive sequence so this may
1340 // apply more generally.
1341 if (Opcode == Instruction::ExtractElement && EltSize == 8) {
1342 if (auto *FVTy = dyn_cast<FixedVectorType>(Val: ValTy)) {
1343 unsigned NumElts = FVTy->getNumElements();
1344 if (NumElts >= 4 && isPowerOf2_32(Value: NumElts))
1345 return 0;
1346 }
1347 }
1348 return BaseT::getVectorInstrCost(Opcode, Val: ValTy, CostKind, Index, Op0, Op1,
1349 VIC);
1350 }
1351
1352 // Extracts are just reads of a subregister, so are free. Inserts are
1353 // considered free because we don't want to have any cost for scalarizing
1354 // operations, and we don't have to copy into a different register class.
1355 return 0;
1356 }
1357 default:
1358 return BaseT::getVectorInstrCost(Opcode, Val: ValTy, CostKind, Index, Op0, Op1,
1359 VIC);
1360 }
1361}
1362
1363/// Analyze if the results of inline asm are divergent. If \p Indices is empty,
1364/// this is analyzing the collective result of all output registers. Otherwise,
1365/// this is only querying a specific result index if this returns multiple
1366/// registers in a struct.
1367bool GCNTTIImpl::isInlineAsmSourceOfDivergence(
1368 const CallInst *CI, ArrayRef<unsigned> Indices) const {
1369 // TODO: Handle complex extract indices
1370 if (Indices.size() > 1)
1371 return true;
1372
1373 const DataLayout &DL = CI->getDataLayout();
1374 const SIRegisterInfo *TRI = ST->getRegisterInfo();
1375 TargetLowering::AsmOperandInfoVector TargetConstraints =
1376 TLI->ParseConstraints(DL, TRI: ST->getRegisterInfo(), Call: *CI);
1377
1378 const int TargetOutputIdx = Indices.empty() ? -1 : Indices[0];
1379
1380 int OutputIdx = 0;
1381 for (auto &TC : TargetConstraints) {
1382 if (TC.Type != InlineAsm::isOutput)
1383 continue;
1384
1385 // Skip outputs we don't care about.
1386 if (TargetOutputIdx != -1 && TargetOutputIdx != OutputIdx++)
1387 continue;
1388
1389 TLI->ComputeConstraintToUse(OpInfo&: TC, Op: SDValue());
1390
1391 const TargetRegisterClass *RC = TLI->getRegForInlineAsmConstraint(
1392 TRI, Constraint: TC.ConstraintCode, VT: TC.ConstraintVT).second;
1393
1394 // For AGPR constraints null is returned on subtargets without AGPRs, so
1395 // assume divergent for null.
1396 if (!RC || !TRI->isSGPRClass(RC))
1397 return true;
1398 }
1399
1400 return false;
1401}
1402
1403bool GCNTTIImpl::isReadRegisterSourceOfDivergence(
1404 const IntrinsicInst *ReadReg) const {
1405 Metadata *MD =
1406 cast<MetadataAsValue>(Val: ReadReg->getArgOperand(i: 0))->getMetadata();
1407 StringRef RegName =
1408 cast<MDString>(Val: cast<MDNode>(Val: MD)->getOperand(I: 0))->getString();
1409
1410 // Special case registers that look like VCC.
1411 MVT VT = MVT::getVT(Ty: ReadReg->getType());
1412 if (VT == MVT::i1)
1413 return true;
1414
1415 // Special case scalar registers that start with 'v'.
1416 if (RegName.starts_with(Prefix: "vcc") || RegName.empty())
1417 return false;
1418
1419 // VGPR or AGPR is divergent. There aren't any specially named vector
1420 // registers.
1421 return RegName[0] == 'v' || RegName[0] == 'a';
1422}
1423
1424/// \returns true if the result of the value could potentially be
1425/// different across workitems in a wavefront.
1426bool GCNTTIImpl::isSourceOfDivergence(const Value *V) const {
1427 if (const Argument *A = dyn_cast<Argument>(Val: V))
1428 return !AMDGPU::isArgPassedInSGPR(Arg: A);
1429
1430 // Loads from the private and flat address spaces are divergent, because
1431 // threads can execute the load instruction with the same inputs and get
1432 // different results.
1433 //
1434 // All other loads are not divergent, because if threads issue loads with the
1435 // same arguments, they will always get the same result.
1436 if (const LoadInst *Load = dyn_cast<LoadInst>(Val: V))
1437 return Load->getPointerAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
1438 Load->getPointerAddressSpace() == AMDGPUAS::FLAT_ADDRESS;
1439
1440 // Atomics are divergent because they are executed sequentially: when an
1441 // atomic operation refers to the same address in each thread, then each
1442 // thread after the first sees the value written by the previous thread as
1443 // original value.
1444 if (isa<AtomicRMWInst, AtomicCmpXchgInst>(Val: V))
1445 return true;
1446
1447 if (const IntrinsicInst *Intrinsic = dyn_cast<IntrinsicInst>(Val: V)) {
1448 Intrinsic::ID IID = Intrinsic->getIntrinsicID();
1449 switch (IID) {
1450 case Intrinsic::read_register:
1451 return isReadRegisterSourceOfDivergence(ReadReg: Intrinsic);
1452 case Intrinsic::amdgcn_workitem_id_y:
1453 case Intrinsic::amdgcn_workitem_id_z: {
1454 const Function *F = Intrinsic->getFunction();
1455 bool HasUniformYZ =
1456 ST->hasWavefrontsEvenlySplittingXDim(F: *F, /*RequiresUniformYZ=*/true);
1457 std::optional<unsigned> ThisDimSize = ST->getReqdWorkGroupSize(
1458 F: *F, Dim: IID == Intrinsic::amdgcn_workitem_id_y ? 1 : 2);
1459 return !HasUniformYZ && (!ThisDimSize || *ThisDimSize != 1);
1460 }
1461 default:
1462 return AMDGPU::isIntrinsicSourceOfDivergence(IntrID: IID);
1463 }
1464 }
1465
1466 // Assume all function calls are a source of divergence.
1467 if (const CallInst *CI = dyn_cast<CallInst>(Val: V)) {
1468 if (CI->isInlineAsm())
1469 return isInlineAsmSourceOfDivergence(CI);
1470 return true;
1471 }
1472
1473 // Assume all function calls are a source of divergence.
1474 if (isa<InvokeInst>(Val: V))
1475 return true;
1476
1477 // If the target supports globally addressable scratch, the mapping from
1478 // scratch memory to the flat aperture changes therefore an address space cast
1479 // is no longer uniform.
1480 if (auto *CastI = dyn_cast<AddrSpaceCastInst>(Val: V)) {
1481 return CastI->getSrcAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS &&
1482 CastI->getDestAddressSpace() == AMDGPUAS::FLAT_ADDRESS &&
1483 ST->hasGloballyAddressableScratch();
1484 }
1485
1486 return false;
1487}
1488
1489bool GCNTTIImpl::isAlwaysUniform(const Value *V) const {
1490 if (const IntrinsicInst *Intrinsic = dyn_cast<IntrinsicInst>(Val: V))
1491 return AMDGPU::isIntrinsicAlwaysUniform(IntrID: Intrinsic->getIntrinsicID());
1492
1493 if (const CallInst *CI = dyn_cast<CallInst>(Val: V)) {
1494 if (CI->isInlineAsm())
1495 return !isInlineAsmSourceOfDivergence(CI);
1496 return false;
1497 }
1498
1499 // In most cases TID / wavefrontsize is uniform.
1500 //
1501 // However, if a kernel has uneven dimesions we can have a value of
1502 // workitem-id-x divided by the wavefrontsize non-uniform. For example
1503 // dimensions (65, 2) will have workitems with address (64, 0) and (0, 1)
1504 // packed into a same wave which gives 1 and 0 after the division by 64
1505 // respectively.
1506 //
1507 // The X dimension doesn't reset within a wave if either both the Y
1508 // and Z dimensions are of length 1, or if the X dimension's required
1509 // size is a power of 2. Note, however, if the X dimension's maximum
1510 // size is a power of 2 < the wavefront size, division by the wavefront
1511 // size is guaranteed to yield 0, so this is also a no-reset case.
1512 bool XDimDoesntResetWithinWaves = false;
1513 if (auto *I = dyn_cast<Instruction>(Val: V)) {
1514 const Function *F = I->getFunction();
1515 XDimDoesntResetWithinWaves = ST->hasWavefrontsEvenlySplittingXDim(F: *F);
1516 }
1517 using namespace llvm::PatternMatch;
1518 uint64_t C;
1519 auto MatchTidXCall = m_Intrinsic<Intrinsic::amdgcn_workitem_id_x>();
1520 auto MaybeMaskedTidX =
1521 m_CombineOr(Ps: m_c_And(L: MatchTidXCall, R: m_Constant()), Ps: MatchTidXCall);
1522 auto MaybeCastTidX = m_CastOrSelf(Op: MaybeMaskedTidX);
1523 auto MaybeMaskedCastTidX =
1524 m_CombineOr(Ps: m_c_And(L: MaybeCastTidX, R: m_Constant()), Ps: MaybeCastTidX);
1525 if (match(V, P: m_LShr(L: MaybeMaskedCastTidX, R: m_ConstantInt(V&: C))))
1526 return C >= ST->getWavefrontSizeLog2() && XDimDoesntResetWithinWaves;
1527
1528 Constant *Mask;
1529 if (match(V, P: m_c_And(
1530 L: m_CastOrSelf(Op: m_Intrinsic<Intrinsic::amdgcn_workitem_id_x>()),
1531 R: m_Constant(C&: Mask)))) {
1532 return computeKnownBits(V: Mask, DL).countMinTrailingZeros() >=
1533 ST->getWavefrontSizeLog2() &&
1534 XDimDoesntResetWithinWaves;
1535 }
1536
1537 const ExtractValueInst *ExtValue = dyn_cast<ExtractValueInst>(Val: V);
1538 if (!ExtValue)
1539 return false;
1540
1541 const CallInst *CI = dyn_cast<CallInst>(Val: ExtValue->getOperand(i_nocapture: 0));
1542 if (!CI)
1543 return false;
1544
1545 if (const IntrinsicInst *Intrinsic = dyn_cast<IntrinsicInst>(Val: CI)) {
1546 switch (Intrinsic->getIntrinsicID()) {
1547 default:
1548 return false;
1549 case Intrinsic::amdgcn_if:
1550 case Intrinsic::amdgcn_else: {
1551 ArrayRef<unsigned> Indices = ExtValue->getIndices();
1552 return Indices.size() == 1 && Indices[0] == 1;
1553 }
1554 }
1555 }
1556
1557 // If we have inline asm returning mixed SGPR and VGPR results, we inferred
1558 // divergent for the overall struct return. We need to override it in the
1559 // case we're extracting an SGPR component here.
1560 if (CI->isInlineAsm())
1561 return !isInlineAsmSourceOfDivergence(CI, Indices: ExtValue->getIndices());
1562
1563 return false;
1564}
1565
1566bool GCNTTIImpl::collectFlatAddressOperands(SmallVectorImpl<int> &OpIndexes,
1567 Intrinsic::ID IID) const {
1568 switch (IID) {
1569 case Intrinsic::amdgcn_is_shared:
1570 case Intrinsic::amdgcn_is_private:
1571 case Intrinsic::amdgcn_flat_atomic_fmax_num:
1572 case Intrinsic::amdgcn_flat_atomic_fmin_num:
1573 case Intrinsic::amdgcn_load_to_lds:
1574 case Intrinsic::amdgcn_make_buffer_rsrc:
1575 OpIndexes.push_back(Elt: 0);
1576 return true;
1577 default:
1578 return false;
1579 }
1580}
1581
1582Value *GCNTTIImpl::rewriteIntrinsicWithAddressSpace(IntrinsicInst *II,
1583 Value *OldV,
1584 Value *NewV) const {
1585 auto IntrID = II->getIntrinsicID();
1586 switch (IntrID) {
1587 case Intrinsic::amdgcn_is_shared:
1588 case Intrinsic::amdgcn_is_private: {
1589 unsigned TrueAS = IntrID == Intrinsic::amdgcn_is_shared ?
1590 AMDGPUAS::LOCAL_ADDRESS : AMDGPUAS::PRIVATE_ADDRESS;
1591 unsigned NewAS = NewV->getType()->getPointerAddressSpace();
1592 LLVMContext &Ctx = NewV->getType()->getContext();
1593 ConstantInt *NewVal = (TrueAS == NewAS) ?
1594 ConstantInt::getTrue(Context&: Ctx) : ConstantInt::getFalse(Context&: Ctx);
1595 return NewVal;
1596 }
1597 case Intrinsic::amdgcn_flat_atomic_fmax_num:
1598 case Intrinsic::amdgcn_flat_atomic_fmin_num: {
1599 Type *DestTy = II->getType();
1600 Type *SrcTy = NewV->getType();
1601 unsigned NewAS = SrcTy->getPointerAddressSpace();
1602 if (!AMDGPU::isExtendedGlobalAddrSpace(AS: NewAS))
1603 return nullptr;
1604 Module *M = II->getModule();
1605 Function *NewDecl = Intrinsic::getOrInsertDeclaration(
1606 M, id: II->getIntrinsicID(), OverloadTys: {DestTy, SrcTy, DestTy});
1607 II->setArgOperand(i: 0, v: NewV);
1608 II->setCalledFunction(NewDecl);
1609 return II;
1610 }
1611 case Intrinsic::amdgcn_load_to_lds: {
1612 Type *SrcTy = NewV->getType();
1613 Module *M = II->getModule();
1614 Function *NewDecl =
1615 Intrinsic::getOrInsertDeclaration(M, id: II->getIntrinsicID(), OverloadTys: {SrcTy});
1616 II->setArgOperand(i: 0, v: NewV);
1617 II->setCalledFunction(NewDecl);
1618 return II;
1619 }
1620 case Intrinsic::amdgcn_make_buffer_rsrc: {
1621 Type *SrcTy = NewV->getType();
1622 Type *DstTy = II->getType();
1623 Type *NumRecordsTy = II->getArgOperand(i: 2)->getType();
1624 Module *M = II->getModule();
1625 Function *NewDecl = Intrinsic::getOrInsertDeclaration(
1626 M, id: II->getIntrinsicID(), OverloadTys: {DstTy, SrcTy, NumRecordsTy});
1627 II->setArgOperand(i: 0, v: NewV);
1628 II->setCalledFunction(NewDecl);
1629 return II;
1630 }
1631 default:
1632 return nullptr;
1633 }
1634}
1635
1636InstructionCost GCNTTIImpl::getShuffleCost(
1637 TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy,
1638 TTI::TargetCostKind CostKind, ArrayRef<int> Mask, int Index,
1639 VectorType *SubTp, ArrayRef<const Value *> Args, const Instruction *CtxI,
1640 TTI::VectorInstrContext VIC) const {
1641 if (!isa<FixedVectorType>(Val: SrcTy))
1642 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
1643 SubTp);
1644
1645 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTy&: SubTp);
1646
1647 unsigned ScalarSize = DL.getTypeSizeInBits(Ty: SrcTy->getElementType());
1648 if (ST->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS &&
1649 (ScalarSize == 16 || ScalarSize == 8)) {
1650 // Larger vector widths may require additional instructions, but are
1651 // typically cheaper than scalarized versions.
1652 //
1653 // We assume that shuffling at a register granularity can be done for free.
1654 // This is not true for vectors fed into memory instructions, but it is
1655 // effectively true for all other shuffling. The emphasis of the logic here
1656 // is to assist generic transform in cleaning up / canonicalizing those
1657 // shuffles.
1658
1659 // With op_sel VOP3P instructions freely can access the low half or high
1660 // half of a register, so any swizzle of two elements is free.
1661 if (auto *SrcVecTy = dyn_cast<FixedVectorType>(Val: SrcTy)) {
1662 unsigned NumSrcElts = SrcVecTy->getNumElements();
1663 if (ST->hasVOP3PInsts() && ScalarSize == 16 && NumSrcElts == 2 &&
1664 (Kind == TTI::SK_Broadcast || Kind == TTI::SK_Reverse ||
1665 Kind == TTI::SK_PermuteSingleSrc))
1666 return 0;
1667 }
1668
1669 unsigned EltsPerReg = 32 / ScalarSize;
1670 switch (Kind) {
1671 case TTI::SK_Broadcast:
1672 // A single v_perm_b32 can be re-used for all destination registers.
1673 return 1;
1674 case TTI::SK_Reverse:
1675 // One instruction per register.
1676 if (auto *DstVecTy = dyn_cast<FixedVectorType>(Val: DstTy))
1677 return divideCeil(Numerator: DstVecTy->getNumElements(), Denominator: EltsPerReg);
1678 return InstructionCost::getInvalid();
1679 case TTI::SK_ExtractSubvector:
1680 if (Index % EltsPerReg == 0)
1681 return 0; // Shuffling at register granularity
1682 if (auto *DstVecTy = dyn_cast<FixedVectorType>(Val: DstTy))
1683 return divideCeil(Numerator: DstVecTy->getNumElements(), Denominator: EltsPerReg);
1684 return InstructionCost::getInvalid();
1685 case TTI::SK_InsertSubvector: {
1686 auto *DstVecTy = dyn_cast<FixedVectorType>(Val: DstTy);
1687 if (!DstVecTy)
1688 return InstructionCost::getInvalid();
1689 unsigned NumDstElts = DstVecTy->getNumElements();
1690 unsigned NumInsertElts = cast<FixedVectorType>(Val: SubTp)->getNumElements();
1691 unsigned EndIndex = Index + NumInsertElts;
1692 unsigned BeginSubIdx = Index % EltsPerReg;
1693 unsigned EndSubIdx = EndIndex % EltsPerReg;
1694 unsigned Cost = 0;
1695
1696 if (BeginSubIdx != 0) {
1697 // Need to shift the inserted vector into place. The cost is the number
1698 // of destination registers overlapped by the inserted vector.
1699 Cost = divideCeil(Numerator: EndIndex, Denominator: EltsPerReg) - (Index / EltsPerReg);
1700 }
1701
1702 // If the last register overlap is partial, there may be three source
1703 // registers feeding into it; that takes an extra instruction.
1704 if (EndIndex < NumDstElts && BeginSubIdx < EndSubIdx)
1705 Cost += 1;
1706
1707 return Cost;
1708 }
1709 case TTI::SK_Splice: {
1710 auto *DstVecTy = dyn_cast<FixedVectorType>(Val: DstTy);
1711 if (!DstVecTy)
1712 return InstructionCost::getInvalid();
1713 unsigned NumElts = DstVecTy->getNumElements();
1714 assert(NumElts == cast<FixedVectorType>(SrcTy)->getNumElements());
1715 // Determine the sub-region of the result vector that requires
1716 // sub-register shuffles / mixing.
1717 unsigned EltsFromLHS = NumElts - Index;
1718 bool LHSIsAligned = (Index % EltsPerReg) == 0;
1719 bool RHSIsAligned = (EltsFromLHS % EltsPerReg) == 0;
1720 if (LHSIsAligned && RHSIsAligned)
1721 return 0;
1722 if (LHSIsAligned && !RHSIsAligned)
1723 return divideCeil(Numerator: NumElts, Denominator: EltsPerReg) - (EltsFromLHS / EltsPerReg);
1724 if (!LHSIsAligned && RHSIsAligned)
1725 return divideCeil(Numerator: EltsFromLHS, Denominator: EltsPerReg);
1726 return divideCeil(Numerator: NumElts, Denominator: EltsPerReg);
1727 }
1728 default:
1729 break;
1730 }
1731
1732 if (!Mask.empty()) {
1733 unsigned NumSrcElts = cast<FixedVectorType>(Val: SrcTy)->getNumElements();
1734
1735 // Generically estimate the cost by assuming that each destination
1736 // register is derived from sources via v_perm_b32 instructions if it
1737 // can't be copied as-is.
1738 //
1739 // For each destination register, derive the cost of obtaining it based
1740 // on the number of source registers that feed into it.
1741 unsigned Cost = 0;
1742 for (unsigned DstIdx = 0; DstIdx < Mask.size(); DstIdx += EltsPerReg) {
1743 SmallVector<int, 4> Regs;
1744 bool Aligned = true;
1745 for (unsigned I = 0; I < EltsPerReg && DstIdx + I < Mask.size(); ++I) {
1746 int SrcIdx = Mask[DstIdx + I];
1747 if (SrcIdx == -1)
1748 continue;
1749 int Reg;
1750 if (SrcIdx < (int)NumSrcElts) {
1751 Reg = SrcIdx / EltsPerReg;
1752 if (SrcIdx % EltsPerReg != I)
1753 Aligned = false;
1754 } else {
1755 Reg = NumSrcElts + (SrcIdx - NumSrcElts) / EltsPerReg;
1756 if ((SrcIdx - NumSrcElts) % EltsPerReg != I)
1757 Aligned = false;
1758 }
1759 if (!llvm::is_contained(Range&: Regs, Element: Reg))
1760 Regs.push_back(Elt: Reg);
1761 }
1762 if (Regs.size() >= 2)
1763 Cost += Regs.size() - 1;
1764 else if (!Aligned)
1765 Cost += 1;
1766 }
1767 return Cost;
1768 }
1769 }
1770
1771 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
1772 SubTp);
1773}
1774
1775/// Whether it is profitable to sink the operands of an
1776/// Instruction I to the basic block of I.
1777/// This helps using several modifiers (like abs and neg) more often.
1778bool GCNTTIImpl::isProfitableToSinkOperands(Instruction *I,
1779 SmallVectorImpl<Use *> &Ops) const {
1780 using namespace PatternMatch;
1781
1782 // The cost model prices this fmul as free assuming it fuses with its
1783 // fadd/fsub user, which needs them in one block. Sink a stranded
1784 // loop-invariant fmul back to the user when they would fuse. Single use only,
1785 // so this stays a move.
1786 if (I->getOpcode() == Instruction::FAdd ||
1787 I->getOpcode() == Instruction::FSub) {
1788 const Instruction *FMul = getFusedFMul(TLI: *TLI, Ty: I->getType(), FAddSub: I);
1789 if (FMul && FMul->getParent() != I->getParent())
1790 Ops.push_back(Elt: &I->getOperandUse(i: I->getOperand(i: 0) == FMul ? 0 : 1));
1791 }
1792
1793 for (auto &Op : I->operands()) {
1794 // Ensure we are not already sinking this operand.
1795 if (any_of(Range&: Ops, P: [&](Use *U) { return U->get() == Op.get(); }))
1796 continue;
1797
1798 if (match(V: &Op, P: m_FAbs(Op0: m_Value())) || match(V: &Op, P: m_FNeg(X: m_Value()))) {
1799 Ops.push_back(Elt: &Op);
1800 continue;
1801 }
1802
1803 // Check for zero-cost multiple use InsertElement/ExtractElement
1804 // instructions
1805 if (Instruction *OpInst = dyn_cast<Instruction>(Val: Op.get())) {
1806 if (OpInst->getType()->isVectorTy() && OpInst->getNumOperands() > 1) {
1807 Instruction *VecOpInst = dyn_cast<Instruction>(Val: OpInst->getOperand(i: 0));
1808 if (VecOpInst && VecOpInst->hasOneUse())
1809 continue;
1810
1811 if (getVectorInstrCost(Opcode: OpInst->getOpcode(), ValTy: OpInst->getType(),
1812 CostKind: TTI::TCK_RecipThroughput, Index: 0,
1813 Op0: OpInst->getOperand(i: 0),
1814 Op1: OpInst->getOperand(i: 1)) == 0) {
1815 Ops.push_back(Elt: &Op);
1816 continue;
1817 }
1818 }
1819 }
1820
1821 if (auto *Shuffle = dyn_cast<ShuffleVectorInst>(Val: Op.get())) {
1822
1823 unsigned EltSize = DL.getTypeSizeInBits(
1824 Ty: cast<VectorType>(Val: Shuffle->getType())->getElementType());
1825
1826 // For i32 (or greater) shufflevectors, these will be lowered into a
1827 // series of insert / extract elements, which will be coalesced away.
1828 if (EltSize < 16 || !ST->has16BitInsts())
1829 continue;
1830
1831 int NumSubElts, SubIndex;
1832 if (Shuffle->changesLength()) {
1833 if (Shuffle->increasesLength() && Shuffle->isIdentityWithPadding()) {
1834 Ops.push_back(Elt: &Op);
1835 continue;
1836 }
1837
1838 if ((Shuffle->isExtractSubvectorMask(Index&: SubIndex) ||
1839 Shuffle->isInsertSubvectorMask(NumSubElts, Index&: SubIndex)) &&
1840 !(SubIndex & 0x1)) {
1841 Ops.push_back(Elt: &Op);
1842 continue;
1843 }
1844 }
1845
1846 if (Shuffle->isReverse() || Shuffle->isZeroEltSplat() ||
1847 Shuffle->isSingleSource()) {
1848 Ops.push_back(Elt: &Op);
1849 continue;
1850 }
1851 }
1852 }
1853
1854 return !Ops.empty();
1855}
1856
1857bool GCNTTIImpl::areInlineCompatible(const Function *Caller,
1858 const Function *Callee) const {
1859 const TargetMachine &TM = getTLI()->getTargetMachine();
1860 const GCNSubtarget *CallerST
1861 = static_cast<const GCNSubtarget *>(TM.getSubtargetImpl(*Caller));
1862 const GCNSubtarget *CalleeST
1863 = static_cast<const GCNSubtarget *>(TM.getSubtargetImpl(*Callee));
1864
1865 if (!BaseT::areInlineCompatible(Caller, Callee))
1866 return false;
1867
1868 // FIXME: dx10_clamp can just take the caller setting, but there seems to be
1869 // no way to support merge for backend defined attributes.
1870 SIModeRegisterDefaults CallerMode(*Caller, *CallerST);
1871 SIModeRegisterDefaults CalleeMode(*Callee, *CalleeST);
1872 if (!CallerMode.isInlineCompatible(CalleeMode))
1873 return false;
1874
1875 if (Callee->hasFnAttribute(Kind: Attribute::AlwaysInline) ||
1876 Callee->hasFnAttribute(Kind: Attribute::InlineHint))
1877 return true;
1878
1879 // Hack to make compile times reasonable.
1880 if (InlineMaxBB) {
1881 // Single BB does not increase total BB amount.
1882 if (Callee->size() == 1)
1883 return true;
1884 size_t BBSize = Caller->size() + Callee->size() - 1;
1885 if (BBSize > InlineMaxBB) {
1886 LLVM_DEBUG(dbgs() << "AMDGPU inline max-BB rejected inlining "
1887 << Callee->getName() << " into " << Caller->getName()
1888 << ": caller BBs=" << Caller->size() << ", callee BBs="
1889 << Callee->size() << ", combined BBs=" << BBSize
1890 << ", max BBs=" << InlineMaxBB << '\n');
1891 return false;
1892 }
1893 }
1894
1895 return true;
1896}
1897
1898static unsigned adjustInliningThresholdUsingCallee(const CallBase *CB,
1899 const SITargetLowering *TLI,
1900 const GCNTTIImpl *TTIImpl) {
1901 const int NrOfSGPRUntilSpill = 26;
1902 const int NrOfVGPRUntilSpill = 32;
1903
1904 const DataLayout &DL = TTIImpl->getDataLayout();
1905
1906 unsigned adjustThreshold = 0;
1907 int SGPRsInUse = 0;
1908 int VGPRsInUse = 0;
1909 for (const Use &A : CB->args()) {
1910 SmallVector<EVT, 4> ValueVTs;
1911 ComputeValueVTs(TLI: *TLI, DL, Ty: A.get()->getType(), ValueVTs);
1912 for (auto ArgVT : ValueVTs) {
1913 unsigned CCRegNum = TLI->getNumRegistersForCallingConv(
1914 Context&: CB->getContext(), CC: CB->getCallingConv(), VT: ArgVT);
1915 if (AMDGPU::isArgPassedInSGPR(CB, ArgNo: CB->getArgOperandNo(U: &A)))
1916 SGPRsInUse += CCRegNum;
1917 else
1918 VGPRsInUse += CCRegNum;
1919 }
1920 }
1921
1922 // The cost of passing function arguments through the stack:
1923 // 1 instruction to put a function argument on the stack in the caller.
1924 // 1 instruction to take a function argument from the stack in callee.
1925 // 1 instruction is explicitly take care of data dependencies in callee
1926 // function.
1927 InstructionCost ArgStackCost(1);
1928 ArgStackCost += const_cast<GCNTTIImpl *>(TTIImpl)->getMemoryOpCost(
1929 Opcode: Instruction::Store, Src: Type::getInt32Ty(C&: CB->getContext()), Alignment: Align(4),
1930 AddressSpace: AMDGPUAS::PRIVATE_ADDRESS, CostKind: TTI::TCK_SizeAndLatency);
1931 ArgStackCost += const_cast<GCNTTIImpl *>(TTIImpl)->getMemoryOpCost(
1932 Opcode: Instruction::Load, Src: Type::getInt32Ty(C&: CB->getContext()), Alignment: Align(4),
1933 AddressSpace: AMDGPUAS::PRIVATE_ADDRESS, CostKind: TTI::TCK_SizeAndLatency);
1934
1935 // The penalty cost is computed relative to the cost of instructions and does
1936 // not model any storage costs.
1937 adjustThreshold += std::max(a: 0, b: SGPRsInUse - NrOfSGPRUntilSpill) *
1938 ArgStackCost.getValue() * InlineConstants::getInstrCost();
1939 adjustThreshold += std::max(a: 0, b: VGPRsInUse - NrOfVGPRUntilSpill) *
1940 ArgStackCost.getValue() * InlineConstants::getInstrCost();
1941 return adjustThreshold;
1942}
1943
1944static unsigned getCallArgsTotalAllocaSize(const CallBase *CB,
1945 const DataLayout &DL) {
1946 // If we have a pointer to a private array passed into a function
1947 // it will not be optimized out, leaving scratch usage.
1948 // This function calculates the total size in bytes of the memory that would
1949 // end in scratch if the call was not inlined.
1950 unsigned AllocaSize = 0;
1951 SmallPtrSet<const AllocaInst *, 8> AIVisited;
1952 for (Value *PtrArg : CB->args()) {
1953 PointerType *Ty = dyn_cast<PointerType>(Val: PtrArg->getType());
1954 if (!Ty)
1955 continue;
1956
1957 unsigned AddrSpace = Ty->getAddressSpace();
1958 if (AddrSpace != AMDGPUAS::FLAT_ADDRESS &&
1959 AddrSpace != AMDGPUAS::PRIVATE_ADDRESS)
1960 continue;
1961
1962 const AllocaInst *AI = dyn_cast<AllocaInst>(Val: getUnderlyingObject(V: PtrArg));
1963 if (!AI || !AI->isStaticAlloca() || !AIVisited.insert(Ptr: AI).second)
1964 continue;
1965
1966 if (auto Size = AI->getAllocationSize(DL))
1967 AllocaSize += Size->getFixedValue();
1968 }
1969 return AllocaSize;
1970}
1971
1972int GCNTTIImpl::getInliningLastCallToStaticBonus() const {
1973 return BaseT::getInliningLastCallToStaticBonus() *
1974 getInliningThresholdMultiplier();
1975}
1976
1977unsigned GCNTTIImpl::adjustInliningThreshold(const CallBase *CB) const {
1978 unsigned Threshold = adjustInliningThresholdUsingCallee(CB, TLI, TTIImpl: this);
1979
1980 // Private object passed as arguments may end up in scratch usage if the call
1981 // is not inlined. Increase the inline threshold to promote inlining.
1982 unsigned AllocaSize = getCallArgsTotalAllocaSize(CB, DL);
1983 if (AllocaSize > 0)
1984 Threshold += ArgAllocaCost;
1985 return Threshold;
1986}
1987
1988unsigned GCNTTIImpl::getCallerAllocaCost(const CallBase *CB,
1989 const AllocaInst *AI) const {
1990
1991 // Below the cutoff, assume that the private memory objects would be
1992 // optimized
1993 auto AllocaSize = getCallArgsTotalAllocaSize(CB, DL);
1994 if (AllocaSize <= ArgAllocaCutoff)
1995 return 0;
1996
1997 // Above the cutoff, we give a cost to each private memory object
1998 // depending its size. If the array can be optimized by SROA this cost is not
1999 // added to the total-cost in the inliner cost analysis.
2000 //
2001 // We choose the total cost of the alloca such that their sum cancels the
2002 // bonus given in the threshold (ArgAllocaCost).
2003 //
2004 // Cost_Alloca_0 + ... + Cost_Alloca_N == ArgAllocaCost
2005 //
2006 // Awkwardly, the ArgAllocaCost bonus is multiplied by threshold-multiplier,
2007 // the single-bb bonus and the vector-bonus.
2008 //
2009 // We compensate the first two multipliers, by repeating logic from the
2010 // inliner-cost in here. The vector-bonus is 0 on AMDGPU.
2011 static_assert(InlinerVectorBonusPercent == 0, "vector bonus assumed to be 0");
2012 unsigned Threshold = ArgAllocaCost * getInliningThresholdMultiplier();
2013
2014 bool SingleBB = none_of(Range&: *CB->getCalledFunction(), P: [](const BasicBlock &BB) {
2015 return BB.getTerminator()->getNumSuccessors() > 1;
2016 });
2017 if (SingleBB) {
2018 Threshold += Threshold / 2;
2019 }
2020
2021 auto ArgAllocaSize = AI->getAllocationSize(DL);
2022 if (!ArgAllocaSize)
2023 return 0;
2024
2025 // Attribute the bonus proportionally to the alloca size
2026 unsigned AllocaThresholdBonus =
2027 (Threshold * ArgAllocaSize->getFixedValue()) / AllocaSize;
2028
2029 return AllocaThresholdBonus;
2030}
2031
2032void GCNTTIImpl::getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
2033 TTI::UnrollingPreferences &UP,
2034 OptimizationRemarkEmitter *ORE) const {
2035 CommonTTI.getUnrollingPreferences(L, SE, UP, ORE);
2036}
2037
2038void GCNTTIImpl::getPeelingPreferences(Loop *L, ScalarEvolution &SE,
2039 TTI::PeelingPreferences &PP) const {
2040 CommonTTI.getPeelingPreferences(L, SE, PP);
2041}
2042
2043int GCNTTIImpl::getTransInstrCost(TTI::TargetCostKind CostKind) const {
2044 return getQuarterRateInstrCost(CostKind);
2045}
2046
2047int GCNTTIImpl::get64BitInstrCost(TTI::TargetCostKind CostKind) const {
2048 return ST->hasFullRate64Ops()
2049 ? getFullRateInstrCost()
2050 : ST->hasHalfRate64Ops() ? getHalfRateInstrCost(CostKind)
2051 : getQuarterRateInstrCost(CostKind);
2052}
2053
2054std::pair<InstructionCost, MVT>
2055GCNTTIImpl::getTypeLegalizationCost(Type *Ty) const {
2056 std::pair<InstructionCost, MVT> Cost = BaseT::getTypeLegalizationCost(Ty);
2057 auto Size = DL.getTypeSizeInBits(Ty);
2058 // Maximum load or store can handle 8 dwords for scalar and 4 for
2059 // vector ALU. Let's assume anything above 8 dwords is expensive
2060 // even if legal.
2061 if (Size <= 256)
2062 return Cost;
2063
2064 Cost.first += (Size + 255) / 256;
2065 return Cost;
2066}
2067
2068unsigned GCNTTIImpl::getCacheLineSize() const {
2069 if (ST->hasVmemPrefInsts() || ST->hasSmemPrefetchInsts())
2070 return ST->getDataCacheLineSize();
2071 return 0;
2072}
2073
2074unsigned GCNTTIImpl::getPrefetchDistance() const {
2075 return ST->hasPrefetch() ? 128 : 0;
2076}
2077
2078bool GCNTTIImpl::shouldPrefetchAddressSpace(unsigned AS) const {
2079 return AMDGPU::isFlatGlobalAddrSpace(AS);
2080}
2081
2082void GCNTTIImpl::collectKernelLaunchBounds(
2083 const Function &F,
2084 SmallVectorImpl<std::pair<StringRef, int64_t>> &LB) const {
2085 SmallVector<unsigned> MaxNumWorkgroups = AMDGPU::getMaxNumWorkGroups(F);
2086 LB.push_back(Elt: {"amdgpu-max-num-workgroups[0]", MaxNumWorkgroups[0]});
2087 LB.push_back(Elt: {"amdgpu-max-num-workgroups[1]", MaxNumWorkgroups[1]});
2088 LB.push_back(Elt: {"amdgpu-max-num-workgroups[2]", MaxNumWorkgroups[2]});
2089 std::pair<unsigned, unsigned> FlatWorkGroupSize =
2090 ST->getFlatWorkGroupSizes(F);
2091 LB.push_back(Elt: {"amdgpu-flat-work-group-size[0]", FlatWorkGroupSize.first});
2092 LB.push_back(Elt: {"amdgpu-flat-work-group-size[1]", FlatWorkGroupSize.second});
2093 std::pair<unsigned, unsigned> WavesPerEU = ST->getWavesPerEU(F);
2094 LB.push_back(Elt: {"amdgpu-waves-per-eu[0]", WavesPerEU.first});
2095 LB.push_back(Elt: {"amdgpu-waves-per-eu[1]", WavesPerEU.second});
2096}
2097
2098GCNTTIImpl::KnownIEEEMode
2099GCNTTIImpl::fpenvIEEEMode(const Instruction &I) const {
2100 if (!ST->hasFeature(Feature: AMDGPU::FeatureDX10ClampAndIEEEMode))
2101 return KnownIEEEMode::On; // Only mode on gfx1170+
2102
2103 const Function *F = I.getFunction();
2104 if (!F)
2105 return KnownIEEEMode::Unknown;
2106
2107 Attribute IEEEAttr = F->getFnAttribute(Kind: "amdgpu-ieee");
2108 if (IEEEAttr.isValid())
2109 return IEEEAttr.getValueAsBool() ? KnownIEEEMode::On : KnownIEEEMode::Off;
2110
2111 return AMDGPU::isShader(CC: F->getCallingConv()) ? KnownIEEEMode::Off
2112 : KnownIEEEMode::On;
2113}
2114
2115InstructionCost GCNTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src,
2116 Align Alignment,
2117 unsigned AddressSpace,
2118 TTI::TargetCostKind CostKind,
2119 TTI::OperandValueInfo OpInfo,
2120 const Instruction *I) const {
2121 if (VectorType *VecTy = dyn_cast<VectorType>(Val: Src)) {
2122 if ((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
2123 CostKind != TTI::TCK_Latency &&
2124 VecTy->getElementType()->isIntegerTy(BitWidth: 8)) {
2125 return divideCeil(Numerator: DL.getTypeSizeInBits(Ty: VecTy) - 1,
2126 Denominator: getLoadStoreVecRegBitWidth(AddrSpace: AddressSpace));
2127 }
2128 }
2129 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace, CostKind,
2130 OpInfo, I);
2131}
2132
2133unsigned GCNTTIImpl::getNumberOfParts(Type *Tp) const {
2134 if (VectorType *VecTy = dyn_cast<VectorType>(Val: Tp)) {
2135 if (VecTy->getElementType()->isIntegerTy(BitWidth: 8)) {
2136 unsigned ElementCount = VecTy->getElementCount().getFixedValue();
2137 return divideCeil(Numerator: ElementCount - 1, Denominator: 4);
2138 }
2139 }
2140 return BaseT::getNumberOfParts(Tp);
2141}
2142
2143ValueUniformity GCNTTIImpl::getValueUniformity(const Value *V) const {
2144 if (const IntrinsicInst *Intrinsic = dyn_cast<IntrinsicInst>(Val: V)) {
2145 switch (Intrinsic->getIntrinsicID()) {
2146 case Intrinsic::amdgcn_wave_shuffle:
2147 return ValueUniformity::Custom;
2148 default:
2149 break;
2150 }
2151 }
2152
2153 if (isAlwaysUniform(V))
2154 return ValueUniformity::AlwaysUniform;
2155
2156 if (isSourceOfDivergence(V))
2157 return ValueUniformity::NeverUniform;
2158
2159 return ValueUniformity::Default;
2160}
2161
2162InstructionCost GCNTTIImpl::getScalingFactorCost(Type *Ty, GlobalValue *BaseGV,
2163 StackOffset BaseOffset,
2164 bool HasBaseReg, int64_t Scale,
2165 unsigned AddrSpace) const {
2166 if (HasBaseReg && Scale != 0) {
2167 // gfx1250+ can fold base+scale*index when scale matches the memory access
2168 // size (scale_offset bit). Supported for flat/global/constant/scratch
2169 // (VMEM, max 128 bits) and constant_32bit (SMRD, capped to 128 bits here).
2170 if (getST()->hasScaleOffset() && Ty && Ty->isSized() &&
2171 (AMDGPU::isExtendedGlobalAddrSpace(AS: AddrSpace) ||
2172 AddrSpace == AMDGPUAS::FLAT_ADDRESS ||
2173 AddrSpace == AMDGPUAS::PRIVATE_ADDRESS)) {
2174 TypeSize StoreSize = getDataLayout().getTypeStoreSize(Ty);
2175 if (TypeSize::isKnownLE(LHS: StoreSize, RHS: TypeSize::getFixed(ExactSize: 16)) &&
2176 static_cast<int64_t>(StoreSize.getFixedValue()) == Scale)
2177 return 0;
2178 }
2179 return 1;
2180 }
2181 return BaseT::getScalingFactorCost(Ty, BaseGV, BaseOffset, HasBaseReg, Scale,
2182 AddrSpace);
2183}
2184
2185bool GCNTTIImpl::isLSRCostLess(const TTI::LSRCost &A,
2186 const TTI::LSRCost &B) const {
2187 // Favor lower per-iteration work over preheader/setup costs.
2188 // AMDGPU lacks rich addressing modes, so ScaleCost is folded into the
2189 // effective instruction count (base+scale*index requires a separate ADD).
2190 unsigned EffInsnsA = A.Insns + A.ScaleCost;
2191 unsigned EffInsnsB = B.Insns + B.ScaleCost;
2192
2193 return std::tie(args&: EffInsnsA, args: A.NumIVMuls, args: A.AddRecCost, args: A.NumBaseAdds,
2194 args: A.SetupCost, args: A.ImmCost, args: A.NumRegs) <
2195 std::tie(args&: EffInsnsB, args: B.NumIVMuls, args: B.AddRecCost, args: B.NumBaseAdds,
2196 args: B.SetupCost, args: B.ImmCost, args: B.NumRegs);
2197}
2198
2199bool GCNTTIImpl::isNumRegsMajorCostOfLSR() const {
2200 // isLSRCostLess de-prioritizes register count; keep consistent.
2201 return false;
2202}
2203
2204bool GCNTTIImpl::shouldDropLSRSolutionIfLessProfitable() const {
2205 // Prefer the baseline when LSR cannot clearly reduce per-iteration work.
2206 return true;
2207}
2208
2209bool GCNTTIImpl::isUniform(const Instruction *I,
2210 const SmallBitVector &UniformArgs) const {
2211 const IntrinsicInst *Intrinsic = cast<IntrinsicInst>(Val: I);
2212 switch (Intrinsic->getIntrinsicID()) {
2213 case Intrinsic::amdgcn_wave_shuffle:
2214 // wave_shuffle(Value, Index): result is uniform when either Value or Index
2215 // is uniform.
2216 return UniformArgs[0] || UniformArgs[1];
2217 default:
2218 llvm_unreachable("unexpected intrinsic in isUniform");
2219 }
2220}
2221