1//===- HexagonTargetTransformInfo.cpp - Hexagon specific TTI pass ---------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7/// \file
8/// This file implements a TargetTransformInfo analysis pass specific to the
9/// Hexagon target machine. It uses the target's detailed information to provide
10/// more precise answers to certain TTI queries, while letting the target
11/// independent and default TTI implementations handle the rest.
12///
13//===----------------------------------------------------------------------===//
14
15#include "HexagonTargetTransformInfo.h"
16#include "HexagonSubtarget.h"
17#include "llvm/Analysis/TargetTransformInfo.h"
18#include "llvm/CodeGen/ValueTypes.h"
19#include "llvm/IR/InstrTypes.h"
20#include "llvm/IR/Instructions.h"
21#include "llvm/IR/User.h"
22#include "llvm/Support/Casting.h"
23#include "llvm/Support/CommandLine.h"
24#include "llvm/Transforms/Utils/LoopPeel.h"
25#include "llvm/Transforms/Utils/UnrollLoop.h"
26
27using namespace llvm;
28
29#define DEBUG_TYPE "hexagontti"
30
31static cl::opt<bool> HexagonAutoHVX("hexagon-autohvx", cl::init(Val: false),
32 cl::Hidden, cl::desc("Enable loop vectorizer for HVX"));
33
34cl::opt<bool> HexagonAllowScatterGatherHVX(
35 "hexagon-allow-scatter-gather-hvx", cl::init(Val: false), cl::Hidden,
36 cl::desc("Allow auto-generation of HVX scatter-gather"));
37
38static cl::opt<bool> EnableV68FloatAutoHVX(
39 "force-hvx-float", cl::Hidden,
40 cl::desc("Enable auto-vectorization of floatint point types on v68."));
41
42static cl::opt<bool> EmitLookupTables("hexagon-emit-lookup-tables",
43 cl::init(Val: true), cl::Hidden,
44 cl::desc("Control lookup table emission on Hexagon target"));
45
46static cl::opt<bool> HexagonMaskedVMem("hexagon-masked-vmem", cl::init(Val: true),
47 cl::Hidden, cl::desc("Enable masked loads/stores for HVX"));
48
49// Constant "cost factor" to make floating point operations more expensive
50// in terms of vectorization cost. This isn't the best way, but it should
51// do. Ultimately, the cost should use cycles.
52static const unsigned FloatFactor = 4;
53
54bool HexagonTTIImpl::useHVX() const {
55 return ST.useHVXOps() && HexagonAutoHVX && !IsHMX;
56}
57
58bool HexagonTTIImpl::isHVXVectorType(Type *Ty) const {
59 auto *VecTy = dyn_cast<VectorType>(Val: Ty);
60 if (!VecTy)
61 return false;
62 if (!ST.isTypeForHVX(VecTy))
63 return false;
64 if (ST.useHVXV69Ops() || !VecTy->getElementType()->isFloatingPointTy())
65 return true;
66 return ST.useHVXV68Ops() && EnableV68FloatAutoHVX;
67}
68
69unsigned HexagonTTIImpl::getTypeNumElements(Type *Ty) const {
70 if (auto *VTy = dyn_cast<FixedVectorType>(Val: Ty))
71 return VTy->getNumElements();
72 assert((Ty->isIntegerTy() || Ty->isFloatingPointTy()) &&
73 "Expecting scalar type");
74 return 1;
75}
76
77TargetTransformInfo::PopcntSupportKind
78HexagonTTIImpl::getPopcntSupport(unsigned IntTyWidthInBit) const {
79 // Return fast hardware support as every input < 64 bits will be promoted
80 // to 64 bits.
81 return TargetTransformInfo::PSK_FastHardware;
82}
83
84// The Hexagon target can unroll loops with run-time trip counts.
85void HexagonTTIImpl::getUnrollingPreferences(
86 Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP,
87 OptimizationRemarkEmitter *ORE) const {
88 UP.Runtime = UP.Partial = true;
89}
90
91void HexagonTTIImpl::getPeelingPreferences(Loop *L, ScalarEvolution &SE,
92 TTI::PeelingPreferences &PP) const {
93 BaseT::getPeelingPreferences(L, SE, PP);
94 // Only try to peel innermost loops with small runtime trip counts.
95 if (L && L->isInnermost() && canPeel(L) &&
96 SE.getSmallConstantTripCount(L) == 0 &&
97 SE.getSmallConstantMaxTripCount(L) > 0 &&
98 SE.getSmallConstantMaxTripCount(L) <= 5) {
99 PP.PeelCount = 2;
100 }
101}
102
103TTI::AddressingModeKind
104HexagonTTIImpl::getPreferredAddressingMode(const Loop *L,
105 ScalarEvolution *SE) const {
106 return TTI::AMK_PostIndexed;
107}
108
109/// --- Vector TTI begin ---
110
111unsigned HexagonTTIImpl::getNumberOfRegisters(unsigned ClassID) const {
112 bool Vector = ClassID == 1;
113 if (Vector)
114 return useHVX() ? 32 : 0;
115 return 32;
116}
117
118unsigned
119HexagonTTIImpl::getMaxInterleaveFactor(ElementCount VF,
120 bool HasUnorderedReductions) const {
121 return useHVX() ? 2 : 1;
122}
123
124TypeSize
125HexagonTTIImpl::getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const {
126 switch (K) {
127 case TargetTransformInfo::RGK_Scalar:
128 return TypeSize::getFixed(ExactSize: 32);
129 case TargetTransformInfo::RGK_FixedWidthVector:
130 return TypeSize::getFixed(ExactSize: getMinVectorRegisterBitWidth());
131 case TargetTransformInfo::RGK_ScalableVector:
132 return TypeSize::getScalable(MinimumSize: 0);
133 }
134
135 llvm_unreachable("Unsupported register kind");
136}
137
138unsigned HexagonTTIImpl::getMinVectorRegisterBitWidth() const {
139 return useHVX() ? ST.getVectorLength()*8 : 32;
140}
141
142ElementCount HexagonTTIImpl::getMinimumVF(unsigned ElemWidth,
143 bool IsScalable) const {
144 assert(!IsScalable && "Scalable VFs are not supported for Hexagon");
145 return ElementCount::getFixed(MinVal: (8 * ST.getVectorLength()) / ElemWidth);
146}
147
148InstructionCost
149HexagonTTIImpl::getCallInstrCost(Function *F, Type *RetTy, ArrayRef<Type *> Tys,
150 TTI::TargetCostKind CostKind) const {
151 return BaseT::getCallInstrCost(F, RetTy, Tys, CostKind);
152}
153
154InstructionCost
155HexagonTTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
156 TTI::TargetCostKind CostKind) const {
157 if (ICA.getID() == Intrinsic::bswap) {
158 std::pair<InstructionCost, MVT> LT =
159 getTypeLegalizationCost(Ty: ICA.getReturnType());
160 return LT.first + 2;
161 }
162 return BaseT::getIntrinsicInstrCost(ICA, CostKind);
163}
164
165InstructionCost
166HexagonTTIImpl::getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE,
167 const SCEV *S,
168 TTI::TargetCostKind CostKind) const {
169 return 0;
170}
171
172InstructionCost HexagonTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src,
173 Align Alignment,
174 unsigned AddressSpace,
175 TTI::TargetCostKind CostKind,
176 TTI::OperandValueInfo OpInfo,
177 const Instruction *I) const {
178 assert(Opcode == Instruction::Load || Opcode == Instruction::Store);
179
180 // FIXME: Load latency isn't handled here
181 if (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency)
182 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
183 CostKind, OpInfo, I);
184
185 // TODO: Handle other cost kinds.
186 if (CostKind != TTI::TCK_RecipThroughput)
187 return 1;
188
189 if (Opcode == Instruction::Store)
190 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
191 CostKind, OpInfo, I);
192
193 if (Src->isVectorTy()) {
194 VectorType *VecTy = cast<VectorType>(Val: Src);
195 unsigned VecWidth = VecTy->getPrimitiveSizeInBits().getFixedValue();
196 if (isHVXVectorType(Ty: VecTy)) {
197 unsigned RegWidth =
198 getRegisterBitWidth(K: TargetTransformInfo::RGK_FixedWidthVector)
199 .getFixedValue();
200 assert(RegWidth && "Non-zero vector register width expected");
201 // Cost of HVX loads.
202 if (VecWidth % RegWidth == 0)
203 return VecWidth / RegWidth;
204 // Cost of constructing HVX vector from scalar loads
205 const Align RegAlign(RegWidth / 8);
206 if (Alignment > RegAlign)
207 Alignment = RegAlign;
208 unsigned AlignWidth = 8 * Alignment.value();
209 unsigned NumLoads = alignTo(Value: VecWidth, Align: AlignWidth) / AlignWidth;
210 return 3 * NumLoads;
211 }
212
213 // Non-HVX vectors.
214 // Add extra cost for floating point types.
215 unsigned Cost =
216 VecTy->getElementType()->isFloatingPointTy() ? FloatFactor : 1;
217
218 // At this point unspecified alignment is considered as Align(1).
219 const Align BoundAlignment = std::min(a: Alignment, b: Align(8));
220 unsigned AlignWidth = 8 * BoundAlignment.value();
221 unsigned NumLoads = alignTo(Value: VecWidth, Align: AlignWidth) / AlignWidth;
222 if (Alignment == Align(4) || Alignment == Align(8))
223 return Cost * NumLoads;
224 // Loads of less than 32 bits will need extra inserts to compose a vector.
225 assert(BoundAlignment <= Align(8));
226 unsigned LogA = Log2(A: BoundAlignment);
227 return (3 - LogA) * Cost * NumLoads;
228 }
229
230 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace, CostKind,
231 OpInfo, I);
232}
233
234InstructionCost HexagonTTIImpl::getShuffleCost(
235 TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy,
236 TTI::TargetCostKind CostKind, ArrayRef<int> Mask, int Index,
237 VectorType *SubTp, ArrayRef<const Value *> Args, const Instruction *CtxI,
238 TTI::VectorInstrContext VIC) const {
239 return 1;
240}
241
242InstructionCost HexagonTTIImpl::getInterleavedMemoryOpCost(
243 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
244 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
245 bool UseMaskForCond, bool UseMaskForGaps) const {
246 if (Indices.size() != Factor || UseMaskForCond || UseMaskForGaps)
247 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
248 Alignment, AddressSpace,
249 CostKind,
250 UseMaskForCond, UseMaskForGaps);
251 return getMemoryOpCost(Opcode, Src: VecTy, Alignment, AddressSpace, CostKind);
252}
253
254InstructionCost HexagonTTIImpl::getCmpSelInstrCost(
255 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
256 TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
257 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
258 if (ValTy->isVectorTy() && CostKind == TTI::TCK_RecipThroughput) {
259 if (!isHVXVectorType(Ty: ValTy) && ValTy->isFPOrFPVectorTy())
260 return InstructionCost::getMax();
261 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: ValTy);
262 if (Opcode == Instruction::FCmp)
263 return LT.first + FloatFactor * getTypeNumElements(Ty: ValTy);
264 }
265 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
266 Op1Info, Op2Info, I);
267}
268
269InstructionCost HexagonTTIImpl::getArithmeticInstrCost(
270 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
271 TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
272 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
273 // TODO: Handle more cost kinds.
274 if (CostKind != TTI::TCK_RecipThroughput)
275 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Opd1Info: Op1Info,
276 Opd2Info: Op2Info, Args, CtxI);
277
278 if (Ty->isVectorTy()) {
279 if (!isHVXVectorType(Ty) && Ty->isFPOrFPVectorTy())
280 return InstructionCost::getMax();
281 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
282 if (LT.second.isFloatingPoint())
283 return LT.first + FloatFactor * getTypeNumElements(Ty);
284 }
285 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Opd1Info: Op1Info, Opd2Info: Op2Info,
286 Args, CtxI);
287}
288
289InstructionCost HexagonTTIImpl::getCastInstrCost(unsigned Opcode, Type *DstTy,
290 Type *SrcTy,
291 TTI::CastContextHint CCH,
292 TTI::TargetCostKind CostKind,
293 const Instruction *I) const {
294 auto isNonHVXFP = [this] (Type *Ty) {
295 return Ty->isVectorTy() && !isHVXVectorType(Ty) && Ty->isFPOrFPVectorTy();
296 };
297 if (isNonHVXFP(SrcTy) || isNonHVXFP(DstTy))
298 return InstructionCost::getMax();
299
300 if (SrcTy->isFPOrFPVectorTy() || DstTy->isFPOrFPVectorTy()) {
301 unsigned SrcN = SrcTy->isFPOrFPVectorTy() ? getTypeNumElements(Ty: SrcTy) : 0;
302 unsigned DstN = DstTy->isFPOrFPVectorTy() ? getTypeNumElements(Ty: DstTy) : 0;
303
304 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Ty: SrcTy);
305 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Ty: DstTy);
306 InstructionCost Cost =
307 std::max(a: SrcLT.first, b: DstLT.first) + FloatFactor * (SrcN + DstN);
308 // TODO: Allow non-throughput costs that aren't binary.
309 if (CostKind != TTI::TCK_RecipThroughput)
310 return Cost == 0 ? 0 : 1;
311 return Cost;
312 }
313 return 1;
314}
315
316InstructionCost HexagonTTIImpl::getVectorInstrCost(
317 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
318 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
319 Type *ElemTy = Val->isVectorTy() ? cast<VectorType>(Val)->getElementType()
320 : Val;
321 if (Opcode == Instruction::InsertElement) {
322 // Need two rotations for non-zero index.
323 unsigned Cost = (Index != 0) ? 2 : 0;
324 if (ElemTy->isIntegerTy(BitWidth: 32))
325 return Cost;
326 // If it's not a 32-bit value, there will need to be an extract.
327 return Cost + getVectorInstrCost(Opcode: Instruction::ExtractElement, Val, CostKind,
328 Index, Op0, Op1, VIC);
329 }
330
331 if (Opcode == Instruction::ExtractElement)
332 return 2;
333
334 return 1;
335}
336
337bool HexagonTTIImpl::shouldExpandReduction(const IntrinsicInst *II) const {
338 switch (II->getIntrinsicID()) {
339 case Intrinsic::vector_reduce_add:
340 return false;
341 }
342 return true;
343}
344
345bool HexagonTTIImpl::isLegalMaskedStore(Type *DataType, Align /*Alignment*/,
346 unsigned /*AddressSpace*/,
347 TTI::MaskKind /*MaskKind*/) const {
348 // This function is called from scalarize-masked-mem-intrin, which runs
349 // in pre-isel. Use ST directly instead of calling isHVXVectorType.
350 return HexagonMaskedVMem && ST.isTypeForHVX(VecTy: DataType);
351}
352
353bool HexagonTTIImpl::isLegalMaskedLoad(Type *DataType, Align /*Alignment*/,
354 unsigned /*AddressSpace*/,
355 TTI::MaskKind /*MaskKind*/) const {
356 // This function is called from scalarize-masked-mem-intrin, which runs
357 // in pre-isel. Use ST directly instead of calling isHVXVectorType.
358 return HexagonMaskedVMem && ST.isTypeForHVX(VecTy: DataType);
359}
360
361bool HexagonTTIImpl::isLegalMaskedGather(Type *Ty, Align Alignment) const {
362 // For now assume we can not deal with all HVX datatypes.
363 if (!Ty->isVectorTy() || !ST.isTypeForHVX(VecTy: Ty) ||
364 !HexagonAllowScatterGatherHVX)
365 return false;
366 // This must be in sync with HexagonVectorCombine pass.
367 switch (Ty->getScalarSizeInBits()) {
368 case 8:
369 return (getTypeNumElements(Ty) == 128);
370 case 16:
371 if (getTypeNumElements(Ty) == 64 || getTypeNumElements(Ty) == 32)
372 return (Alignment >= 2);
373 break;
374 case 32:
375 if (getTypeNumElements(Ty) == 32)
376 return (Alignment >= 4);
377 break;
378 default:
379 break;
380 }
381 return false;
382}
383
384bool HexagonTTIImpl::isLegalMaskedScatter(Type *Ty, Align Alignment) const {
385 if (!Ty->isVectorTy() || !ST.isTypeForHVX(VecTy: Ty) ||
386 !HexagonAllowScatterGatherHVX)
387 return false;
388 // This must be in sync with HexagonVectorCombine pass.
389 switch (Ty->getScalarSizeInBits()) {
390 case 8:
391 return (getTypeNumElements(Ty) == 128);
392 case 16:
393 if (getTypeNumElements(Ty) == 64)
394 return (Alignment >= 2);
395 break;
396 case 32:
397 if (getTypeNumElements(Ty) == 32)
398 return (Alignment >= 4);
399 break;
400 default:
401 break;
402 }
403 return false;
404}
405
406bool HexagonTTIImpl::forceScalarizeMaskedGather(VectorType *VTy,
407 Align Alignment) const {
408 return !isLegalMaskedGather(Ty: VTy, Alignment);
409}
410
411bool HexagonTTIImpl::forceScalarizeMaskedScatter(VectorType *VTy,
412 Align Alignment) const {
413 return !isLegalMaskedScatter(Ty: VTy, Alignment);
414}
415
416/// --- Vector TTI end ---
417
418unsigned HexagonTTIImpl::getPrefetchDistance() const {
419 return ST.getL1PrefetchDistance();
420}
421
422unsigned HexagonTTIImpl::getCacheLineSize() const {
423 return ST.getL1CacheLineSize();
424}
425
426InstructionCost
427HexagonTTIImpl::getInstructionCost(const User *U,
428 ArrayRef<const Value *> Operands,
429 TTI::TargetCostKind CostKind) const {
430 auto isCastFoldedIntoLoad = [this](const CastInst *CI) -> bool {
431 if (!CI->isIntegerCast())
432 return false;
433 // Only extensions from an integer type shorter than 32-bit to i32
434 // can be folded into the load.
435 const DataLayout &DL = getDataLayout();
436 unsigned SBW = DL.getTypeSizeInBits(Ty: CI->getSrcTy());
437 unsigned DBW = DL.getTypeSizeInBits(Ty: CI->getDestTy());
438 if (DBW != 32 || SBW >= DBW)
439 return false;
440
441 const LoadInst *LI = dyn_cast<const LoadInst>(Val: CI->getOperand(i_nocapture: 0));
442 // Technically, this code could allow multiple uses of the load, and
443 // check if all the uses are the same extension operation, but this
444 // should be sufficient for most cases.
445 return LI && LI->hasOneUse();
446 };
447
448 if (const CastInst *CI = dyn_cast<const CastInst>(Val: U))
449 if (isCastFoldedIntoLoad(CI))
450 return TargetTransformInfo::TCC_Free;
451 return BaseT::getInstructionCost(U, Operands, CostKind);
452}
453
454bool HexagonTTIImpl::shouldBuildLookupTables() const {
455 return EmitLookupTables;
456}
457
458bool HexagonTTIImpl::areInlineCompatible(const Function *Caller,
459 const Function *Callee) const {
460 // HVX contexts are a fixed hardware resource, held by threads dedicated to
461 // HVX. HVX reaching a thread without one stalls, and deadlocks if the two
462 // then meet at a barrier.
463 if (Caller->hasFnAttribute(Kind: "hexagon_hmx") !=
464 Callee->hasFnAttribute(Kind: "hexagon_hmx"))
465 return false;
466 if (Callee->hasFnAttribute(Kind: "hexagon_hvx") &&
467 !Caller->hasFnAttribute(Kind: "hexagon_hvx"))
468 return false;
469 return BaseT::areInlineCompatible(Caller, Callee);
470}
471