1//===-- X86TargetTransformInfo.cpp - X86 specific TTI pass ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements a TargetTransformInfo analysis pass specific to the
10/// X86 target machine. It uses the target's detailed information to provide
11/// more precise answers to certain TTI queries, while letting the target
12/// independent and default TTI implementations handle the rest.
13///
14//===----------------------------------------------------------------------===//
15/// About Cost Model numbers used below it's necessary to say the following:
16/// the numbers correspond to some "generic" X86 CPU instead of usage of a
17/// specific CPU model. Usually the numbers correspond to the CPU where the
18/// feature first appeared. For example, if we do Subtarget.hasSSE42() in
19/// the lookups below the cost is based on Nehalem as that was the first CPU
20/// to support that feature level and thus has most likely the worst case cost,
21/// although we may discard an outlying worst cost from one CPU (e.g. Atom).
22///
23/// Some examples of other technologies/CPUs:
24/// SSE 3 - Pentium4 / Athlon64
25/// SSE 4.1 - Penryn
26/// SSE 4.2 - Nehalem / Silvermont
27/// AVX - Sandy Bridge / Jaguar / Bulldozer
28/// AVX2 - Haswell / Ryzen
29/// AVX-512 - Xeon Phi / Skylake
30///
31/// And some examples of instruction target dependent costs (latency)
32/// divss sqrtss rsqrtss
33/// AMD K7 11-16 19 3
34/// Piledriver 9-24 13-15 5
35/// Jaguar 14 16 2
36/// Pentium II,III 18 30 2
37/// Nehalem 7-14 7-18 3
38/// Haswell 10-13 11 5
39///
40/// Interpreting the 4 TargetCostKind types:
41/// TCK_RecipThroughput and TCK_Latency should try to match the worst case
42/// values reported by the CPU scheduler models (and llvm-mca).
43/// TCK_CodeSize should match the instruction count (e.g. divss = 1), NOT the
44/// actual encoding size of the instruction.
45/// TCK_SizeAndLatency should match the worst case micro-op counts reported by
46/// by the CPU scheduler models (and llvm-mca), to ensure that they are
47/// compatible with the MicroOpBufferSize and LoopMicroOpBufferSize values which are
48/// often used as the cost thresholds where TCK_SizeAndLatency is requested.
49//===----------------------------------------------------------------------===//
50
51#include "X86TargetTransformInfo.h"
52#include "llvm/ADT/SmallBitVector.h"
53#include "llvm/Analysis/TargetTransformInfo.h"
54#include "llvm/CodeGen/Analysis.h"
55#include "llvm/CodeGen/BasicTTIImpl.h"
56#include "llvm/CodeGen/CostTable.h"
57#include "llvm/CodeGen/TargetLowering.h"
58#include "llvm/IR/InstIterator.h"
59#include "llvm/IR/IntrinsicInst.h"
60#include <optional>
61
62using namespace llvm;
63
64#define DEBUG_TYPE "x86tti"
65
66//===----------------------------------------------------------------------===//
67//
68// X86 cost model.
69//
70//===----------------------------------------------------------------------===//
71
72// Helper struct to store/access costs for each cost kind.
73// TODO: Move this to allow other targets to use it?
74struct CostKindCosts {
75 unsigned RecipThroughputCost = ~0U;
76 unsigned LatencyCost = ~0U;
77 unsigned CodeSizeCost = ~0U;
78 unsigned SizeAndLatencyCost = ~0U;
79
80 std::optional<unsigned>
81 operator[](TargetTransformInfo::TargetCostKind Kind) const {
82 unsigned Cost = ~0U;
83 switch (Kind) {
84 case TargetTransformInfo::TCK_RecipThroughput:
85 Cost = RecipThroughputCost;
86 break;
87 case TargetTransformInfo::TCK_Latency:
88 Cost = LatencyCost;
89 break;
90 case TargetTransformInfo::TCK_CodeSize:
91 Cost = CodeSizeCost;
92 break;
93 case TargetTransformInfo::TCK_SizeAndLatency:
94 Cost = SizeAndLatencyCost;
95 break;
96 }
97 if (Cost == ~0U)
98 return std::nullopt;
99 return Cost;
100 }
101};
102using CostKindTblEntry = CostTblEntryT<CostKindCosts>;
103using TypeConversionCostKindTblEntry = TypeConversionCostTblEntryT<CostKindCosts>;
104
105TargetTransformInfo::PopcntSupportKind
106X86TTIImpl::getPopcntSupport(unsigned TyWidth) const {
107 assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2");
108 // TODO: Currently the __builtin_popcount() implementation using SSE3
109 // instructions is inefficient. Once the problem is fixed, we should
110 // call ST->hasSSE3() instead of ST->hasPOPCNT().
111 return ST->hasPOPCNT() ? TTI::PSK_FastHardware : TTI::PSK_Software;
112}
113
114std::optional<unsigned> X86TTIImpl::getCacheSize(
115 TargetTransformInfo::CacheLevel Level) const {
116 switch (Level) {
117 case TargetTransformInfo::CacheLevel::L1D:
118 // - Penryn
119 // - Nehalem
120 // - Westmere
121 // - Sandy Bridge
122 // - Ivy Bridge
123 // - Haswell
124 // - Broadwell
125 // - Skylake
126 // - Kabylake
127 return 32 * 1024; // 32 KiB
128 case TargetTransformInfo::CacheLevel::L2D:
129 // - Penryn
130 // - Nehalem
131 // - Westmere
132 // - Sandy Bridge
133 // - Ivy Bridge
134 // - Haswell
135 // - Broadwell
136 // - Skylake
137 // - Kabylake
138 return 256 * 1024; // 256 KiB
139 }
140
141 llvm_unreachable("Unknown TargetTransformInfo::CacheLevel");
142}
143
144std::optional<unsigned> X86TTIImpl::getCacheAssociativity(
145 TargetTransformInfo::CacheLevel Level) const {
146 // - Penryn
147 // - Nehalem
148 // - Westmere
149 // - Sandy Bridge
150 // - Ivy Bridge
151 // - Haswell
152 // - Broadwell
153 // - Skylake
154 // - Kabylake
155 switch (Level) {
156 case TargetTransformInfo::CacheLevel::L1D:
157 [[fallthrough]];
158 case TargetTransformInfo::CacheLevel::L2D:
159 return 8;
160 }
161
162 llvm_unreachable("Unknown TargetTransformInfo::CacheLevel");
163}
164
165enum ClassIDEnum { GPRClass = 0, VectorClass = 1, ScalarFPClass = 2 };
166
167unsigned X86TTIImpl::getRegisterClassForType(bool Vector, Type *Ty) const {
168 return Vector ? VectorClass
169 : Ty && Ty->isFloatingPointTy() ? ScalarFPClass
170 : GPRClass;
171}
172
173unsigned X86TTIImpl::getNumberOfRegisters(unsigned ClassID) const {
174 if (ClassID == VectorClass && !ST->hasSSE1())
175 return 0;
176
177 if (!ST->is64Bit())
178 return 8;
179
180 if ((ClassID == GPRClass && ST->hasEGPR()) ||
181 (ClassID != GPRClass && ST->hasAVX512()))
182 return 32;
183
184 return 16;
185}
186
187bool X86TTIImpl::hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const {
188 if (!ST->hasCF())
189 return false;
190 if (!Ty)
191 return true;
192 // Conditional faulting is supported by CFCMOV, which only accepts
193 // 16/32/64-bit operands.
194 // TODO: Support f32/f64 with VMOVSS/VMOVSD with zero mask when it's
195 // profitable.
196 auto *VTy = dyn_cast<FixedVectorType>(Val: Ty);
197 if (!Ty->isIntegerTy() && (!VTy || VTy->getNumElements() != 1))
198 return false;
199 auto *ScalarTy = Ty->getScalarType();
200 switch (cast<IntegerType>(Val: ScalarTy)->getBitWidth()) {
201 default:
202 return false;
203 case 16:
204 case 32:
205 case 64:
206 return true;
207 }
208}
209
210TypeSize
211X86TTIImpl::getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const {
212 unsigned PreferVectorWidth = ST->getPreferVectorWidth();
213 switch (K) {
214 case TargetTransformInfo::RGK_Scalar:
215 return TypeSize::getFixed(ExactSize: ST->is64Bit() ? 64 : 32);
216 case TargetTransformInfo::RGK_FixedWidthVector:
217 if (ST->hasAVX512() && PreferVectorWidth >= 512)
218 return TypeSize::getFixed(ExactSize: 512);
219 if (ST->hasAVX() && PreferVectorWidth >= 256)
220 return TypeSize::getFixed(ExactSize: 256);
221 if (ST->hasSSE1() && PreferVectorWidth >= 128)
222 return TypeSize::getFixed(ExactSize: 128);
223 return TypeSize::getFixed(ExactSize: 0);
224 case TargetTransformInfo::RGK_ScalableVector:
225 return TypeSize::getScalable(MinimumSize: 0);
226 }
227
228 llvm_unreachable("Unsupported register kind");
229}
230
231unsigned X86TTIImpl::getLoadStoreVecRegBitWidth(unsigned) const {
232 return getRegisterBitWidth(K: TargetTransformInfo::RGK_FixedWidthVector)
233 .getFixedValue();
234}
235
236unsigned X86TTIImpl::getMaxInterleaveFactor(ElementCount VF,
237 bool HasUnorderedReductions) const {
238 // If the loop will not be vectorized, don't interleave the loop.
239 // Let regular unroll to unroll the loop, which saves the overflow
240 // check and memory check cost.
241 if (VF.isScalar())
242 return 1;
243
244 if (ST->isAtom())
245 return 1;
246
247 // Sandybridge and Haswell have multiple execution ports and pipelined
248 // vector units.
249 if (ST->hasAVX())
250 return 4;
251
252 return 2;
253}
254
255InstructionCost X86TTIImpl::getArithmeticInstrCost(
256 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
257 TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
258 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
259
260 // vXi8 multiplications are always promoted to vXi16.
261 // Sub-128-bit types can be extended/packed more efficiently.
262 if (Opcode == Instruction::Mul && Ty->isVectorTy() &&
263 Ty->getPrimitiveSizeInBits() <= 64 && Ty->getScalarSizeInBits() == 8) {
264 Type *WideVecTy =
265 VectorType::getExtendedElementVectorType(VTy: cast<VectorType>(Val: Ty));
266 return getCastInstrCost(Opcode: Instruction::ZExt, Dst: WideVecTy, Src: Ty,
267 CCH: TargetTransformInfo::CastContextHint::None,
268 CostKind) +
269 getCastInstrCost(Opcode: Instruction::Trunc, Dst: Ty, Src: WideVecTy,
270 CCH: TargetTransformInfo::CastContextHint::None,
271 CostKind) +
272 getArithmeticInstrCost(Opcode, Ty: WideVecTy, CostKind, Op1Info, Op2Info);
273 }
274
275 // Legalize the type.
276 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
277
278 int ISD = TLI->InstructionOpcodeToISD(Opcode);
279 assert(ISD && "Invalid opcode");
280
281 if (ISD == ISD::MUL && Args.size() == 2 && LT.second.isVector() &&
282 (LT.second.getScalarType() == MVT::i32 ||
283 LT.second.getScalarType() == MVT::i64)) {
284 // Check if the operands can be represented as a smaller datatype.
285 bool Op1Signed = false, Op2Signed = false;
286 unsigned Op1MinSize = BaseT::minRequiredElementSize(Val: Args[0], isSigned&: Op1Signed);
287 unsigned Op2MinSize = BaseT::minRequiredElementSize(Val: Args[1], isSigned&: Op2Signed);
288 unsigned OpMinSize = std::max(a: Op1MinSize, b: Op2MinSize);
289 bool SignedMode = Op1Signed || Op2Signed;
290
291 // If both vXi32 are representable as i15 and at least one is constant,
292 // zero-extended, or sign-extended from vXi16 (or less pre-SSE41) then we
293 // can treat this as PMADDWD which has the same costs as a vXi16 multiply.
294 if (OpMinSize <= 15 && !ST->isPMADDWDSlow() &&
295 LT.second.getScalarType() == MVT::i32) {
296 bool Op1Constant =
297 isa<ConstantDataVector>(Val: Args[0]) || isa<ConstantVector>(Val: Args[0]);
298 bool Op2Constant =
299 isa<ConstantDataVector>(Val: Args[1]) || isa<ConstantVector>(Val: Args[1]);
300 bool Op1Sext = isa<SExtInst>(Val: Args[0]) &&
301 (Op1MinSize == 15 || (Op1MinSize < 15 && !ST->hasSSE41()));
302 bool Op2Sext = isa<SExtInst>(Val: Args[1]) &&
303 (Op2MinSize == 15 || (Op2MinSize < 15 && !ST->hasSSE41()));
304
305 bool IsZeroExtended = !Op1Signed || !Op2Signed;
306 bool IsConstant = Op1Constant || Op2Constant;
307 bool IsSext = Op1Sext || Op2Sext;
308 if (IsConstant || IsZeroExtended || IsSext)
309 LT.second =
310 MVT::getVectorVT(VT: MVT::i16, NumElements: 2 * LT.second.getVectorNumElements());
311 }
312
313 // Check if the vXi32 operands can be shrunk into a smaller datatype.
314 // This should match the codegen from reduceVMULWidth.
315 // TODO: Make this generic (!ST->SSE41 || ST->isPMULLDSlow()).
316 if (ST->useSLMArithCosts() && LT.second == MVT::v4i32) {
317 if (OpMinSize <= 7)
318 return LT.first * 3; // pmullw/sext
319 if (!SignedMode && OpMinSize <= 8)
320 return LT.first * 3; // pmullw/zext
321 if (OpMinSize <= 15)
322 return LT.first * 5; // pmullw/pmulhw/pshuf
323 if (!SignedMode && OpMinSize <= 16)
324 return LT.first * 5; // pmullw/pmulhw/pshuf
325 }
326
327 // If both vXi64 are representable as (unsigned) i32, then we can perform
328 // the multiple with a single PMULUDQ instruction.
329 // TODO: Add (SSE41+) PMULDQ handling for signed extensions.
330 if (!SignedMode && OpMinSize <= 32 && LT.second.getScalarType() == MVT::i64)
331 ISD = X86ISD::PMULUDQ;
332 }
333
334 // Vector multiply by pow2 will be simplified to shifts.
335 // Vector multiply by -pow2 will be simplified to shifts/negates.
336 if (ISD == ISD::MUL && Op2Info.isConstant() &&
337 (Op2Info.isPowerOf2() || Op2Info.isNegatedPowerOf2())) {
338 InstructionCost Cost =
339 getArithmeticInstrCost(Opcode: Instruction::Shl, Ty, CostKind,
340 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
341 if (Op2Info.isNegatedPowerOf2())
342 Cost += getArithmeticInstrCost(Opcode: Instruction::Sub, Ty, CostKind);
343 return Cost;
344 }
345
346 // On X86, vector signed division by constants power-of-two are
347 // normally expanded to the sequence SRA + SRL + ADD + SRA.
348 // The OperandValue properties may not be the same as that of the previous
349 // operation; conservatively assume OP_None.
350 if ((ISD == ISD::SDIV || ISD == ISD::SREM) &&
351 Op2Info.isConstant() && Op2Info.isPowerOf2()) {
352 InstructionCost Cost =
353 2 * getArithmeticInstrCost(Opcode: Instruction::AShr, Ty, CostKind,
354 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
355 Cost += getArithmeticInstrCost(Opcode: Instruction::LShr, Ty, CostKind,
356 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
357 Cost += getArithmeticInstrCost(Opcode: Instruction::Add, Ty, CostKind,
358 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
359
360 if (ISD == ISD::SREM) {
361 // For SREM: (X % C) is the equivalent of (X - (X/C)*C)
362 Cost += getArithmeticInstrCost(Opcode: Instruction::Mul, Ty, CostKind, Op1Info: Op1Info.getNoProps(),
363 Op2Info: Op2Info.getNoProps());
364 Cost += getArithmeticInstrCost(Opcode: Instruction::Sub, Ty, CostKind, Op1Info: Op1Info.getNoProps(),
365 Op2Info: Op2Info.getNoProps());
366 }
367
368 return Cost;
369 }
370
371 // Vector unsigned division/remainder will be simplified to shifts/masks.
372 if ((ISD == ISD::UDIV || ISD == ISD::UREM) &&
373 Op2Info.isConstant() && Op2Info.isPowerOf2()) {
374 if (ISD == ISD::UDIV)
375 return getArithmeticInstrCost(Opcode: Instruction::LShr, Ty, CostKind,
376 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
377 // UREM
378 return getArithmeticInstrCost(Opcode: Instruction::And, Ty, CostKind,
379 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
380 }
381
382 // A scalar integer divide/remainder by a constant is not a hardware divide;
383 // it lowers to a magic-number multiply-high plus a few fixup ops. Cost it as
384 // that sequence rather than the generic single-instruction divide, so the
385 // vectorizers do not compare against an artificially cheap scalar lane. The
386 // power-of-two cases are handled above; negated powers of two are left to the
387 // generic handling.
388 if (!Ty->isVectorTy() && Op2Info.isConstant() && !Op2Info.isNegatedPowerOf2() &&
389 (ISD == ISD::UDIV || ISD == ISD::SDIV || ISD == ISD::UREM ||
390 ISD == ISD::SREM)) {
391 unsigned Cost = ISD == ISD::UREM || ISD == ISD::SREM ? 6 : 5;
392 if (CostKind == TTI::TCK_CodeSize || CostKind == TTI::TCK_SizeAndLatency)
393 Cost += 2;
394 return LT.first * Cost;
395 }
396
397 static const CostKindTblEntry GFNIUniformConstCostTable[] = {
398 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
399 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
400 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
401 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
402 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
403 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
404 { .ISD: ISD::SHL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
405 { .ISD: ISD::SRL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
406 { .ISD: ISD::SRA, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
407 };
408
409 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasGFNI())
410 if (const auto *Entry =
411 CostTableLookup(Table: GFNIUniformConstCostTable, ISD, Ty: LT.second))
412 if (auto KindCost = Entry->Cost[CostKind])
413 return LT.first * *KindCost;
414
415 static const CostKindTblEntry AVX512BWUniformConstCostTable[] = {
416 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psllw + pand.
417 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrlw + pand.
418 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // psrlw, pand, pxor, psubb.
419 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psllw + pand.
420 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrlw + pand.
421 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // psrlw, pand, pxor, psubb.
422 { .ISD: ISD::SHL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psllw + pand.
423 { .ISD: ISD::SRL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrlw + pand.
424 { .ISD: ISD::SRA, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } }, // psrlw, pand, pxor, psubb.
425
426 { .ISD: ISD::SHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllw
427 { .ISD: ISD::SRL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlw
428 { .ISD: ISD::SRA, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlw
429 { .ISD: ISD::SHL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllw
430 { .ISD: ISD::SRL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlw
431 { .ISD: ISD::SRA, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlw
432 };
433
434 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasBWI())
435 if (const auto *Entry =
436 CostTableLookup(Table: AVX512BWUniformConstCostTable, ISD, Ty: LT.second))
437 if (auto KindCost = Entry->Cost[CostKind])
438 return LT.first * *KindCost;
439
440 static const CostKindTblEntry AVX512DQUniformConstCostTable[] = {
441 { .ISD: ISD::SDIV, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 15 } }, // vpmullq-based MULHS sequence
442 { .ISD: ISD::SREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 17 } }, // vpmullq-based MULHS+mul+sub sequence
443 { .ISD: ISD::SDIV, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 15 } }, // vpmullq-based MULHS sequence
444 { .ISD: ISD::SREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 17 } }, // vpmullq-based MULHS+mul+sub sequence
445 // The remainder's multiply-back is a single vpmullq with DQ, just like the
446 // pmulld the vXi32 entries above rely on. Without DQ it is another
447 // vpmuludq schoolbook, so the AVX512/AVX2 tables charge more.
448 { .ISD: ISD::UREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 17 } }, // MULHU + vpmullq + sub sequence
449 { .ISD: ISD::UREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 17 } }, // MULHU + vpmullq + sub sequence
450 };
451
452 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasDQI())
453 if (const auto *Entry =
454 CostTableLookup(Table: AVX512DQUniformConstCostTable, ISD, Ty: LT.second))
455 if (auto KindCost = Entry->Cost[CostKind])
456 return LT.first * *KindCost;
457
458 static const CostKindTblEntry AVX512UniformConstCostTable[] = {
459 { .ISD: ISD::SHL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 12, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psllw + pand.
460 { .ISD: ISD::SRL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 12, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psrlw + pand.
461 { .ISD: ISD::SRA, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 12, .SizeAndLatencyCost: 12 } }, // psrlw, pand, pxor, psubb.
462
463 { .ISD: ISD::SHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // psllw + split.
464 { .ISD: ISD::SRL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // psrlw + split.
465 { .ISD: ISD::SRA, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // psraw + split.
466
467 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pslld
468 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrld
469 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrad
470 { .ISD: ISD::SHL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pslld
471 { .ISD: ISD::SRL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrld
472 { .ISD: ISD::SRA, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrad
473
474 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psraq
475 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllq
476 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlq
477 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psraq
478 { .ISD: ISD::SHL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllq
479 { .ISD: ISD::SRL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlq
480 { .ISD: ISD::SRA, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psraq
481
482 { .ISD: ISD::SDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 6 } }, // pmuludq sequence
483 { .ISD: ISD::SREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 8 } }, // pmuludq+mul+sub sequence
484 { .ISD: ISD::UDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 5 } }, // pmuludq sequence
485 { .ISD: ISD::UREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 7 } }, // pmuludq+mul+sub sequence
486
487 { .ISD: ISD::UDIV, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 15 } }, // pmuludq-based MULHU sequence
488 { .ISD: ISD::UREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 21 } }, // pmuludq-based MULHU+mul+sub sequence
489 };
490
491 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasAVX512())
492 if (const auto *Entry =
493 CostTableLookup(Table: AVX512UniformConstCostTable, ISD, Ty: LT.second))
494 if (auto KindCost = Entry->Cost[CostKind])
495 return LT.first * *KindCost;
496
497 static const CostKindTblEntry AVX2UniformConstCostTable[] = {
498 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psllw + pand.
499 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrlw + pand.
500 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 10, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psrlw, pand, pxor, psubb.
501 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // psllw + pand.
502 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // psrlw + pand.
503 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 5, .SizeAndLatencyCost: 9 } }, // psrlw, pand, pxor, psubb.
504
505 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllw
506 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlw
507 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psraw
508 { .ISD: ISD::SHL, .Type: MVT::v16i16,.Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllw
509 { .ISD: ISD::SRL, .Type: MVT::v16i16,.Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlw
510 { .ISD: ISD::SRA, .Type: MVT::v16i16,.Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psraw
511
512 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pslld
513 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrld
514 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrad
515 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pslld
516 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrld
517 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrad
518
519 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllq
520 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlq
521 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // psrad + shuffle.
522 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllq
523 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlq
524 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 3, .SizeAndLatencyCost: 6 } }, // psrad + shuffle + split.
525
526 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 6 } }, // pmuludq sequence
527 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 8 } }, // pmuludq+mul+sub sequence
528 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 5 } }, // pmuludq sequence
529 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 7 } }, // pmuludq+mul+sub sequence
530
531 { .ISD: ISD::UDIV, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 15 } }, // pmuludq-based MULHU sequence
532 { .ISD: ISD::UREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 21 } }, // pmuludq-based MULHU+mul+sub sequence
533 };
534
535 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasAVX2())
536 if (const auto *Entry =
537 CostTableLookup(Table: AVX2UniformConstCostTable, ISD, Ty: LT.second))
538 if (auto KindCost = Entry->Cost[CostKind])
539 return LT.first * *KindCost;
540
541 static const CostKindTblEntry AVXUniformConstCostTable[] = {
542 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psllw + pand.
543 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrlw + pand.
544 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psrlw, pand, pxor, psubb.
545 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 8 } }, // 2*(psllw + pand) + split.
546 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 8 } }, // 2*(psrlw + pand) + split.
547 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 12, .SizeAndLatencyCost: 13 } }, // 2*(psrlw, pand, pxor, psubb) + split.
548
549 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllw.
550 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlw.
551 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psraw.
552 { .ISD: ISD::SHL, .Type: MVT::v16i16,.Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // psllw + split.
553 { .ISD: ISD::SRL, .Type: MVT::v16i16,.Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // psrlw + split.
554 { .ISD: ISD::SRA, .Type: MVT::v16i16,.Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // psraw + split.
555
556 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pslld.
557 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrld.
558 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrad.
559 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // pslld + split.
560 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // psrld + split.
561 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // psrad + split.
562
563 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllq.
564 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlq.
565 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // psrad + shuffle.
566 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // 2 x psllq + split.
567 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // 2 x psllq + split.
568 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 7, .CodeSizeCost: 8, .SizeAndLatencyCost: 9 } }, // 2 x psrad + shuffle + split.
569
570 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } }, // 2*pmuludq sequence + split.
571 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 18 } }, // 2*pmuludq+mul+sub sequence + split.
572 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 12 } }, // 2*pmuludq sequence + split.
573 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 16 } }, // 2*pmuludq+mul+sub sequence + split.
574 };
575
576 // XOP has faster vXi8 shifts.
577 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasAVX() &&
578 (!ST->hasXOP() || LT.second.getScalarSizeInBits() != 8))
579 if (const auto *Entry =
580 CostTableLookup(Table: AVXUniformConstCostTable, ISD, Ty: LT.second))
581 if (auto KindCost = Entry->Cost[CostKind])
582 return LT.first * *KindCost;
583
584 static const CostKindTblEntry SSE2UniformConstCostTable[] = {
585 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psllw + pand.
586 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrlw + pand.
587 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psrlw, pand, pxor, psubb.
588
589 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllw.
590 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlw.
591 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psraw.
592
593 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pslld
594 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrld.
595 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrad.
596
597 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psllq.
598 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psrlq.
599 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } }, // 2 x psrad + shuffle.
600
601 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 6 } }, // pmuludq sequence
602 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8 } }, // pmuludq+mul+sub sequence
603 { .ISD: ISD::UDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 5 } }, // pmuludq sequence
604 { .ISD: ISD::UREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7 } }, // pmuludq+mul+sub sequence
605 };
606
607 // XOP has faster vXi8 shifts.
608 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasSSE2() &&
609 (!ST->hasXOP() || LT.second.getScalarSizeInBits() != 8))
610 if (const auto *Entry =
611 CostTableLookup(Table: SSE2UniformConstCostTable, ISD, Ty: LT.second))
612 if (auto KindCost = Entry->Cost[CostKind])
613 return LT.first * *KindCost;
614
615 static const CostKindTblEntry AVX512BWConstCostTable[] = {
616 { .ISD: ISD::SDIV, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 14 } }, // 2*ext+2*pmulhw sequence
617 { .ISD: ISD::SREM, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
618 { .ISD: ISD::UDIV, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 14 } }, // 2*ext+2*pmulhw sequence
619 { .ISD: ISD::UREM, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
620
621 { .ISD: ISD::SDIV, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 6 } }, // vpmulhw sequence
622 { .ISD: ISD::SREM, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 8 } }, // vpmulhw+mul+sub sequence
623 { .ISD: ISD::UDIV, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 6 } }, // vpmulhuw sequence
624 { .ISD: ISD::UREM, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 8 } }, // vpmulhuw+mul+sub sequence
625 };
626
627 if (Op2Info.isConstant() && ST->hasBWI())
628 if (const auto *Entry =
629 CostTableLookup(Table: AVX512BWConstCostTable, ISD, Ty: LT.second))
630 if (auto KindCost = Entry->Cost[CostKind])
631 return LT.first * *KindCost;
632
633 static const CostKindTblEntry AVX512DQConstCostTable[] = {
634 { .ISD: ISD::SDIV, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 19 } }, // vpmullq-based MULHS sequence
635 { .ISD: ISD::SREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 21 } }, // vpmullq-based MULHS+mul+sub sequence
636 { .ISD: ISD::SDIV, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 19 } }, // vpmullq-based MULHS sequence
637 { .ISD: ISD::SREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 21 } }, // vpmullq-based MULHS+mul+sub sequence
638 // The remainder's multiply-back is a single vpmullq with DQ, whereas the
639 // AVX512/AVX2 tables have to charge for another vpmuludq schoolbook.
640 { .ISD: ISD::UREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 24 } }, // MULHU + vpmullq + sub sequence
641 { .ISD: ISD::UREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 24 } }, // MULHU + vpmullq + sub sequence
642 };
643
644 if (Op2Info.isConstant() && ST->hasDQI())
645 if (const auto *Entry =
646 CostTableLookup(Table: AVX512DQConstCostTable, ISD, Ty: LT.second))
647 if (auto KindCost = Entry->Cost[CostKind])
648 return LT.first * *KindCost;
649
650 static const CostKindTblEntry AVX512ConstCostTable[] = {
651 { .ISD: ISD::SDIV, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 28 } }, // 4*ext+4*pmulhw sequence
652 { .ISD: ISD::SREM, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 32 } }, // 4*ext+4*pmulhw+mul+sub sequence
653 { .ISD: ISD::UDIV, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 28 } }, // 4*ext+4*pmulhw sequence
654 { .ISD: ISD::UREM, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 32 } }, // 4*ext+4*pmulhw+mul+sub sequence
655
656 { .ISD: ISD::SDIV, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 12 } }, // 2*vpmulhw sequence
657 { .ISD: ISD::SREM, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 16 } }, // 2*vpmulhw+mul+sub sequence
658 { .ISD: ISD::UDIV, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 12 } }, // 2*vpmulhuw sequence
659 { .ISD: ISD::UREM, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 16 } }, // 2*vpmulhuw+mul+sub sequence
660
661 { .ISD: ISD::SDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 15 } }, // vpmuldq sequence
662 { .ISD: ISD::SREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 17 } }, // vpmuldq+mul+sub sequence
663 { .ISD: ISD::UDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 15 } }, // vpmuludq sequence
664 { .ISD: ISD::UREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 17 } }, // vpmuludq+mul+sub sequence
665
666 { .ISD: ISD::UDIV, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 22 } }, // vpmuludq-based MULHU sequence
667 { .ISD: ISD::UREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 28 } }, // vpmuludq-based MULHU+mul+sub sequence
668 };
669
670 if (Op2Info.isConstant() && ST->hasAVX512())
671 if (const auto *Entry =
672 CostTableLookup(Table: AVX512ConstCostTable, ISD, Ty: LT.second))
673 if (auto KindCost = Entry->Cost[CostKind])
674 return LT.first * *KindCost;
675
676 static const CostKindTblEntry AVX2ConstCostTable[] = {
677 { .ISD: ISD::SDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 14 } }, // 2*ext+2*pmulhw sequence
678 { .ISD: ISD::SREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
679 { .ISD: ISD::UDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 14 } }, // 2*ext+2*pmulhw sequence
680 { .ISD: ISD::UREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
681
682 { .ISD: ISD::SDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 6 } }, // vpmulhw sequence
683 { .ISD: ISD::SREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 8 } }, // vpmulhw+mul+sub sequence
684 { .ISD: ISD::UDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 6 } }, // vpmulhuw sequence
685 { .ISD: ISD::UREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 8 } }, // vpmulhuw+mul+sub sequence
686
687 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 15 } }, // vpmuldq sequence
688 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 19 } }, // vpmuldq+mul+sub sequence
689 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 15 } }, // vpmuludq sequence
690 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 19 } }, // vpmuludq+mul+sub sequence
691
692 { .ISD: ISD::UDIV, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 22 } }, // vpmuludq-based MULHU sequence
693 { .ISD: ISD::UREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 28 } }, // vpmuludq-based MULHU+mul+sub sequence
694 };
695
696 if (Op2Info.isConstant() && ST->hasAVX2())
697 if (const auto *Entry = CostTableLookup(Table: AVX2ConstCostTable, ISD, Ty: LT.second))
698 if (auto KindCost = Entry->Cost[CostKind])
699 return LT.first * *KindCost;
700
701 static const CostKindTblEntry AVXConstCostTable[] = {
702 { .ISD: ISD::SDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 30 } }, // 4*ext+4*pmulhw sequence + split.
703 { .ISD: ISD::SREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 34 } }, // 4*ext+4*pmulhw+mul+sub sequence + split.
704 { .ISD: ISD::UDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 30 } }, // 4*ext+4*pmulhw sequence + split.
705 { .ISD: ISD::UREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 34 } }, // 4*ext+4*pmulhw+mul+sub sequence + split.
706
707 { .ISD: ISD::SDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 14 } }, // 2*pmulhw sequence + split.
708 { .ISD: ISD::SREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 18 } }, // 2*pmulhw+mul+sub sequence + split.
709 { .ISD: ISD::UDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 14 } }, // 2*pmulhuw sequence + split.
710 { .ISD: ISD::UREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 18 } }, // 2*pmulhuw+mul+sub sequence + split.
711
712 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 32 } }, // vpmuludq sequence
713 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 38 } }, // vpmuludq+mul+sub sequence
714 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 32 } }, // 2*pmuludq sequence + split.
715 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 42 } }, // 2*pmuludq+mul+sub sequence + split.
716 };
717
718 if (Op2Info.isConstant() && ST->hasAVX())
719 if (const auto *Entry = CostTableLookup(Table: AVXConstCostTable, ISD, Ty: LT.second))
720 if (auto KindCost = Entry->Cost[CostKind])
721 return LT.first * *KindCost;
722
723 static const CostKindTblEntry SSE41ConstCostTable[] = {
724 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 15 } }, // vpmuludq sequence
725 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 20 } }, // vpmuludq+mul+sub sequence
726 };
727
728 if (Op2Info.isConstant() && ST->hasSSE41())
729 if (const auto *Entry =
730 CostTableLookup(Table: SSE41ConstCostTable, ISD, Ty: LT.second))
731 if (auto KindCost = Entry->Cost[CostKind])
732 return LT.first * *KindCost;
733
734 static const CostKindTblEntry SSE2ConstCostTable[] = {
735 { .ISD: ISD::SDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 14 } }, // 2*ext+2*pmulhw sequence
736 { .ISD: ISD::SREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
737 { .ISD: ISD::UDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 14 } }, // 2*ext+2*pmulhw sequence
738 { .ISD: ISD::UREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
739
740 { .ISD: ISD::SDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 6 } }, // pmulhw sequence
741 { .ISD: ISD::SREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 8 } }, // pmulhw+mul+sub sequence
742 { .ISD: ISD::UDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 6 } }, // pmulhuw sequence
743 { .ISD: ISD::UREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 8 } }, // pmulhuw+mul+sub sequence
744
745 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 19 } }, // pmuludq sequence
746 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 24 } }, // pmuludq+mul+sub sequence
747 { .ISD: ISD::UDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 15 } }, // pmuludq sequence
748 { .ISD: ISD::UREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 20 } }, // pmuludq+mul+sub sequence
749 };
750
751 if (Op2Info.isConstant() && ST->hasSSE2())
752 if (const auto *Entry = CostTableLookup(Table: SSE2ConstCostTable, ISD, Ty: LT.second))
753 if (auto KindCost = Entry->Cost[CostKind])
754 return LT.first * *KindCost;
755
756 // rem matches div because the divider returns the remainder for free.
757 static const CostKindTblEntry ScalarVarDivCostTable[] = {
758 { .ISD: ISD::SDIV, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 20, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
759 { .ISD: ISD::UDIV, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 20, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
760 { .ISD: ISD::SREM, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 20, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
761 { .ISD: ISD::UREM, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 20, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
762 { .ISD: ISD::SDIV, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 20, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
763 { .ISD: ISD::UDIV, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 20, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
764 { .ISD: ISD::SREM, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 20, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
765 { .ISD: ISD::UREM, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 20, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
766 { .ISD: ISD::SDIV, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 25, .LatencyCost: 22, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
767 { .ISD: ISD::UDIV, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 25, .LatencyCost: 22, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
768 { .ISD: ISD::SREM, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 25, .LatencyCost: 22, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
769 { .ISD: ISD::UREM, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 25, .LatencyCost: 22, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
770 { .ISD: ISD::SDIV, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 41, .LatencyCost: 24, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
771 { .ISD: ISD::UDIV, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 41, .LatencyCost: 24, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
772 { .ISD: ISD::SREM, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 41, .LatencyCost: 24, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
773 { .ISD: ISD::UREM, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 41, .LatencyCost: 24, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
774 };
775
776 if (!LT.second.isVector() && !Op2Info.isConstant())
777 if (const auto *Entry =
778 CostTableLookup(Table: ScalarVarDivCostTable, ISD, Ty: LT.second))
779 if (auto KindCost = Entry->Cost[CostKind])
780 return LT.first * *KindCost;
781
782 // Variable divisors lower through a float divide. strictfp needs SAE
783 // rounding which is 512-bit only.
784 bool IsStrictFP =
785 CtxI && CtxI->getFunction()->hasFnAttribute(Kind: Attribute::StrictFP);
786 bool IsDivRem = ISD == ISD::UDIV || ISD == ISD::SDIV || ISD == ISD::UREM ||
787 ISD == ISD::SREM;
788 bool VarDivToFP = IsDivRem && !Op2Info.isConstant() &&
789 (!IsStrictFP || ST->useAVX512Regs());
790
791 // i64 needs the qq converts, which are AVX512DQ only. Two tables because the
792 // lowering picks by operand value and not by type.
793 static const CostKindTblEntry AVX512DQExactVarDivCostTable[] = {
794 { .ISD: ISD::UDIV, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5 } }, // cvt+divpd sequence
795 { .ISD: ISD::SDIV, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5 } },
796 { .ISD: ISD::UREM, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5 } },
797 { .ISD: ISD::SREM, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5 } },
798 { .ISD: ISD::UDIV, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 8 } },
799 { .ISD: ISD::SDIV, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 8 } },
800 { .ISD: ISD::UREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 8 } },
801 { .ISD: ISD::SREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 8 } },
802 { .ISD: ISD::UDIV, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 16 } },
803 { .ISD: ISD::SDIV, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 16 } },
804 { .ISD: ISD::UREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 16 } },
805 { .ISD: ISD::SREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 16 } },
806 };
807
808 static const CostKindTblEntry AVX512DQVarDivCostTable[] = {
809 { .ISD: ISD::UDIV, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 16 } },
810 { .ISD: ISD::SDIV, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 16 } },
811 { .ISD: ISD::UREM, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 16 } },
812 { .ISD: ISD::SREM, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 16 } },
813 { .ISD: ISD::UDIV, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 16 } },
814 { .ISD: ISD::SDIV, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 16 } },
815 { .ISD: ISD::UREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 16 } },
816 { .ISD: ISD::SREM, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 16 } },
817 { .ISD: ISD::UDIV, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 16 } },
818 { .ISD: ISD::SDIV, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 18 } },
819 { .ISD: ISD::UREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 16 } },
820 { .ISD: ISD::SREM, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 18 } },
821 };
822
823 // The DAG combine picks between the two sequences with these same two
824 // queries, so the cost cannot disagree with what codegen emits.
825 bool IsSignedDiv = ISD == ISD::SDIV || ISD == ISD::SREM;
826 auto OperandsFit = [&](unsigned Mantissa) {
827 if (Args.size() != 2 || !CtxI)
828 return false;
829 unsigned EltBits = LT.second.getScalarSizeInBits();
830 const DataLayout &DL = CtxI->getDataLayout();
831 auto Fits = [&](const Value *V) {
832 if (IsSignedDiv)
833 return ComputeNumSignBits(Op: V, DL, /*AC=*/nullptr, CtxI) + Mantissa >
834 EltBits;
835 return computeKnownBits(V, DL, /*AC=*/nullptr, CtxI)
836 .countMaxActiveBits() <= Mantissa;
837 };
838 return Fits(Args[0]) && Fits(Args[1]);
839 };
840
841 // An i32 divide goes through f32 when both operands fit in 24 bits, and
842 // through f64 at twice the vector width when they do not.
843 bool ExactI32 =
844 VarDivToFP && LT.second.getScalarType() == MVT::i32 &&
845 OperandsFit(APFloat::semanticsPrecision(APFloat::IEEEsingle()));
846
847 if (VarDivToFP && ST->hasDQI() && ST->useAVX512Regs() &&
848 LT.second.getScalarType() == MVT::i64) {
849 bool ExactFPDiv =
850 OperandsFit(APFloat::semanticsPrecision(APFloat::IEEEdouble()));
851
852 // Only the reciprocal chain multiplies, so only it is vpmullq gated.
853 bool SlowMultiply =
854 !ExactFPDiv && LT.second == MVT::v2i64 && ST->isPMULLQSlow();
855
856 if (!SlowMultiply) {
857 ArrayRef<CostKindTblEntry> Tbl =
858 ExactFPDiv ? AVX512DQExactVarDivCostTable : AVX512DQVarDivCostTable;
859 if (const auto *Entry = CostTableLookup(Tbl, ISD, Ty: LT.second))
860 if (auto KindCost = Entry->Cost[CostKind])
861 return LT.first * *KindCost;
862 }
863 }
864
865 static const CostKindTblEntry AVX512BWVarDivCostTable[] = {
866 { .ISD: ISD::UDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 10 } }, // unpack+cvt+divps sequence
867 { .ISD: ISD::SDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 10 } },
868 { .ISD: ISD::UREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 10 } },
869 { .ISD: ISD::SREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 10 } },
870 { .ISD: ISD::UDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 20 } },
871 { .ISD: ISD::SDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 20 } },
872 { .ISD: ISD::UREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 20 } },
873 { .ISD: ISD::SREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 20 } },
874 { .ISD: ISD::UDIV, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 40 } },
875 { .ISD: ISD::SDIV, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 40 } },
876 { .ISD: ISD::UREM, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 40 } },
877 { .ISD: ISD::SREM, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 40 } },
878 { .ISD: ISD::UDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5 } },
879 { .ISD: ISD::SDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5 } },
880 { .ISD: ISD::UREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5 } },
881 { .ISD: ISD::SREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5 } },
882 { .ISD: ISD::UDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 10 } },
883 { .ISD: ISD::SDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 10 } },
884 { .ISD: ISD::UREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 10 } },
885 { .ISD: ISD::SREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 10 } },
886 { .ISD: ISD::UDIV, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 20 } },
887 { .ISD: ISD::SDIV, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 20 } },
888 { .ISD: ISD::UREM, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 20 } },
889 { .ISD: ISD::SREM, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 20 } },
890 { .ISD: ISD::UDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8 } }, // cvt+divpd sequence
891 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8 } },
892 { .ISD: ISD::UREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8 } },
893 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8 } },
894 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 16 } },
895 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 16 } },
896 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 16 } },
897 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 16 } },
898 { .ISD: ISD::UDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 32 } },
899 { .ISD: ISD::SDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 32 } },
900 { .ISD: ISD::UREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 32 } },
901 { .ISD: ISD::SREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 32 } },
902 };
903
904 static const CostKindTblEntry AVX512BWExactVarDivCostTable[] = {
905 { .ISD: ISD::UDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3 } }, // cvt+divps sequence
906 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3 } },
907 { .ISD: ISD::UREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3 } },
908 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3 } },
909 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 5 } },
910 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 5 } },
911 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 5 } },
912 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 5 } },
913 { .ISD: ISD::UDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 10 } },
914 { .ISD: ISD::SDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 10 } },
915 { .ISD: ISD::UREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 10 } },
916 { .ISD: ISD::SREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 10 } },
917 };
918
919 if (VarDivToFP && ST->hasBWI()) {
920 ArrayRef<CostKindTblEntry> Tbl = AVX512BWVarDivCostTable;
921 if (ExactI32)
922 Tbl = AVX512BWExactVarDivCostTable;
923 if (const auto *Entry = CostTableLookup(Tbl, ISD, Ty: LT.second))
924 if (auto KindCost = Entry->Cost[CostKind])
925 return LT.first * *KindCost;
926 }
927
928 static const CostKindTblEntry AVX512VarDivCostTable[] = {
929 { .ISD: ISD::UDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 14 } }, // unpack+cvt+divps sequence
930 { .ISD: ISD::SDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 14 } },
931 { .ISD: ISD::UREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 14 } },
932 { .ISD: ISD::SREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 14 } },
933 { .ISD: ISD::UDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 28 } },
934 { .ISD: ISD::SDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 28 } },
935 { .ISD: ISD::UREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 28 } },
936 { .ISD: ISD::SREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 28 } },
937 { .ISD: ISD::UDIV, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 56 } },
938 { .ISD: ISD::SDIV, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 56 } },
939 { .ISD: ISD::UREM, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 56 } },
940 { .ISD: ISD::SREM, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 56 } },
941 { .ISD: ISD::UDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 14 } },
942 { .ISD: ISD::SDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 14 } },
943 { .ISD: ISD::UREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 14 } },
944 { .ISD: ISD::SREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 14 } },
945 { .ISD: ISD::UDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 14 } },
946 { .ISD: ISD::SDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 14 } },
947 { .ISD: ISD::UREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 14 } },
948 { .ISD: ISD::SREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 14 } },
949 { .ISD: ISD::UDIV, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 28 } },
950 { .ISD: ISD::SDIV, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 28 } },
951 { .ISD: ISD::UREM, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 28 } },
952 { .ISD: ISD::SREM, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 28 } },
953 { .ISD: ISD::UDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 28 } }, // cvt+divpd sequence
954 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 28 } },
955 { .ISD: ISD::UREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 28 } },
956 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 28 } },
957 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 28 } },
958 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 28 } },
959 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 28 } },
960 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 28 } },
961 { .ISD: ISD::UDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 56 } },
962 { .ISD: ISD::SDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 56 } },
963 { .ISD: ISD::UREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 56 } },
964 { .ISD: ISD::SREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 56 } },
965 };
966
967 static const CostKindTblEntry AVX512ExactVarDivCostTable[] = {
968 { .ISD: ISD::UDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7 } }, // cvt+divps sequence
969 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7 } },
970 { .ISD: ISD::UREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7 } },
971 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7 } },
972 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } },
973 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } },
974 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } },
975 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } },
976 { .ISD: ISD::UDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 14 } },
977 { .ISD: ISD::SDIV, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 14 } },
978 { .ISD: ISD::UREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 14 } },
979 { .ISD: ISD::SREM, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 14 } },
980 };
981
982 if (VarDivToFP && ST->hasAVX512()) {
983 ArrayRef<CostKindTblEntry> Tbl = AVX512VarDivCostTable;
984 if (ExactI32)
985 Tbl = AVX512ExactVarDivCostTable;
986 if (const auto *Entry = CostTableLookup(Tbl, ISD, Ty: LT.second))
987 if (auto KindCost = Entry->Cost[CostKind])
988 return LT.first * *KindCost;
989 }
990
991 static const CostKindTblEntry AVX2VarDivCostTable[] = {
992 { .ISD: ISD::UDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 28 } }, // unpack+cvt+divps sequence
993 { .ISD: ISD::SDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 28 } },
994 { .ISD: ISD::UREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 28 } },
995 { .ISD: ISD::SREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 28 } },
996 { .ISD: ISD::UDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 56 } },
997 { .ISD: ISD::SDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 56 } },
998 { .ISD: ISD::UREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 56 } },
999 { .ISD: ISD::SREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 56 } },
1000 { .ISD: ISD::UDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 14 } },
1001 { .ISD: ISD::SDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 14 } },
1002 { .ISD: ISD::UREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 14 } },
1003 { .ISD: ISD::SREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 14 } },
1004 { .ISD: ISD::UDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 28 } },
1005 { .ISD: ISD::SDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 28 } },
1006 { .ISD: ISD::UREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 28 } },
1007 { .ISD: ISD::SREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 28 } },
1008 { .ISD: ISD::UDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 28 } }, // cvt+divpd sequence
1009 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 28 } },
1010 { .ISD: ISD::UREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 28 } },
1011 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 28 } },
1012 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 56 } },
1013 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 56 } },
1014 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 56 } },
1015 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 56 } },
1016 };
1017
1018 static const CostKindTblEntry AVX2ExactVarDivCostTable[] = {
1019 { .ISD: ISD::UDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 9 } }, // cvt+divps sequence
1020 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8 } },
1021 { .ISD: ISD::UREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 9 } },
1022 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8 } },
1023 { .ISD: ISD::UDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } },
1024 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } },
1025 { .ISD: ISD::UREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } },
1026 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14 } },
1027 };
1028
1029 if (VarDivToFP && ST->hasAVX2()) {
1030 ArrayRef<CostKindTblEntry> Tbl = AVX2VarDivCostTable;
1031 if (ExactI32)
1032 Tbl = AVX2ExactVarDivCostTable;
1033 if (const auto *Entry = CostTableLookup(Tbl, ISD, Ty: LT.second))
1034 if (auto KindCost = Entry->Cost[CostKind])
1035 return LT.first * *KindCost;
1036 }
1037
1038 // No unsigned i32 entries below AVX2, where the u32 to f64 converts are
1039 // emulated and the fold stays off.
1040 static const CostKindTblEntry AVX1VarDivCostTable[] = {
1041 { .ISD: ISD::UDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 56 } }, // unpack+cvt+divps sequence
1042 { .ISD: ISD::SDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 56 } },
1043 { .ISD: ISD::UREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 56 } },
1044 { .ISD: ISD::SREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 56 } },
1045 { .ISD: ISD::UDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 112 } },
1046 { .ISD: ISD::SDIV, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 112 } },
1047 { .ISD: ISD::UREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 112 } },
1048 { .ISD: ISD::SREM, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 112 } },
1049 { .ISD: ISD::UDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 28 } },
1050 { .ISD: ISD::SDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 28 } },
1051 { .ISD: ISD::UREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 28 } },
1052 { .ISD: ISD::SREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 28 } },
1053 { .ISD: ISD::UDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 56 } },
1054 { .ISD: ISD::SDIV, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 56 } },
1055 { .ISD: ISD::UREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 56 } },
1056 { .ISD: ISD::SREM, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 56 } },
1057 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 44 } }, // cvt+divpd sequence
1058 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 44 } },
1059 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 88 } },
1060 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 88 } },
1061 };
1062
1063 static const CostKindTblEntry AVX1ExactVarDivCostTable[] = {
1064 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 14 } }, // cvt+divps sequence
1065 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 14 } },
1066 { .ISD: ISD::SDIV, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 28 } },
1067 { .ISD: ISD::SREM, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 28 } },
1068 };
1069
1070 if (VarDivToFP && ST->hasAVX()) {
1071 ArrayRef<CostKindTblEntry> Tbl = AVX1VarDivCostTable;
1072 if (ExactI32)
1073 Tbl = AVX1ExactVarDivCostTable;
1074 if (const auto *Entry = CostTableLookup(Tbl, ISD, Ty: LT.second))
1075 if (auto KindCost = Entry->Cost[CostKind])
1076 return LT.first * *KindCost;
1077 }
1078
1079 static const CostKindTblEntry SSE2VarDivCostTable[] = {
1080 { .ISD: ISD::UDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 56 } }, // unpack+cvt+divps sequence
1081 { .ISD: ISD::SDIV, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 56 } },
1082 { .ISD: ISD::UREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 56 } },
1083 { .ISD: ISD::SREM, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 56 } },
1084 { .ISD: ISD::UDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 28 } },
1085 { .ISD: ISD::SDIV, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 28 } },
1086 { .ISD: ISD::UREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 28 } },
1087 { .ISD: ISD::SREM, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 28 } },
1088 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 44 } }, // cvt+divpd sequence
1089 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 44 } },
1090 };
1091
1092 static const CostKindTblEntry SSE2ExactVarDivCostTable[] = {
1093 { .ISD: ISD::SDIV, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 14 } }, // cvt+divps sequence
1094 { .ISD: ISD::SREM, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 14 } },
1095 };
1096
1097 if (VarDivToFP && ST->hasSSE2()) {
1098 ArrayRef<CostKindTblEntry> Tbl = SSE2VarDivCostTable;
1099 if (ExactI32)
1100 Tbl = SSE2ExactVarDivCostTable;
1101 if (const auto *Entry = CostTableLookup(Tbl, ISD, Ty: LT.second))
1102 if (auto KindCost = Entry->Cost[CostKind])
1103 return LT.first * *KindCost;
1104 }
1105
1106 static const CostKindTblEntry AVX512BWUniformCostTable[] = {
1107 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psllw + pand.
1108 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 5, .SizeAndLatencyCost: 8 } }, // psrlw + pand.
1109 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4,.LatencyCost: 12, .CodeSizeCost: 8,.SizeAndLatencyCost: 12 } }, // psrlw, pand, pxor, psubb.
1110 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 6, .SizeAndLatencyCost: 8 } }, // psllw + pand.
1111 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 7, .SizeAndLatencyCost: 9 } }, // psrlw + pand.
1112 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 5,.LatencyCost: 10,.CodeSizeCost: 10,.SizeAndLatencyCost: 13 } }, // psrlw, pand, pxor, psubb.
1113 { .ISD: ISD::SHL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 6, .SizeAndLatencyCost: 8 } }, // psllw + pand.
1114 { .ISD: ISD::SRL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 7,.SizeAndLatencyCost: 10 } }, // psrlw + pand.
1115 { .ISD: ISD::SRA, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 5,.LatencyCost: 10,.CodeSizeCost: 10,.SizeAndLatencyCost: 15 } }, // psrlw, pand, pxor, psubb.
1116
1117 { .ISD: ISD::SHL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psllw
1118 { .ISD: ISD::SRL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrlw
1119 { .ISD: ISD::SRA, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrqw
1120 };
1121
1122 if (ST->hasBWI() && Op2Info.isUniform())
1123 if (const auto *Entry =
1124 CostTableLookup(Table: AVX512BWUniformCostTable, ISD, Ty: LT.second))
1125 if (auto KindCost = Entry->Cost[CostKind])
1126 return LT.first * *KindCost;
1127
1128 static const CostKindTblEntry AVX512UniformCostTable[] = {
1129 { .ISD: ISD::SHL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 5,.LatencyCost: 10, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psllw + split.
1130 { .ISD: ISD::SRL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 5,.LatencyCost: 10, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psrlw + split.
1131 { .ISD: ISD::SRA, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 5,.LatencyCost: 10, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psraw + split.
1132
1133 { .ISD: ISD::SHL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // pslld
1134 { .ISD: ISD::SRL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrld
1135 { .ISD: ISD::SRA, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrad
1136
1137 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psraq
1138 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllq
1139 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlq
1140 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psraq
1141 { .ISD: ISD::SHL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllq
1142 { .ISD: ISD::SRL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlq
1143 { .ISD: ISD::SRA, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psraq
1144 };
1145
1146 if (ST->hasAVX512() && Op2Info.isUniform())
1147 if (const auto *Entry =
1148 CostTableLookup(Table: AVX512UniformCostTable, ISD, Ty: LT.second))
1149 if (auto KindCost = Entry->Cost[CostKind])
1150 return LT.first * *KindCost;
1151
1152 static const CostKindTblEntry AVX2UniformCostTable[] = {
1153 // Uniform splats are cheaper for the following instructions.
1154 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psllw + pand.
1155 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 5, .SizeAndLatencyCost: 8 } }, // psrlw + pand.
1156 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 5, .CodeSizeCost: 9,.SizeAndLatencyCost: 13 } }, // psrlw, pand, pxor, psubb.
1157 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 6, .SizeAndLatencyCost: 8 } }, // psllw + pand.
1158 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 7, .SizeAndLatencyCost: 9 } }, // psrlw + pand.
1159 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 9,.CodeSizeCost: 11,.SizeAndLatencyCost: 16 } }, // psrlw, pand, pxor, psubb.
1160
1161 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllw.
1162 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlw.
1163 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psraw.
1164 { .ISD: ISD::SHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psllw.
1165 { .ISD: ISD::SRL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrlw.
1166 { .ISD: ISD::SRA, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psraw.
1167
1168 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pslld
1169 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrld
1170 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrad
1171 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // pslld
1172 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrld
1173 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // psrad
1174
1175 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllq
1176 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlq
1177 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // 2 x psrad + shuffle.
1178 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllq
1179 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlq
1180 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 9 } }, // 2 x psrad + shuffle.
1181 };
1182
1183 if (ST->hasAVX2() && Op2Info.isUniform())
1184 if (const auto *Entry =
1185 CostTableLookup(Table: AVX2UniformCostTable, ISD, Ty: LT.second))
1186 if (auto KindCost = Entry->Cost[CostKind])
1187 return LT.first * *KindCost;
1188
1189 static const CostKindTblEntry AVXUniformCostTable[] = {
1190 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 6, .SizeAndLatencyCost: 8 } }, // psllw + pand.
1191 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 8 } }, // psrlw + pand.
1192 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 6, .CodeSizeCost: 9,.SizeAndLatencyCost: 13 } }, // psrlw, pand, pxor, psubb.
1193 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 8,.CodeSizeCost: 11,.SizeAndLatencyCost: 14 } }, // psllw + pand + split.
1194 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 9,.CodeSizeCost: 10,.SizeAndLatencyCost: 14 } }, // psrlw + pand + split.
1195 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 10,.LatencyCost: 11,.CodeSizeCost: 16,.SizeAndLatencyCost: 21 } }, // psrlw, pand, pxor, psubb + split.
1196
1197 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllw.
1198 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlw.
1199 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psraw.
1200 { .ISD: ISD::SHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psllw + split.
1201 { .ISD: ISD::SRL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psrlw + split.
1202 { .ISD: ISD::SRA, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psraw + split.
1203
1204 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pslld.
1205 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrld.
1206 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrad.
1207 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // pslld + split.
1208 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psrld + split.
1209 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // psrad + split.
1210
1211 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllq.
1212 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlq.
1213 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // 2 x psrad + shuffle.
1214 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } }, // psllq + split.
1215 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } }, // psrlq + split.
1216 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7,.CodeSizeCost: 10,.SizeAndLatencyCost: 13 } }, // 2 x (2 x psrad + shuffle) + split.
1217 };
1218
1219 // XOP has faster vXi8 shifts.
1220 if (ST->hasAVX() && Op2Info.isUniform() &&
1221 (!ST->hasXOP() || LT.second.getScalarSizeInBits() != 8))
1222 if (const auto *Entry =
1223 CostTableLookup(Table: AVXUniformCostTable, ISD, Ty: LT.second))
1224 if (auto KindCost = Entry->Cost[CostKind])
1225 return LT.first * *KindCost;
1226
1227 static const CostKindTblEntry SSE2UniformCostTable[] = {
1228 // Uniform splats are cheaper for the following instructions.
1229 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 10, .CodeSizeCost: 6, .SizeAndLatencyCost: 9 } }, // psllw + pand.
1230 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 13, .CodeSizeCost: 5, .SizeAndLatencyCost: 9 } }, // psrlw + pand.
1231 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 15, .CodeSizeCost: 9,.SizeAndLatencyCost: 13 } }, // pcmpgtb sequence.
1232
1233 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllw.
1234 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlw.
1235 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psraw.
1236
1237 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pslld
1238 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrld.
1239 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrad.
1240
1241 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psllq.
1242 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psrlq.
1243 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 9, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // 2*psrlq + xor + sub.
1244 };
1245
1246 if (ST->hasSSE2() && Op2Info.isUniform() &&
1247 (!ST->hasXOP() || LT.second.getScalarSizeInBits() != 8))
1248 if (const auto *Entry =
1249 CostTableLookup(Table: SSE2UniformCostTable, ISD, Ty: LT.second))
1250 if (auto KindCost = Entry->Cost[CostKind])
1251 return LT.first * *KindCost;
1252
1253 static const CostKindTblEntry AVX512DQCostTable[] = {
1254 { .ISD: ISD::MUL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 15, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // pmullq
1255 { .ISD: ISD::MUL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 15, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // pmullq
1256 { .ISD: ISD::MUL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 15, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } } // pmullq
1257 };
1258
1259 // Look for AVX512DQ lowering tricks for custom cases.
1260 if (ST->hasDQI())
1261 if (const auto *Entry = CostTableLookup(Table: AVX512DQCostTable, ISD, Ty: LT.second))
1262 if (auto KindCost = Entry->Cost[CostKind])
1263 return LT.first * *KindCost;
1264
1265 static const CostKindTblEntry AVX512BWCostTable[] = {
1266 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // extend/vpsllvw/pack sequence.
1267 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // extend/vpsrlvw/pack sequence.
1268 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // extend/vpsravw/pack sequence.
1269 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 23,.CodeSizeCost: 11,.SizeAndLatencyCost: 16 } }, // extend/vpsllvw/pack sequence.
1270 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 30,.CodeSizeCost: 12,.SizeAndLatencyCost: 18 } }, // extend/vpsrlvw/pack sequence.
1271 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 13,.CodeSizeCost: 24,.SizeAndLatencyCost: 30 } }, // extend/vpsravw/pack sequence.
1272 { .ISD: ISD::SHL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 19,.CodeSizeCost: 13,.SizeAndLatencyCost: 15 } }, // extend/vpsllvw/pack sequence.
1273 { .ISD: ISD::SRL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 27,.CodeSizeCost: 15,.SizeAndLatencyCost: 18 } }, // extend/vpsrlvw/pack sequence.
1274 { .ISD: ISD::SRA, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 15,.CodeSizeCost: 30,.SizeAndLatencyCost: 30 } }, // extend/vpsravw/pack sequence.
1275
1276 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsllvw
1277 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsrlvw
1278 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsravw
1279 { .ISD: ISD::SHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsllvw
1280 { .ISD: ISD::SRL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsrlvw
1281 { .ISD: ISD::SRA, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsravw
1282 { .ISD: ISD::SHL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsllvw
1283 { .ISD: ISD::SRL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsrlvw
1284 { .ISD: ISD::SRA, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsravw
1285
1286 { .ISD: ISD::ADD, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // paddb
1287 { .ISD: ISD::ADD, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // paddw
1288
1289 { .ISD: ISD::ADD, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // paddb
1290 { .ISD: ISD::ADD, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // paddw
1291 { .ISD: ISD::ADD, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // paddd
1292 { .ISD: ISD::ADD, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // paddq
1293
1294 { .ISD: ISD::SUB, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psubb
1295 { .ISD: ISD::SUB, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psubw
1296
1297 { .ISD: ISD::MUL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 12, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // extend/pmullw/trunc
1298 { .ISD: ISD::MUL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 7,.SizeAndLatencyCost: 10 } }, // pmaddubsw
1299 { .ISD: ISD::MUL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 7,.SizeAndLatencyCost: 10 } }, // pmaddubsw
1300 { .ISD: ISD::MUL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pmullw
1301
1302 { .ISD: ISD::SUB, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psubb
1303 { .ISD: ISD::SUB, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psubw
1304 { .ISD: ISD::SUB, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psubd
1305 { .ISD: ISD::SUB, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psubq
1306 };
1307
1308 // Look for AVX512BW lowering tricks for custom cases.
1309 if (ST->hasBWI())
1310 if (const auto *Entry = CostTableLookup(Table: AVX512BWCostTable, ISD, Ty: LT.second))
1311 if (auto KindCost = Entry->Cost[CostKind])
1312 return LT.first * *KindCost;
1313
1314 static const CostKindTblEntry AVX512CostTable[] = {
1315 { .ISD: ISD::SHL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 19,.CodeSizeCost: 27,.SizeAndLatencyCost: 33 } }, // vpblendv+split sequence.
1316 { .ISD: ISD::SRL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 19,.CodeSizeCost: 30,.SizeAndLatencyCost: 36 } }, // vpblendv+split sequence.
1317 { .ISD: ISD::SRA, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 37, .LatencyCost: 37,.CodeSizeCost: 51,.SizeAndLatencyCost: 63 } }, // vpblendv+split sequence.
1318
1319 { .ISD: ISD::SHL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 16,.CodeSizeCost: 11,.SizeAndLatencyCost: 15 } }, // 2*extend/vpsrlvd/pack sequence.
1320 { .ISD: ISD::SRL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 16,.CodeSizeCost: 11,.SizeAndLatencyCost: 15 } }, // 2*extend/vpsrlvd/pack sequence.
1321 { .ISD: ISD::SRA, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 16,.CodeSizeCost: 11,.SizeAndLatencyCost: 15 } }, // 2*extend/vpsravd/pack sequence.
1322
1323 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1324 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1325 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1326 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1327 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1328 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1329 { .ISD: ISD::SHL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1330 { .ISD: ISD::SRL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1331 { .ISD: ISD::SRA, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1332
1333 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1334 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1335 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1336 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1337 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1338 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1339 { .ISD: ISD::SHL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1340 { .ISD: ISD::SRL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1341 { .ISD: ISD::SRA, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1342
1343 { .ISD: ISD::ADD, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } }, // 2*paddb + split
1344 { .ISD: ISD::ADD, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } }, // 2*paddw + split
1345
1346 { .ISD: ISD::SUB, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } }, // 2*psubb + split
1347 { .ISD: ISD::SUB, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } }, // 2*psubw + split
1348
1349 { .ISD: ISD::AND, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1350 { .ISD: ISD::AND, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1351 { .ISD: ISD::AND, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1352 { .ISD: ISD::AND, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1353
1354 { .ISD: ISD::OR, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1355 { .ISD: ISD::OR, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1356 { .ISD: ISD::OR, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1357 { .ISD: ISD::OR, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1358
1359 { .ISD: ISD::XOR, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1360 { .ISD: ISD::XOR, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1361 { .ISD: ISD::XOR, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1362 { .ISD: ISD::XOR, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1363
1364 { .ISD: ISD::MUL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pmulld (Skylake from agner.org)
1365 { .ISD: ISD::MUL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pmulld (Skylake from agner.org)
1366 { .ISD: ISD::MUL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pmulld (Skylake from agner.org)
1367 { .ISD: ISD::MUL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 9, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } }, // 3*pmuludq/3*shift/2*add
1368 { .ISD: ISD::MUL, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1 } }, // Skylake from http://www.agner.org/
1369
1370 { .ISD: X86ISD::PMULUDQ, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1371
1372 { .ISD: ISD::FNEG, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // Skylake from http://www.agner.org/
1373 { .ISD: ISD::FADD, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1374 { .ISD: ISD::FADD, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1375 { .ISD: ISD::FSUB, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1376 { .ISD: ISD::FSUB, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1377 { .ISD: ISD::FMUL, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1378 { .ISD: ISD::FMUL, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1379 { .ISD: ISD::FMUL, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1380 { .ISD: ISD::FMUL, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1381
1382 { .ISD: ISD::FDIV, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 14, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1383 { .ISD: ISD::FDIV, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 14, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1384 { .ISD: ISD::FDIV, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 14, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1385 { .ISD: ISD::FDIV, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 23, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // Skylake from http://www.agner.org/
1386
1387 { .ISD: ISD::FNEG, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // Skylake from http://www.agner.org/
1388 { .ISD: ISD::FADD, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1389 { .ISD: ISD::FADD, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1390 { .ISD: ISD::FSUB, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1391 { .ISD: ISD::FSUB, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1392 { .ISD: ISD::FMUL, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1393 { .ISD: ISD::FMUL, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1394 { .ISD: ISD::FMUL, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1395 { .ISD: ISD::FMUL, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1396
1397 { .ISD: ISD::FDIV, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1398 { .ISD: ISD::FDIV, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1399 { .ISD: ISD::FDIV, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
1400 { .ISD: ISD::FDIV, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 18, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // Skylake from http://www.agner.org/
1401 };
1402
1403 if (ST->hasAVX512())
1404 if (const auto *Entry = CostTableLookup(Table: AVX512CostTable, ISD, Ty: LT.second))
1405 if (auto KindCost = Entry->Cost[CostKind])
1406 return LT.first * *KindCost;
1407
1408 static const CostKindTblEntry AVX2ShiftCostTable[] = {
1409 // Shifts on vXi64/vXi32 on AVX2 is legal even though we declare to
1410 // customize them to detect the cases where shift amount is a scalar one.
1411 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vpsllvd (Haswell from agner.org)
1412 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vpsrlvd (Haswell from agner.org)
1413 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vpsravd (Haswell from agner.org)
1414 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vpsllvd (Haswell from agner.org)
1415 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vpsrlvd (Haswell from agner.org)
1416 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vpsravd (Haswell from agner.org)
1417 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsllvq (Haswell from agner.org)
1418 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsrlvq (Haswell from agner.org)
1419 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpsllvq (Haswell from agner.org)
1420 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpsrlvq (Haswell from agner.org)
1421 };
1422
1423 if (ST->hasAVX512()) {
1424 if (ISD == ISD::SHL && LT.second == MVT::v32i16 && Op2Info.isConstant())
1425 // On AVX512, a packed v32i16 shift left by a constant build_vector
1426 // is lowered into a vector multiply (vpmullw).
1427 return getArithmeticInstrCost(Opcode: Instruction::Mul, Ty, CostKind,
1428 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
1429 }
1430
1431 // Look for AVX2 lowering tricks (XOP is always better at v4i32 shifts).
1432 if (ST->hasAVX2() && !(ST->hasXOP() && LT.second == MVT::v4i32)) {
1433 if (ISD == ISD::SHL && LT.second == MVT::v16i16 &&
1434 Op2Info.isConstant())
1435 // On AVX2, a packed v16i16 shift left by a constant build_vector
1436 // is lowered into a vector multiply (vpmullw).
1437 return getArithmeticInstrCost(Opcode: Instruction::Mul, Ty, CostKind,
1438 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
1439
1440 if (const auto *Entry = CostTableLookup(Table: AVX2ShiftCostTable, ISD, Ty: LT.second))
1441 if (auto KindCost = Entry->Cost[CostKind])
1442 return LT.first * *KindCost;
1443 }
1444
1445 static const CostKindTblEntry XOPShiftCostTable[] = {
1446 // 128bit shifts take 1cy, but right shifts require negation beforehand.
1447 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1448 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1449 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1450 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1451 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1452 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1453 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1454 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1455 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1456 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1457 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1458 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1459 // 256bit shifts require splitting if AVX2 didn't catch them above.
1460 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1461 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1462 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1463 { .ISD: ISD::SHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1464 { .ISD: ISD::SRL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1465 { .ISD: ISD::SRA, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1466 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1467 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1468 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1469 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1470 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1471 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
1472 };
1473
1474 // Look for XOP lowering tricks.
1475 if (ST->hasXOP()) {
1476 // If the right shift is constant then we'll fold the negation so
1477 // it's as cheap as a left shift.
1478 int ShiftISD = ISD;
1479 if ((ShiftISD == ISD::SRL || ShiftISD == ISD::SRA) && Op2Info.isConstant())
1480 ShiftISD = ISD::SHL;
1481 if (const auto *Entry =
1482 CostTableLookup(Table: XOPShiftCostTable, ISD: ShiftISD, Ty: LT.second))
1483 if (auto KindCost = Entry->Cost[CostKind])
1484 return LT.first * *KindCost;
1485 }
1486
1487 if (ISD == ISD::SHL && !Op2Info.isUniform() && Op2Info.isConstant()) {
1488 MVT VT = LT.second;
1489 // Vector shift left by non uniform constant can be lowered
1490 // into vector multiply.
1491 if (((VT == MVT::v8i16 || VT == MVT::v4i32) && ST->hasSSE2()) ||
1492 ((VT == MVT::v16i16 || VT == MVT::v8i32) && ST->hasAVX()))
1493 ISD = ISD::MUL;
1494 }
1495
1496 static const CostKindTblEntry GLMCostTable[] = {
1497 { .ISD: ISD::FDIV, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 19, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // divss
1498 { .ISD: ISD::FDIV, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 35, .LatencyCost: 36, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // divps
1499 { .ISD: ISD::FDIV, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 33, .LatencyCost: 34, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // divsd
1500 { .ISD: ISD::FDIV, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 65, .LatencyCost: 66, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // divpd
1501 };
1502
1503 if (ST->useGLMDivSqrtCosts())
1504 if (const auto *Entry = CostTableLookup(Table: GLMCostTable, ISD, Ty: LT.second))
1505 if (auto KindCost = Entry->Cost[CostKind])
1506 return LT.first * *KindCost;
1507
1508 static const CostKindTblEntry SLMCostTable[] = {
1509 { .ISD: ISD::MUL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } }, // pmulld
1510 { .ISD: ISD::MUL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pmullw
1511 { .ISD: ISD::FMUL, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // mulsd
1512 { .ISD: ISD::FMUL, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // mulss
1513 { .ISD: ISD::FMUL, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // mulpd
1514 { .ISD: ISD::FMUL, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // mulps
1515 { .ISD: ISD::FDIV, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 19, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // divss
1516 { .ISD: ISD::FDIV, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 39, .LatencyCost: 39, .CodeSizeCost: 1, .SizeAndLatencyCost: 6 } }, // divps
1517 { .ISD: ISD::FDIV, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 32, .LatencyCost: 34, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // divsd
1518 { .ISD: ISD::FDIV, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 69, .LatencyCost: 69, .CodeSizeCost: 1, .SizeAndLatencyCost: 6 } }, // divpd
1519 { .ISD: ISD::FADD, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // addpd
1520 { .ISD: ISD::FSUB, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // subpd
1521 // v2i64/v4i64 mul is custom lowered as a series of long:
1522 // multiplies(3), shifts(3) and adds(2)
1523 // slm muldq version throughput is 2 and addq throughput 4
1524 // thus: 3X2 (muldq throughput) + 3X1 (shift throughput) +
1525 // 3X4 (addq throughput) = 17
1526 { .ISD: ISD::MUL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 22, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } },
1527 // slm addq\subq throughput is 4
1528 { .ISD: ISD::ADD, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
1529 { .ISD: ISD::SUB, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
1530 };
1531
1532 if (ST->useSLMArithCosts())
1533 if (const auto *Entry = CostTableLookup(Table: SLMCostTable, ISD, Ty: LT.second))
1534 if (auto KindCost = Entry->Cost[CostKind])
1535 return LT.first * *KindCost;
1536
1537 static const CostKindTblEntry AVX2CostTable[] = {
1538 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 21,.CodeSizeCost: 11,.SizeAndLatencyCost: 16 } }, // vpblendvb sequence.
1539 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 23,.CodeSizeCost: 11,.SizeAndLatencyCost: 22 } }, // vpblendvb sequence.
1540 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 18, .CodeSizeCost: 5,.SizeAndLatencyCost: 10 } }, // extend/vpsrlvd/pack sequence.
1541 { .ISD: ISD::SHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 10,.CodeSizeCost: 10,.SizeAndLatencyCost: 14 } }, // extend/vpsrlvd/pack sequence.
1542
1543 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 27,.CodeSizeCost: 12,.SizeAndLatencyCost: 18 } }, // vpblendvb sequence.
1544 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 30,.CodeSizeCost: 12,.SizeAndLatencyCost: 24 } }, // vpblendvb sequence.
1545 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 5,.SizeAndLatencyCost: 10 } }, // extend/vpsrlvd/pack sequence.
1546 { .ISD: ISD::SRL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 10,.CodeSizeCost: 10,.SizeAndLatencyCost: 14 } }, // extend/vpsrlvd/pack sequence.
1547
1548 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 17,.CodeSizeCost: 24,.SizeAndLatencyCost: 30 } }, // vpblendvb sequence.
1549 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 20,.CodeSizeCost: 24,.SizeAndLatencyCost: 43 } }, // vpblendvb sequence.
1550 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 5,.SizeAndLatencyCost: 10 } }, // extend/vpsravd/pack sequence.
1551 { .ISD: ISD::SRA, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 10,.CodeSizeCost: 10,.SizeAndLatencyCost: 14 } }, // extend/vpsravd/pack sequence.
1552 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } }, // srl/xor/sub sequence.
1553 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 9 } }, // srl/xor/sub sequence.
1554
1555 { .ISD: ISD::SUB, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psubb
1556 { .ISD: ISD::ADD, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // paddb
1557 { .ISD: ISD::SUB, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psubw
1558 { .ISD: ISD::ADD, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // paddw
1559 { .ISD: ISD::SUB, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psubd
1560 { .ISD: ISD::ADD, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // paddd
1561 { .ISD: ISD::SUB, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psubq
1562 { .ISD: ISD::ADD, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // paddq
1563
1564 { .ISD: ISD::MUL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 18, .CodeSizeCost: 6,.SizeAndLatencyCost: 12 } }, // extend/pmullw/pack
1565 { .ISD: ISD::MUL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 8,.SizeAndLatencyCost: 16 } }, // pmaddubsw
1566 { .ISD: ISD::MUL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pmullw
1567 { .ISD: ISD::MUL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pmulld
1568 { .ISD: ISD::MUL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pmulld
1569 { .ISD: ISD::MUL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 10, .CodeSizeCost: 8,.SizeAndLatencyCost: 13 } }, // 3*pmuludq/3*shift/2*add
1570 { .ISD: ISD::MUL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 10, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } }, // 3*pmuludq/3*shift/2*add
1571
1572 { .ISD: X86ISD::PMULUDQ, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1573
1574 { .ISD: ISD::FNEG, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vxorpd
1575 { .ISD: ISD::FNEG, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vxorps
1576
1577 { .ISD: ISD::FADD, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vaddsd
1578 { .ISD: ISD::FADD, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vaddss
1579 { .ISD: ISD::FADD, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vaddpd
1580 { .ISD: ISD::FADD, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vaddps
1581 { .ISD: ISD::FADD, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vaddpd
1582 { .ISD: ISD::FADD, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vaddps
1583
1584 { .ISD: ISD::FSUB, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsubsd
1585 { .ISD: ISD::FSUB, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsubss
1586 { .ISD: ISD::FSUB, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsubpd
1587 { .ISD: ISD::FSUB, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsubps
1588 { .ISD: ISD::FSUB, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vsubpd
1589 { .ISD: ISD::FSUB, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vsubps
1590
1591 { .ISD: ISD::FMUL, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vmulsd
1592 { .ISD: ISD::FMUL, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vmulss
1593 { .ISD: ISD::FMUL, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vmulpd
1594 { .ISD: ISD::FMUL, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vmulps
1595 { .ISD: ISD::FMUL, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vmulpd
1596 { .ISD: ISD::FMUL, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vmulps
1597
1598 { .ISD: ISD::FDIV, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 13, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vdivss
1599 { .ISD: ISD::FDIV, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 13, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vdivps
1600 { .ISD: ISD::FDIV, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 21, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vdivps
1601 { .ISD: ISD::FDIV, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 20, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vdivsd
1602 { .ISD: ISD::FDIV, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 20, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vdivpd
1603 { .ISD: ISD::FDIV, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 28, .LatencyCost: 35, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vdivpd
1604 };
1605
1606 // Look for AVX2 lowering tricks for custom cases.
1607 if (ST->hasAVX2())
1608 if (const auto *Entry = CostTableLookup(Table: AVX2CostTable, ISD, Ty: LT.second))
1609 if (auto KindCost = Entry->Cost[CostKind])
1610 return LT.first * *KindCost;
1611
1612 static const CostKindTblEntry AVX1CostTable[] = {
1613 // We don't have to scalarize unsupported ops. We can issue two half-sized
1614 // operations and we only need to extract the upper YMM half.
1615 // Two ops + 1 extract + 1 insert = 4.
1616 { .ISD: ISD::MUL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 11, .CodeSizeCost: 18, .SizeAndLatencyCost: 19 } }, // pmaddubsw + split
1617 { .ISD: ISD::MUL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 6, .CodeSizeCost: 8, .SizeAndLatencyCost: 12 } }, // 2*pmaddubsw/3*and/psllw/or
1618 { .ISD: ISD::MUL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // pmullw + split
1619 { .ISD: ISD::MUL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 10 } }, // pmulld + split
1620 { .ISD: ISD::MUL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // pmulld
1621 { .ISD: ISD::MUL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 15, .CodeSizeCost: 19, .SizeAndLatencyCost: 20 } },
1622
1623 { .ISD: X86ISD::PMULUDQ, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // pmuludq + split
1624
1625 { .ISD: ISD::AND, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vandps
1626 { .ISD: ISD::AND, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vandps
1627 { .ISD: ISD::AND, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vandps
1628 { .ISD: ISD::AND, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vandps
1629
1630 { .ISD: ISD::OR, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vorps
1631 { .ISD: ISD::OR, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vorps
1632 { .ISD: ISD::OR, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vorps
1633 { .ISD: ISD::OR, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vorps
1634
1635 { .ISD: ISD::XOR, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vxorps
1636 { .ISD: ISD::XOR, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vxorps
1637 { .ISD: ISD::XOR, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vxorps
1638 { .ISD: ISD::XOR, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vxorps
1639
1640 { .ISD: ISD::SUB, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psubb + split
1641 { .ISD: ISD::ADD, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // paddb + split
1642 { .ISD: ISD::SUB, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psubw + split
1643 { .ISD: ISD::ADD, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // paddw + split
1644 { .ISD: ISD::SUB, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psubd + split
1645 { .ISD: ISD::ADD, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // paddd + split
1646 { .ISD: ISD::SUB, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // psubq + split
1647 { .ISD: ISD::ADD, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // paddq + split
1648 { .ISD: ISD::SUB, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // psubq
1649 { .ISD: ISD::ADD, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // paddq
1650
1651 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 21,.CodeSizeCost: 11,.SizeAndLatencyCost: 17 } }, // pblendvb sequence.
1652 { .ISD: ISD::SHL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 22, .LatencyCost: 22,.CodeSizeCost: 27,.SizeAndLatencyCost: 40 } }, // pblendvb sequence + split.
1653 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 9,.CodeSizeCost: 11,.SizeAndLatencyCost: 11 } }, // pblendvb sequence.
1654 { .ISD: ISD::SHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 16,.CodeSizeCost: 24,.SizeAndLatencyCost: 25 } }, // pblendvb sequence + split.
1655 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } }, // pslld/paddd/cvttps2dq/pmulld
1656 { .ISD: ISD::SHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 11,.CodeSizeCost: 12,.SizeAndLatencyCost: 17 } }, // pslld/paddd/cvttps2dq/pmulld + split
1657 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } }, // Shift each lane + blend.
1658 { .ISD: ISD::SHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7,.CodeSizeCost: 11,.SizeAndLatencyCost: 15 } }, // Shift each lane + blend + split.
1659
1660 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 27,.CodeSizeCost: 12,.SizeAndLatencyCost: 18 } }, // pblendvb sequence.
1661 { .ISD: ISD::SRL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 23, .LatencyCost: 23,.CodeSizeCost: 30,.SizeAndLatencyCost: 43 } }, // pblendvb sequence + split.
1662 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 16,.CodeSizeCost: 14,.SizeAndLatencyCost: 22 } }, // pblendvb sequence.
1663 { .ISD: ISD::SRL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 28, .LatencyCost: 30,.CodeSizeCost: 31,.SizeAndLatencyCost: 48 } }, // pblendvb sequence + split.
1664 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7,.CodeSizeCost: 12,.SizeAndLatencyCost: 16 } }, // Shift each lane + blend.
1665 { .ISD: ISD::SRL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14,.CodeSizeCost: 26,.SizeAndLatencyCost: 34 } }, // Shift each lane + blend + split.
1666 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } }, // Shift each lane + blend.
1667 { .ISD: ISD::SRL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7,.CodeSizeCost: 11,.SizeAndLatencyCost: 15 } }, // Shift each lane + blend + split.
1668
1669 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 21, .LatencyCost: 22,.CodeSizeCost: 24,.SizeAndLatencyCost: 36 } }, // pblendvb sequence.
1670 { .ISD: ISD::SRA, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 44, .LatencyCost: 45,.CodeSizeCost: 51,.SizeAndLatencyCost: 76 } }, // pblendvb sequence + split.
1671 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 16,.CodeSizeCost: 14,.SizeAndLatencyCost: 22 } }, // pblendvb sequence.
1672 { .ISD: ISD::SRA, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 28, .LatencyCost: 30,.CodeSizeCost: 31,.SizeAndLatencyCost: 48 } }, // pblendvb sequence + split.
1673 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 7,.CodeSizeCost: 12,.SizeAndLatencyCost: 16 } }, // Shift each lane + blend.
1674 { .ISD: ISD::SRA, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14,.CodeSizeCost: 26,.SizeAndLatencyCost: 34 } }, // Shift each lane + blend + split.
1675 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 6,.CodeSizeCost: 10,.SizeAndLatencyCost: 14 } }, // Shift each lane + blend.
1676 { .ISD: ISD::SRA, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 12,.CodeSizeCost: 22,.SizeAndLatencyCost: 30 } }, // Shift each lane + blend + split.
1677
1678 { .ISD: ISD::FNEG, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BTVER2 from http://www.agner.org/
1679 { .ISD: ISD::FNEG, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BTVER2 from http://www.agner.org/
1680
1681 { .ISD: ISD::FADD, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BDVER2 from http://www.agner.org/
1682 { .ISD: ISD::FADD, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BDVER2 from http://www.agner.org/
1683 { .ISD: ISD::FADD, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BDVER2 from http://www.agner.org/
1684 { .ISD: ISD::FADD, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BDVER2 from http://www.agner.org/
1685 { .ISD: ISD::FADD, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BDVER2 from http://www.agner.org/
1686 { .ISD: ISD::FADD, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BDVER2 from http://www.agner.org/
1687
1688 { .ISD: ISD::FSUB, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BDVER2 from http://www.agner.org/
1689 { .ISD: ISD::FSUB, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BDVER2 from http://www.agner.org/
1690 { .ISD: ISD::FSUB, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BDVER2 from http://www.agner.org/
1691 { .ISD: ISD::FSUB, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BDVER2 from http://www.agner.org/
1692 { .ISD: ISD::FSUB, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BDVER2 from http://www.agner.org/
1693 { .ISD: ISD::FSUB, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BDVER2 from http://www.agner.org/
1694
1695 { .ISD: ISD::FMUL, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BTVER2 from http://www.agner.org/
1696 { .ISD: ISD::FMUL, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BTVER2 from http://www.agner.org/
1697 { .ISD: ISD::FMUL, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BTVER2 from http://www.agner.org/
1698 { .ISD: ISD::FMUL, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // BTVER2 from http://www.agner.org/
1699 { .ISD: ISD::FMUL, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BTVER2 from http://www.agner.org/
1700 { .ISD: ISD::FMUL, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BTVER2 from http://www.agner.org/
1701
1702 { .ISD: ISD::FDIV, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // SNB from http://www.agner.org/
1703 { .ISD: ISD::FDIV, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // SNB from http://www.agner.org/
1704 { .ISD: ISD::FDIV, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 28, .LatencyCost: 29, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // SNB from http://www.agner.org/
1705 { .ISD: ISD::FDIV, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 22, .LatencyCost: 22, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // SNB from http://www.agner.org/
1706 { .ISD: ISD::FDIV, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 22, .LatencyCost: 22, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // SNB from http://www.agner.org/
1707 { .ISD: ISD::FDIV, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 44, .LatencyCost: 45, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // SNB from http://www.agner.org/
1708 };
1709
1710 if (ST->hasAVX())
1711 if (const auto *Entry = CostTableLookup(Table: AVX1CostTable, ISD, Ty: LT.second))
1712 if (auto KindCost = Entry->Cost[CostKind])
1713 return LT.first * *KindCost;
1714
1715 static const CostKindTblEntry SSE42CostTable[] = {
1716 { .ISD: ISD::FADD, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1717 { .ISD: ISD::FADD, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1718 { .ISD: ISD::FADD, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1719 { .ISD: ISD::FADD, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1720
1721 { .ISD: ISD::FSUB, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1722 { .ISD: ISD::FSUB, .Type: MVT::f32 , .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1723 { .ISD: ISD::FSUB, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1724 { .ISD: ISD::FSUB, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1725
1726 { .ISD: ISD::FMUL, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1727 { .ISD: ISD::FMUL, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1728 { .ISD: ISD::FMUL, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1729 { .ISD: ISD::FMUL, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1730
1731 { .ISD: ISD::FDIV, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1732 { .ISD: ISD::FDIV, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1733 { .ISD: ISD::FDIV, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 22, .LatencyCost: 22, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1734 { .ISD: ISD::FDIV, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 22, .LatencyCost: 22, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
1735
1736 { .ISD: ISD::MUL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 10,.CodeSizeCost: 10,.SizeAndLatencyCost: 10 } } // 3*pmuludq/3*shift/2*add
1737 };
1738
1739 if (ST->hasSSE42())
1740 if (const auto *Entry = CostTableLookup(Table: SSE42CostTable, ISD, Ty: LT.second))
1741 if (auto KindCost = Entry->Cost[CostKind])
1742 return LT.first * *KindCost;
1743
1744 static const CostKindTblEntry SSE41CostTable[] = {
1745 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 24,.CodeSizeCost: 17,.SizeAndLatencyCost: 22 } }, // pblendvb sequence.
1746 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 14,.CodeSizeCost: 11,.SizeAndLatencyCost: 11 } }, // pblendvb sequence.
1747 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 20, .CodeSizeCost: 4,.SizeAndLatencyCost: 10 } }, // pslld/paddd/cvttps2dq/pmulld
1748
1749 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 27,.CodeSizeCost: 18,.SizeAndLatencyCost: 24 } }, // pblendvb sequence.
1750 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 22, .LatencyCost: 26,.CodeSizeCost: 23,.SizeAndLatencyCost: 27 } }, // pblendvb sequence.
1751 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 17,.CodeSizeCost: 15,.SizeAndLatencyCost: 19 } }, // Shift each lane + blend.
1752 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // splat+shuffle sequence.
1753
1754 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 38, .LatencyCost: 41,.CodeSizeCost: 30,.SizeAndLatencyCost: 36 } }, // pblendvb sequence.
1755 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 22, .LatencyCost: 26,.CodeSizeCost: 23,.SizeAndLatencyCost: 27 } }, // pblendvb sequence.
1756 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 17,.CodeSizeCost: 15,.SizeAndLatencyCost: 19 } }, // Shift each lane + blend.
1757 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 17, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // splat+shuffle sequence.
1758
1759 { .ISD: ISD::MUL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } } // pmulld (Nehalem from agner.org)
1760 };
1761
1762 if (ST->hasSSE41())
1763 if (const auto *Entry = CostTableLookup(Table: SSE41CostTable, ISD, Ty: LT.second))
1764 if (auto KindCost = Entry->Cost[CostKind])
1765 return LT.first * *KindCost;
1766
1767 static const CostKindTblEntry SSSE3CostTable[] = {
1768 { .ISD: ISD::MUL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 18,.CodeSizeCost: 10,.SizeAndLatencyCost: 12 } }, // 2*pmaddubsw/3*and/psllw/or
1769 };
1770
1771 if (ST->hasSSSE3())
1772 if (const auto *Entry = CostTableLookup(Table: SSSE3CostTable, ISD, Ty: LT.second))
1773 if (auto KindCost = Entry->Cost[CostKind])
1774 return LT.first * *KindCost;
1775
1776 static const CostKindTblEntry SSE2CostTable[] = {
1777 // We don't correctly identify costs of casts because they are marked as
1778 // custom.
1779 { .ISD: ISD::SHL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 21,.CodeSizeCost: 26,.SizeAndLatencyCost: 28 } }, // cmpgtb sequence.
1780 { .ISD: ISD::SHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 24, .LatencyCost: 27,.CodeSizeCost: 16,.SizeAndLatencyCost: 20 } }, // cmpgtw sequence.
1781 { .ISD: ISD::SHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 19,.CodeSizeCost: 10,.SizeAndLatencyCost: 12 } }, // pslld/paddd/cvttps2dq/pmuludq.
1782 { .ISD: ISD::SHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // splat+shuffle sequence.
1783
1784 { .ISD: ISD::SRL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 28,.CodeSizeCost: 27,.SizeAndLatencyCost: 30 } }, // cmpgtb sequence.
1785 { .ISD: ISD::SRL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 19,.CodeSizeCost: 31,.SizeAndLatencyCost: 31 } }, // cmpgtw sequence.
1786 { .ISD: ISD::SRL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 12,.CodeSizeCost: 15,.SizeAndLatencyCost: 19 } }, // Shift each lane + blend.
1787 { .ISD: ISD::SRL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } }, // splat+shuffle sequence.
1788
1789 { .ISD: ISD::SRA, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 27, .LatencyCost: 30,.CodeSizeCost: 54,.SizeAndLatencyCost: 54 } }, // unpacked cmpgtb sequence.
1790 { .ISD: ISD::SRA, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 19,.CodeSizeCost: 31,.SizeAndLatencyCost: 31 } }, // cmpgtw sequence.
1791 { .ISD: ISD::SRA, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 12,.CodeSizeCost: 15,.SizeAndLatencyCost: 19 } }, // Shift each lane + blend.
1792 { .ISD: ISD::SRA, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 11,.CodeSizeCost: 12,.SizeAndLatencyCost: 16 } }, // srl/xor/sub splat+shuffle sequence.
1793
1794 { .ISD: ISD::AND, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pand
1795 { .ISD: ISD::AND, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pand
1796 { .ISD: ISD::AND, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pand
1797 { .ISD: ISD::AND, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pand
1798
1799 { .ISD: ISD::OR, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // por
1800 { .ISD: ISD::OR, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // por
1801 { .ISD: ISD::OR, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // por
1802 { .ISD: ISD::OR, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // por
1803
1804 { .ISD: ISD::XOR, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pxor
1805 { .ISD: ISD::XOR, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pxor
1806 { .ISD: ISD::XOR, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pxor
1807 { .ISD: ISD::XOR, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pxor
1808
1809 { .ISD: ISD::ADD, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // paddq
1810 { .ISD: ISD::SUB, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // psubq
1811
1812 { .ISD: ISD::MUL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 18,.CodeSizeCost: 12,.SizeAndLatencyCost: 12 } }, // 2*unpack/2*pmullw/2*and/pack
1813 { .ISD: ISD::MUL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pmullw
1814 { .ISD: ISD::MUL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 8, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } }, // 3*pmuludq/4*shuffle
1815 { .ISD: ISD::MUL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 10,.CodeSizeCost: 10,.SizeAndLatencyCost: 10 } }, // 3*pmuludq/3*shift/2*add
1816
1817 { .ISD: X86ISD::PMULUDQ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1818
1819 { .ISD: ISD::FDIV, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 23, .LatencyCost: 23, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1820 { .ISD: ISD::FDIV, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 39, .LatencyCost: 39, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1821 { .ISD: ISD::FDIV, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 38, .LatencyCost: 38, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1822 { .ISD: ISD::FDIV, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 69, .LatencyCost: 69, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1823
1824 { .ISD: ISD::FNEG, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1825 { .ISD: ISD::FNEG, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1826 { .ISD: ISD::FNEG, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1827 { .ISD: ISD::FNEG, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1828
1829 { .ISD: ISD::FADD, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1830 { .ISD: ISD::FADD, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1831 { .ISD: ISD::FADD, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1832
1833 { .ISD: ISD::FSUB, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1834 { .ISD: ISD::FSUB, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1835 { .ISD: ISD::FSUB, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1836
1837 { .ISD: ISD::FMUL, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1838 { .ISD: ISD::FMUL, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium IV from http://www.agner.org/
1839 };
1840
1841 if (ST->hasSSE2())
1842 if (const auto *Entry = CostTableLookup(Table: SSE2CostTable, ISD, Ty: LT.second))
1843 if (auto KindCost = Entry->Cost[CostKind])
1844 return LT.first * *KindCost;
1845
1846 static const CostKindTblEntry SSE1CostTable[] = {
1847 { .ISD: ISD::FDIV, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 18, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium III from http://www.agner.org/
1848 { .ISD: ISD::FDIV, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 34, .LatencyCost: 48, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium III from http://www.agner.org/
1849
1850 { .ISD: ISD::FNEG, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // Pentium III from http://www.agner.org/
1851 { .ISD: ISD::FNEG, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // Pentium III from http://www.agner.org/
1852
1853 { .ISD: ISD::FADD, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium III from http://www.agner.org/
1854 { .ISD: ISD::FADD, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium III from http://www.agner.org/
1855
1856 { .ISD: ISD::FSUB, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium III from http://www.agner.org/
1857 { .ISD: ISD::FSUB, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium III from http://www.agner.org/
1858
1859 { .ISD: ISD::FMUL, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium III from http://www.agner.org/
1860 { .ISD: ISD::FMUL, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Pentium III from http://www.agner.org/
1861 };
1862
1863 if (ST->hasSSE1())
1864 if (const auto *Entry = CostTableLookup(Table: SSE1CostTable, ISD, Ty: LT.second))
1865 if (auto KindCost = Entry->Cost[CostKind])
1866 return LT.first * *KindCost;
1867
1868 static const CostKindTblEntry X64CostTbl[] = { // 64-bit targets
1869 { .ISD: ISD::ADD, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1 } }, // Core (Merom) from http://www.agner.org/
1870 { .ISD: ISD::SUB, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1 } }, // Core (Merom) from http://www.agner.org/
1871 { .ISD: ISD::MUL, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
1872 };
1873
1874 if (ST->is64Bit())
1875 if (const auto *Entry = CostTableLookup(Table: X64CostTbl, ISD, Ty: LT.second))
1876 if (auto KindCost = Entry->Cost[CostKind])
1877 return LT.first * *KindCost;
1878
1879 static const CostKindTblEntry X86CostTbl[] = { // 32 or 64-bit targets
1880 { .ISD: ISD::ADD, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 1 } }, // Pentium III from http://www.agner.org/
1881 { .ISD: ISD::ADD, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1 } }, // Pentium III from http://www.agner.org/
1882 { .ISD: ISD::ADD, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1 } }, // Pentium III from http://www.agner.org/
1883
1884 { .ISD: ISD::SUB, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 1 } }, // Pentium III from http://www.agner.org/
1885 { .ISD: ISD::SUB, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1 } }, // Pentium III from http://www.agner.org/
1886 { .ISD: ISD::SUB, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1 } }, // Pentium III from http://www.agner.org/
1887
1888 { .ISD: ISD::MUL, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1889 { .ISD: ISD::MUL, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1890 { .ISD: ISD::MUL, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
1891
1892 { .ISD: ISD::FNEG, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // (x87)
1893 { .ISD: ISD::FADD, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // (x87)
1894 { .ISD: ISD::FSUB, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // (x87)
1895 { .ISD: ISD::FMUL, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // (x87)
1896 { .ISD: ISD::FDIV, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 38, .LatencyCost: 38, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // (x87)
1897 };
1898
1899 if (const auto *Entry = CostTableLookup(Table: X86CostTbl, ISD, Ty: LT.second))
1900 if (auto KindCost = Entry->Cost[CostKind])
1901 return LT.first * *KindCost;
1902
1903 // It is not a good idea to vectorize division. We have to scalarize it and
1904 // in the process we will often end up having to spilling regular
1905 // registers. The overhead of division is going to dominate most kernels
1906 // anyways so try hard to prevent vectorization of division - it is
1907 // generally a bad idea. Assume somewhat arbitrarily that we have to be able
1908 // to hide "20 cycles" for each lane.
1909 if (CostKind == TTI::TCK_RecipThroughput && LT.second.isVector() &&
1910 (ISD == ISD::SDIV || ISD == ISD::SREM || ISD == ISD::UDIV ||
1911 ISD == ISD::UREM)) {
1912 InstructionCost ScalarCost =
1913 getArithmeticInstrCost(Opcode, Ty: Ty->getScalarType(), CostKind,
1914 Op1Info: Op1Info.getNoProps(), Op2Info: Op2Info.getNoProps());
1915 return 20 * LT.first * LT.second.getVectorNumElements() * ScalarCost;
1916 }
1917
1918 // Handle some basic single instruction code size cases.
1919 if (CostKind == TTI::TCK_CodeSize) {
1920 switch (ISD) {
1921 case ISD::FADD:
1922 case ISD::FSUB:
1923 case ISD::FMUL:
1924 case ISD::FDIV:
1925 case ISD::FNEG:
1926 case ISD::AND:
1927 case ISD::OR:
1928 case ISD::XOR:
1929 return LT.first;
1930 break;
1931 }
1932 }
1933
1934 // Fallback to the default implementation.
1935 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Opd1Info: Op1Info, Opd2Info: Op2Info,
1936 Args, CtxI);
1937}
1938
1939InstructionCost
1940X86TTIImpl::getAltInstrCost(VectorType *VecTy, unsigned Opcode0,
1941 unsigned Opcode1, const SmallBitVector &OpcodeMask,
1942 TTI::TargetCostKind CostKind,
1943 ArrayRef<const Value *> Scalars) const {
1944 if (isLegalAltInstr(VecTy, Opcode0, Opcode1, OpcodeMask, Scalars))
1945 return TTI::TCC_Basic;
1946 return InstructionCost::getInvalid();
1947}
1948
1949InstructionCost X86TTIImpl::getShuffleCost(
1950 TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy,
1951 TTI::TargetCostKind CostKind, ArrayRef<int> Mask, int Index,
1952 VectorType *SubTp, ArrayRef<const Value *> Args, const Instruction *CtxI,
1953 TTI::VectorInstrContext VIC) const {
1954 assert((Mask.empty() || DstTy->isScalableTy() ||
1955 Mask.size() == DstTy->getElementCount().getKnownMinValue()) &&
1956 "Expected the Mask to match the return size if given");
1957 assert(SrcTy->getScalarType() == DstTy->getScalarType() &&
1958 "Expected the same scalar types");
1959
1960 // 64-bit packed float vectors (v2f32) are widened to type v4f32.
1961 // 64-bit packed integer vectors (v2i32) are widened to type v4i32.
1962 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: SrcTy);
1963
1964 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTy&: SubTp);
1965
1966 // If all args are constant than this will be constant folded away.
1967 if (!Args.empty() &&
1968 all_of(Range&: Args, P: [](const Value *Arg) { return isa<Constant>(Val: Arg); }))
1969 return TTI::TCC_Free;
1970
1971 // Recognize a basic concat_vector shuffle.
1972 if (Kind == TTI::SK_PermuteTwoSrc &&
1973 Mask.size() == (2 * SrcTy->getElementCount().getKnownMinValue()) &&
1974 ShuffleVectorInst::isIdentityMask(Mask, NumSrcElts: Mask.size()))
1975 return getShuffleCost(Kind: TTI::SK_InsertSubvector,
1976 DstTy: VectorType::getDoubleElementsVectorType(VTy: SrcTy),
1977 SrcTy: VectorType::getDoubleElementsVectorType(VTy: SrcTy),
1978 CostKind, Mask, Index: Mask.size() / 2, SubTp: SrcTy);
1979
1980 // Treat Transpose as 2-op shuffles - there's no difference in lowering.
1981 if (Kind == TTI::SK_Transpose)
1982 if (LT.second != MVT::v4f64 && LT.second != MVT::v4i64)
1983 Kind = TTI::SK_PermuteTwoSrc;
1984
1985 if (Kind == TTI::SK_Broadcast) {
1986 // For Broadcasts we are splatting the first element from the first input
1987 // register, so only need to reference that input and all the output
1988 // registers are the same.
1989 LT.first = 1;
1990
1991 // If we're broadcasting a load then AVX/AVX2 can do this for free.
1992 // If many-used-load whose every use is one of a small set of operations
1993 // that SLP can rewrite into a single vector lane, codegen can fold it into
1994 // the free broadcast.
1995 using namespace PatternMatch;
1996 auto IsBroadcastLoadFoldUser = [&](const User *U) {
1997 if (isa<InsertElementInst>(Val: U) && U->getOperand(i: 1) == Args[0])
1998 return true;
1999 if (U->getType()->isVectorTy())
2000 return false;
2001 // Terminators (return/branch/switch/indirectbr/resume/invoke EH)
2002 // and phis carry the value across control flow.
2003 if (const auto *I = dyn_cast<Instruction>(Val: U))
2004 if (I->isTerminator() ||
2005 isa<PHINode, AtomicRMWInst, AtomicCmpXchgInst>(Val: I))
2006 return false;
2007 // Only pure calls can be folded.
2008 if (const auto *CB = dyn_cast<CallBase>(Val: U))
2009 return CB->doesNotAccessMemory() && !CB->mayHaveSideEffects();
2010 return true;
2011 };
2012 auto IsFoldableSLPBroadcastLoad = [&]() {
2013 if (!match(V: Args[0], P: m_Load(Op: m_Value())))
2014 return false;
2015 auto *FVT = dyn_cast<FixedVectorType>(Val: DstTy);
2016 if (!FVT)
2017 return false;
2018 // getNumUses() counts each Use, matching the per-lane broadcast
2019 // accounting (a use like `op %x, %x` consumes two broadcast lanes).
2020 if (Args[0]->getNumUses() != FVT->getNumElements())
2021 return false;
2022 return all_of(Range: Args[0]->users(), P: IsBroadcastLoadFoldUser);
2023 };
2024 if (!Args.empty() &&
2025 (match(V: Args[0], P: m_OneUse(SubPattern: m_Load(Op: m_Value()))) ||
2026 IsFoldableSLPBroadcastLoad()) &&
2027 (ST->hasAVX2() ||
2028 (ST->hasAVX() && LT.second.getScalarSizeInBits() >= 32)))
2029 return TTI::TCC_Free;
2030 }
2031
2032 // Attempt to detect a cheaper inlane shuffle, avoiding 128-bit subvector
2033 // permutation.
2034 // Attempt to detect a shuffle mask with a single defined element.
2035 bool IsInLaneShuffle = false;
2036 bool IsSingleElementMask = false;
2037 if (SrcTy->getPrimitiveSizeInBits() > 0 &&
2038 (SrcTy->getPrimitiveSizeInBits() % 128) == 0 &&
2039 SrcTy->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
2040 Mask.size() == SrcTy->getElementCount().getKnownMinValue()) {
2041 unsigned NumLanes = SrcTy->getPrimitiveSizeInBits() / 128;
2042 unsigned NumEltsPerLane = Mask.size() / NumLanes;
2043 if ((Mask.size() % NumLanes) == 0) {
2044 IsInLaneShuffle = all_of(Range: enumerate(First&: Mask), P: [&](const auto &P) {
2045 return P.value() == PoisonMaskElem ||
2046 ((P.value() % Mask.size()) / NumEltsPerLane) ==
2047 (P.index() / NumEltsPerLane);
2048 });
2049 IsSingleElementMask =
2050 (Mask.size() - 1) == static_cast<unsigned>(count_if(Range&: Mask, P: [](int M) {
2051 return M == PoisonMaskElem;
2052 }));
2053 }
2054 }
2055
2056 // Treat <X x bfloat> shuffles as <X x half>.
2057 if (LT.second.isVectorOf(EltVT: MVT::bf16))
2058 LT.second = LT.second.changeVectorElementType(EltVT: MVT::f16);
2059
2060 // Subvector extractions are free if they start at the beginning of a
2061 // vector and cheap if the subvectors are aligned.
2062 if (Kind == TTI::SK_ExtractSubvector && LT.second.isVector()) {
2063 int NumElts = LT.second.getVectorNumElements();
2064 if ((Index % NumElts) == 0)
2065 return TTI::TCC_Free;
2066 std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(Ty: SubTp);
2067 if (SubLT.second.isVector()) {
2068 int NumSubElts = SubLT.second.getVectorNumElements();
2069 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
2070 return SubLT.first;
2071 // Handle some cases for widening legalization. For now we only handle
2072 // cases where the original subvector was naturally aligned and evenly
2073 // fit in its legalized subvector type.
2074 // FIXME: Remove some of the alignment restrictions.
2075 // FIXME: We can use permq for 64-bit or larger extracts from 256-bit
2076 // vectors.
2077 int OrigSubElts = cast<FixedVectorType>(Val: SubTp)->getNumElements();
2078 if (NumSubElts > OrigSubElts && (Index % OrigSubElts) == 0 &&
2079 (NumSubElts % OrigSubElts) == 0 &&
2080 LT.second.getVectorElementType() ==
2081 SubLT.second.getVectorElementType() &&
2082 LT.second.getVectorElementType().getSizeInBits() ==
2083 SrcTy->getElementType()->getPrimitiveSizeInBits()) {
2084 assert(NumElts >= NumSubElts && NumElts > OrigSubElts &&
2085 "Unexpected number of elements!");
2086 auto *VecTy = FixedVectorType::get(ElementType: SrcTy->getElementType(),
2087 NumElts: LT.second.getVectorNumElements());
2088 auto *SubTy = FixedVectorType::get(ElementType: SrcTy->getElementType(),
2089 NumElts: SubLT.second.getVectorNumElements());
2090 int ExtractIndex = alignDown(Value: (Index % NumElts), Align: NumSubElts);
2091 InstructionCost ExtractCost =
2092 getShuffleCost(Kind: TTI::SK_ExtractSubvector, DstTy: VecTy, SrcTy: VecTy, CostKind, Mask: {},
2093 Index: ExtractIndex, SubTp: SubTy);
2094
2095 // If the original size is 32-bits or more, we can use pshufd. Otherwise
2096 // if we have SSSE3 we can use pshufb.
2097 if (SubTp->getPrimitiveSizeInBits() >= 32 || ST->hasSSSE3())
2098 return ExtractCost + 1; // pshufd or pshufb
2099
2100 assert(SubTp->getPrimitiveSizeInBits() == 16 &&
2101 "Unexpected vector size");
2102
2103 return ExtractCost + 2; // worst case pshufhw + pshufd
2104 }
2105 }
2106 // If the extract subvector is not optimal, treat it as single op shuffle.
2107 Kind = TTI::SK_PermuteSingleSrc;
2108 }
2109
2110 // Subvector insertions are cheap if the subvectors are aligned.
2111 // Note that in general, the insertion starting at the beginning of a vector
2112 // isn't free, because we need to preserve the rest of the wide vector,
2113 // but if the destination vector legalizes to the same width as the subvector
2114 // then the insertion will simplify to a (free) register copy.
2115 if (Kind == TTI::SK_InsertSubvector && LT.second.isVector()) {
2116 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Ty: DstTy);
2117 int NumElts = DstLT.second.getVectorNumElements();
2118 std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(Ty: SubTp);
2119 if (SubLT.second.isVector()) {
2120 int NumSubElts = SubLT.second.getVectorNumElements();
2121 bool MatchingTypes =
2122 NumElts == NumSubElts &&
2123 (SubTp->getElementCount().getKnownMinValue() % NumSubElts) == 0;
2124 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
2125 return MatchingTypes ? TTI::TCC_Free : SubLT.first;
2126 }
2127
2128 // Attempt to match MOVSS (Idx == 0) or INSERTPS pattern. This will have
2129 // been matched by improveShuffleKindFromMask as a SK_InsertSubvector of
2130 // v1f32 (legalised to f32) into a v4f32.
2131 if (LT.first == 1 && LT.second == MVT::v4f32 && SubLT.first == 1 &&
2132 SubLT.second == MVT::f32 && (Index == 0 || ST->hasSSE41()))
2133 return 1;
2134
2135 // If the insertion is the lowest subvector then it will be blended
2136 // otherwise treat it like a 2-op shuffle.
2137 Kind =
2138 (Index == 0 && LT.first == 1) ? TTI::SK_Select : TTI::SK_PermuteTwoSrc;
2139 }
2140
2141 // Handle some common (illegal) sub-vector types as they are often very cheap
2142 // to shuffle even on targets without PSHUFB.
2143 EVT VT = TLI->getValueType(DL, Ty: SrcTy);
2144 if (VT.isSimple() && VT.isVector() && VT.getSizeInBits() < 128 &&
2145 !ST->hasSSSE3()) {
2146 static const CostKindTblEntry SSE2SubVectorShuffleTbl[] = {
2147 {.ISD: TTI::SK_Broadcast, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pshuflw
2148 {.ISD: TTI::SK_Broadcast, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pshuflw
2149 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck/pshuflw
2150 {.ISD: TTI::SK_Broadcast, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck/pshuflw
2151 {.ISD: TTI::SK_Broadcast, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // punpck
2152
2153 {.ISD: TTI::SK_Reverse, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pshuflw
2154 {.ISD: TTI::SK_Reverse, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pshuflw
2155 {.ISD: TTI::SK_Reverse, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 3}}, // punpck/pshuflw/packus
2156 {.ISD: TTI::SK_Reverse, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // punpck
2157
2158 {.ISD: TTI::SK_Splice, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck+psrldq
2159 {.ISD: TTI::SK_Splice, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck+psrldq
2160 {.ISD: TTI::SK_Splice, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck+psrldq
2161 {.ISD: TTI::SK_Splice, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck+psrldq
2162
2163 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck/pshuflw
2164 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck/pshuflw
2165 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 7,.LatencyCost: 7,.CodeSizeCost: 7,.SizeAndLatencyCost: 7}}, // punpck/pshuflw
2166 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 4,.CodeSizeCost: 4,.SizeAndLatencyCost: 4}}, // punpck/pshuflw
2167 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // punpck
2168
2169 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pshuflw
2170 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pshuflw
2171 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 5,.CodeSizeCost: 5,.SizeAndLatencyCost: 5}}, // punpck/pshuflw
2172 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 3}}, // punpck/pshuflw
2173 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // punpck
2174 };
2175
2176 if (ST->hasSSE2())
2177 if (const auto *Entry =
2178 CostTableLookup(Table: SSE2SubVectorShuffleTbl, ISD: Kind, Ty: VT.getSimpleVT()))
2179 if (auto KindCost = Entry->Cost[CostKind])
2180 return LT.first * *KindCost;
2181 }
2182
2183 // We are going to permute multiple sources and the result will be in multiple
2184 // destinations. Providing an accurate cost only for splits where the element
2185 // type remains the same.
2186 if (LT.first != 1) {
2187 MVT LegalVT = LT.second;
2188 if (LegalVT.isVector() &&
2189 LegalVT.getVectorElementType().getSizeInBits() ==
2190 SrcTy->getElementType()->getPrimitiveSizeInBits() &&
2191 LegalVT.getVectorNumElements() <
2192 cast<FixedVectorType>(Val: SrcTy)->getNumElements()) {
2193 unsigned VecTySize = DL.getTypeStoreSize(Ty: SrcTy);
2194 unsigned LegalVTSize = LegalVT.getStoreSize();
2195 // Number of source vectors after legalization:
2196 unsigned NumOfSrcs = (VecTySize + LegalVTSize - 1) / LegalVTSize;
2197 // Number of destination vectors after legalization:
2198 InstructionCost NumOfDests = LT.first;
2199
2200 auto *SingleOpTy = FixedVectorType::get(ElementType: SrcTy->getElementType(),
2201 NumElts: LegalVT.getVectorNumElements());
2202
2203 if (!Mask.empty() && NumOfDests.isValid()) {
2204 // Try to perform better estimation of the permutation.
2205 // 1. Split the source/destination vectors into real registers.
2206 // 2. Do the mask analysis to identify which real registers are
2207 // permuted. If more than 1 source registers are used for the
2208 // destination register building, the cost for this destination register
2209 // is (Number_of_source_register - 1) * Cost_PermuteTwoSrc. If only one
2210 // source register is used, build mask and calculate the cost as a cost
2211 // of PermuteSingleSrc.
2212 // Also, for the single register permute we try to identify if the
2213 // destination register is just a copy of the source register or the
2214 // copy of the previous destination register (the cost is
2215 // TTI::TCC_Basic). If the source register is just reused, the cost for
2216 // this operation is TTI::TCC_Free.
2217 NumOfDests =
2218 getTypeLegalizationCost(
2219 Ty: FixedVectorType::get(ElementType: SrcTy->getElementType(), NumElts: Mask.size()))
2220 .first;
2221 unsigned E = NumOfDests.getValue();
2222 unsigned NormalizedVF =
2223 LegalVT.getVectorNumElements() * std::max(a: NumOfSrcs, b: E);
2224 unsigned NumOfSrcRegs = NormalizedVF / LegalVT.getVectorNumElements();
2225 unsigned NumOfDestRegs = NormalizedVF / LegalVT.getVectorNumElements();
2226 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
2227 copy(Range&: Mask, Out: NormalizedMask.begin());
2228 unsigned PrevSrcReg = 0;
2229 ArrayRef<int> PrevRegMask;
2230 InstructionCost Cost = 0;
2231 processShuffleMasks(
2232 Mask: NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfUsedRegs: NumOfDestRegs, NoInputAction: []() {},
2233 SingleInputAction: [this, SingleOpTy, CostKind, &PrevSrcReg, &PrevRegMask,
2234 &Cost](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
2235 if (!ShuffleVectorInst::isIdentityMask(Mask: RegMask, NumSrcElts: RegMask.size())) {
2236 // Check if the previous register can be just copied to the next
2237 // one.
2238 if (PrevRegMask.empty() || PrevSrcReg != SrcReg ||
2239 PrevRegMask != RegMask)
2240 Cost +=
2241 getShuffleCost(Kind: TTI::SK_PermuteSingleSrc, DstTy: SingleOpTy,
2242 SrcTy: SingleOpTy, CostKind, Mask: RegMask, Index: 0, SubTp: nullptr);
2243 else
2244 // Just a copy of previous destination register.
2245 Cost += TTI::TCC_Basic;
2246 return;
2247 }
2248 if (SrcReg != DestReg &&
2249 any_of(Range&: RegMask, P: not_equal_to(Arg: PoisonMaskElem))) {
2250 // Just a copy of the source register.
2251 Cost += TTI::TCC_Free;
2252 }
2253 PrevSrcReg = SrcReg;
2254 PrevRegMask = RegMask;
2255 },
2256 ManyInputsAction: [this, SingleOpTy, CostKind,
2257 &Cost](ArrayRef<int> RegMask, unsigned /*Unused*/,
2258 unsigned /*Unused*/, bool /*Unused*/) {
2259 Cost += getShuffleCost(Kind: TTI::SK_PermuteTwoSrc, DstTy: SingleOpTy,
2260 SrcTy: SingleOpTy, CostKind, Mask: RegMask, Index: 0, SubTp: nullptr);
2261 });
2262 return Cost;
2263 }
2264
2265 InstructionCost NumOfShuffles = (NumOfSrcs - 1) * NumOfDests;
2266 return NumOfShuffles * getShuffleCost(Kind: TTI::SK_PermuteTwoSrc, DstTy: SingleOpTy,
2267 SrcTy: SingleOpTy, CostKind, Mask: {}, Index: 0,
2268 SubTp: nullptr);
2269 }
2270
2271 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
2272 SubTp);
2273 }
2274
2275 // If we're just moving a single element around (probably as an alternative to
2276 // extracting it), we can assume this is cheap.
2277 if (LT.first == 1 && IsInLaneShuffle && IsSingleElementMask)
2278 return TTI::TCC_Basic;
2279
2280 static const CostKindTblEntry AVX512VBMIShuffleTbl[] = {
2281 { .ISD: TTI::SK_Reverse, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermb
2282 { .ISD: TTI::SK_Reverse, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermb
2283 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermb
2284 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermb
2285 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermt2b
2286 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermt2b
2287 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } } // vpermt2b
2288 };
2289
2290 if (ST->hasVBMI())
2291 if (const auto *Entry =
2292 CostTableLookup(Table: AVX512VBMIShuffleTbl, ISD: Kind, Ty: LT.second))
2293 if (auto KindCost = Entry->Cost[CostKind])
2294 return LT.first * *KindCost;
2295
2296 static const CostKindTblEntry AVX512BWShuffleTbl[] = {
2297 { .ISD: TTI::SK_Broadcast, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastw
2298 { .ISD: TTI::SK_Broadcast, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastw
2299 { .ISD: TTI::SK_Broadcast, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastb
2300
2301 { .ISD: TTI::SK_Reverse, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // vpermw
2302 { .ISD: TTI::SK_Reverse, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // vpermw
2303 { .ISD: TTI::SK_Reverse, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermw
2304 { .ISD: TTI::SK_Reverse, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermw
2305 { .ISD: TTI::SK_Reverse, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // pshufb + vshufi64x2
2306
2307 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermw
2308 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermw
2309 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermw
2310 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermw
2311 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 8, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } }, // extend to v32i16
2312
2313 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32i16,.Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermt2w
2314 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32f16,.Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermt2w
2315 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i16,.Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermt2w
2316 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vpermt2w
2317 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 19, .LatencyCost: 19, .CodeSizeCost: 19, .SizeAndLatencyCost: 19 } }, // 6 * v32i8 + 1
2318
2319 { .ISD: TTI::SK_Select, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vblendmw
2320 { .ISD: TTI::SK_Select, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vblendmb
2321
2322 { .ISD: TTI::SK_Splice, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vshufi64x2 + palignr
2323 { .ISD: TTI::SK_Splice, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vshufi64x2 + palignr
2324 { .ISD: TTI::SK_Splice, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vshufi64x2 + palignr
2325 };
2326
2327 if (ST->hasBWI())
2328 if (const auto *Entry =
2329 CostTableLookup(Table: AVX512BWShuffleTbl, ISD: Kind, Ty: LT.second))
2330 if (auto KindCost = Entry->Cost[CostKind])
2331 return LT.first * *KindCost;
2332
2333 static const CostKindTblEntry AVX512InLaneShuffleTbl[] = {
2334 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2335 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2336 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2337 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2338 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2339 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2340 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2341 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2342 };
2343
2344 if (IsInLaneShuffle && ST->hasAVX512())
2345 if (const auto *Entry =
2346 CostTableLookup(Table: AVX512InLaneShuffleTbl, ISD: Kind, Ty: LT.second))
2347 if (auto KindCost = Entry->Cost[CostKind])
2348 return LT.first * *KindCost;
2349
2350 static const CostKindTblEntry AVX512ShuffleTbl[] = {
2351 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vbroadcastsd
2352 {.ISD: TTI::SK_Broadcast, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vbroadcastsd
2353 {.ISD: TTI::SK_Broadcast, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vbroadcastss
2354 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vbroadcastss
2355 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastq
2356 {.ISD: TTI::SK_Broadcast, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastq
2357 {.ISD: TTI::SK_Broadcast, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastd
2358 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastd
2359 {.ISD: TTI::SK_Broadcast, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastw
2360 {.ISD: TTI::SK_Broadcast, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastw
2361 {.ISD: TTI::SK_Broadcast, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastw
2362 {.ISD: TTI::SK_Broadcast, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastw
2363 {.ISD: TTI::SK_Broadcast, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastb
2364 {.ISD: TTI::SK_Broadcast, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 }}, // vpbroadcastb
2365
2366 {.ISD: TTI::SK_Reverse, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // vpermpd
2367 {.ISD: TTI::SK_Reverse, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // vpermps
2368 {.ISD: TTI::SK_Reverse, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // vpermq
2369 {.ISD: TTI::SK_Reverse, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // vpermd
2370 {.ISD: TTI::SK_Reverse, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } }, // per mca
2371 {.ISD: TTI::SK_Reverse, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } }, // per mca
2372 {.ISD: TTI::SK_Reverse, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } }, // per mca
2373
2374 {.ISD: TTI::SK_Splice, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpalignd
2375 {.ISD: TTI::SK_Splice, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpalignd
2376 {.ISD: TTI::SK_Splice, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpalignd
2377 {.ISD: TTI::SK_Splice, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpalignd
2378 {.ISD: TTI::SK_Splice, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpalignd
2379 {.ISD: TTI::SK_Splice, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpalignd
2380 {.ISD: TTI::SK_Splice, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpalignd
2381 {.ISD: TTI::SK_Splice, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpalignd
2382 {.ISD: TTI::SK_Splice, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // split + palignr
2383 {.ISD: TTI::SK_Splice, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // split + palignr
2384 {.ISD: TTI::SK_Splice, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // split + palignr
2385
2386 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermpd
2387 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermpd
2388 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermpd
2389 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermps
2390 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermps
2391 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermps
2392 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermq
2393 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermq
2394 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermq
2395 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermd
2396 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermd
2397 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermd
2398 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // pshufb
2399
2400 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermt2pd
2401 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermt2ps
2402 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermt2q
2403 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermt2d
2404 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermt2pd
2405 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermt2ps
2406 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermt2q
2407 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermt2d
2408 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2409 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2410 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2411 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2412
2413 // FIXME: This just applies the type legalization cost rules above
2414 // assuming these completely split.
2415 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
2416 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
2417 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 14, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
2418 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 42, .LatencyCost: 42, .CodeSizeCost: 42, .SizeAndLatencyCost: 42 } },
2419 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 42, .LatencyCost: 42, .CodeSizeCost: 42, .SizeAndLatencyCost: 42 } },
2420 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 42, .LatencyCost: 42, .CodeSizeCost: 42, .SizeAndLatencyCost: 42 } },
2421
2422 {.ISD: TTI::SK_Select, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq
2423 {.ISD: TTI::SK_Select, .Type: MVT::v32f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq
2424 {.ISD: TTI::SK_Select, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq
2425 {.ISD: TTI::SK_Select, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vblendmpd
2426 {.ISD: TTI::SK_Select, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vblendmps
2427 {.ISD: TTI::SK_Select, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vblendmq
2428 {.ISD: TTI::SK_Select, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vblendmd
2429 };
2430
2431 if (ST->hasAVX512())
2432 if (const auto *Entry = CostTableLookup(Table: AVX512ShuffleTbl, ISD: Kind, Ty: LT.second))
2433 if (auto KindCost = Entry->Cost[CostKind])
2434 return LT.first * *KindCost;
2435
2436 static const CostKindTblEntry AVX2InLaneShuffleTbl[] = {
2437 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpshufb
2438 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpshufb
2439 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpshufb
2440
2441 { .ISD: TTI::SK_Transpose, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vshufpd/vunpck
2442 { .ISD: TTI::SK_Transpose, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vshufpd/vunpck
2443
2444 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vshufpd + vblendpd
2445 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vshufps + vblendps
2446 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vpshufd + vpblendd
2447 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vpshufd + vpblendd
2448 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vpshufb + vpor
2449 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vpshufb + vpor
2450 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vpshufb + vpor
2451 };
2452
2453 if (IsInLaneShuffle && ST->hasAVX2())
2454 if (const auto *Entry =
2455 CostTableLookup(Table: AVX2InLaneShuffleTbl, ISD: Kind, Ty: LT.second))
2456 if (auto KindCost = Entry->Cost[CostKind])
2457 return LT.first * *KindCost;
2458
2459 static const CostKindTblEntry AVX2ShuffleTbl[] = {
2460 { .ISD: TTI::SK_Broadcast, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vbroadcastpd
2461 { .ISD: TTI::SK_Broadcast, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vbroadcastps
2462 { .ISD: TTI::SK_Broadcast, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpbroadcastq
2463 { .ISD: TTI::SK_Broadcast, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpbroadcastd
2464 { .ISD: TTI::SK_Broadcast, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpbroadcastw
2465 { .ISD: TTI::SK_Broadcast, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastw
2466 { .ISD: TTI::SK_Broadcast, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpbroadcastw
2467 { .ISD: TTI::SK_Broadcast, .Type: MVT::v8f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastw
2468 { .ISD: TTI::SK_Broadcast, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpbroadcastb
2469 { .ISD: TTI::SK_Broadcast, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpbroadcastb
2470
2471 { .ISD: TTI::SK_Reverse, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpermpd
2472 { .ISD: TTI::SK_Reverse, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // vpermps
2473 { .ISD: TTI::SK_Reverse, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vpermq
2474 { .ISD: TTI::SK_Reverse, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // vpermd
2475 { .ISD: TTI::SK_Reverse, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // vperm2i128 + pshufb
2476 { .ISD: TTI::SK_Reverse, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // vperm2i128 + pshufb
2477 { .ISD: TTI::SK_Reverse, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // vperm2i128 + pshufb
2478
2479 { .ISD: TTI::SK_Select, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpblendvb
2480 { .ISD: TTI::SK_Select, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpblendvb
2481 { .ISD: TTI::SK_Select, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpblendvb
2482
2483 { .ISD: TTI::SK_Splice, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2i128 + vpalignr
2484 { .ISD: TTI::SK_Splice, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2i128 + vpalignr
2485 { .ISD: TTI::SK_Splice, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2i128 + vpalignr
2486 { .ISD: TTI::SK_Splice, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2i128 + vpalignr
2487 { .ISD: TTI::SK_Splice, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2i128 + vpalignr
2488
2489 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermpd
2490 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermps
2491 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermq
2492 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermd
2493 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
2494 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
2495 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
2496
2497 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // 2*vpermpd + vblendpd
2498 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // 2*vpermps + vblendps
2499 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // 2*vpermq + vpblendd
2500 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // 2*vpermd + vpblendd
2501 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
2502 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
2503 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
2504 };
2505
2506 if (ST->hasAVX2())
2507 if (const auto *Entry = CostTableLookup(Table: AVX2ShuffleTbl, ISD: Kind, Ty: LT.second))
2508 if (auto KindCost = Entry->Cost[CostKind])
2509 return LT.first * *KindCost;
2510
2511 static const CostKindTblEntry XOPShuffleTbl[] = {
2512 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2f128 + vpermil2pd
2513 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2f128 + vpermil2ps
2514 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2f128 + vpermil2pd
2515 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // vperm2f128 + vpermil2ps
2516 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i16,.Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // vextractf128 + 2*vpperm
2517 // + vinsertf128
2518 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // vextractf128 + 2*vpperm
2519 // + vinsertf128
2520
2521 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 9, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } }, // 2*vextractf128 + 6*vpperm
2522 // + vinsertf128
2523
2524 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpperm
2525 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 9, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } }, // 2*vextractf128 + 6*vpperm
2526 // + vinsertf128
2527 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpperm
2528 };
2529
2530 if (ST->hasXOP())
2531 if (const auto *Entry = CostTableLookup(Table: XOPShuffleTbl, ISD: Kind, Ty: LT.second))
2532 if (auto KindCost = Entry->Cost[CostKind])
2533 return LT.first * *KindCost;
2534
2535 static const CostKindTblEntry AVX1InLaneShuffleTbl[] = {
2536 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermilpd
2537 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermilpd
2538 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermilps
2539 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpermilps
2540
2541 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // vextractf128 + 2*pshufb
2542 // + vpor + vinsertf128
2543 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // vextractf128 + 2*pshufb
2544 // + vpor + vinsertf128
2545 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } }, // vextractf128 + 2*pshufb
2546 // + vpor + vinsertf128
2547
2548 { .ISD: TTI::SK_Transpose, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vshufpd/vunpck
2549 { .ISD: TTI::SK_Transpose, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vshufpd/vunpck
2550
2551 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vshufpd + vblendpd
2552 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vshufps + vblendps
2553 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vpermilpd + vblendpd
2554 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // 2*vpermilps + vblendps
2555 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 9, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } }, // 2*vextractf128 + 4*pshufb
2556 // + 2*vpor + vinsertf128
2557 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16f16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 9, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } }, // 2*vextractf128 + 4*pshufb
2558 // + 2*vpor + vinsertf128
2559 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 9, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } }, // 2*vextractf128 + 4*pshufb
2560 // + 2*vpor + vinsertf128
2561 };
2562
2563 if (IsInLaneShuffle && ST->hasAVX())
2564 if (const auto *Entry =
2565 CostTableLookup(Table: AVX1InLaneShuffleTbl, ISD: Kind, Ty: LT.second))
2566 if (auto KindCost = Entry->Cost[CostKind])
2567 return LT.first * *KindCost;
2568
2569 static const CostKindTblEntry AVX1ShuffleTbl[] = {
2570 {.ISD: TTI::SK_Broadcast, .Type: MVT::v4f64, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 3,.CodeSizeCost: 2,.SizeAndLatencyCost: 3}}, // vperm2f128 + vpermilpd
2571 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8f32, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 3,.CodeSizeCost: 2,.SizeAndLatencyCost: 3}}, // vperm2f128 + vpermilps
2572 {.ISD: TTI::SK_Broadcast, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 3,.CodeSizeCost: 2,.SizeAndLatencyCost: 3}}, // vperm2f128 + vpermilpd
2573 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 3,.CodeSizeCost: 2,.SizeAndLatencyCost: 3}}, // vperm2f128 + vpermilps
2574 {.ISD: TTI::SK_Broadcast, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 4}}, // vpshuflw + vpshufd + vinsertf128
2575 {.ISD: TTI::SK_Broadcast, .Type: MVT::v16f16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 4}}, // vpshuflw + vpshufd + vinsertf128
2576 {.ISD: TTI::SK_Broadcast, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 4,.CodeSizeCost: 3,.SizeAndLatencyCost: 6}}, // vpshufb + vinsertf128
2577
2578 {.ISD: TTI::SK_Reverse, .Type: MVT::v4f64, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 6,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // vperm2f128 + vpermilpd
2579 {.ISD: TTI::SK_Reverse, .Type: MVT::v8f32, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 7,.CodeSizeCost: 2,.SizeAndLatencyCost: 4}}, // vperm2f128 + vpermilps
2580 {.ISD: TTI::SK_Reverse, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 6,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // vperm2f128 + vpermilpd
2581 {.ISD: TTI::SK_Reverse, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 7,.CodeSizeCost: 2,.SizeAndLatencyCost: 4}}, // vperm2f128 + vpermilps
2582 {.ISD: TTI::SK_Reverse, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 9,.CodeSizeCost: 5,.SizeAndLatencyCost: 5}}, // vextractf128 + 2*pshufb
2583 // + vinsertf128
2584 {.ISD: TTI::SK_Reverse, .Type: MVT::v16f16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 9,.CodeSizeCost: 5,.SizeAndLatencyCost: 5}}, // vextractf128 + 2*pshufb
2585 // + vinsertf128
2586 {.ISD: TTI::SK_Reverse, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 9,.CodeSizeCost: 5,.SizeAndLatencyCost: 5}}, // vextractf128 + 2*pshufb
2587 // + vinsertf128
2588
2589 {.ISD: TTI::SK_Select, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // vblendpd
2590 {.ISD: TTI::SK_Select, .Type: MVT::v4f64, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // vblendpd
2591 {.ISD: TTI::SK_Select, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // vblendps
2592 {.ISD: TTI::SK_Select, .Type: MVT::v8f32, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // vblendps
2593 {.ISD: TTI::SK_Select, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 3}}, // vpand + vpandn + vpor
2594 {.ISD: TTI::SK_Select, .Type: MVT::v16f16, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 3}}, // vpand + vpandn + vpor
2595 {.ISD: TTI::SK_Select, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 3}}, // vpand + vpandn + vpor
2596
2597 {.ISD: TTI::SK_Splice, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // vperm2f128 + shufpd
2598 {.ISD: TTI::SK_Splice, .Type: MVT::v4f64, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // vperm2f128 + shufpd
2599 {.ISD: TTI::SK_Splice, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 4,.CodeSizeCost: 4,.SizeAndLatencyCost: 4}}, // 2*vperm2f128 + 2*vshufps
2600 {.ISD: TTI::SK_Splice, .Type: MVT::v8f32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 4,.CodeSizeCost: 4,.SizeAndLatencyCost: 4}}, // 2*vperm2f128 + 2*vshufps
2601 {.ISD: TTI::SK_Splice, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 5,.CodeSizeCost: 5,.SizeAndLatencyCost: 5}}, // 2*vperm2f128 + 2*vpalignr + vinsertf128
2602 {.ISD: TTI::SK_Splice, .Type: MVT::v16f16, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 5,.CodeSizeCost: 5,.SizeAndLatencyCost: 5}}, // 2*vperm2f128 + 2*vpalignr + vinsertf128
2603 {.ISD: TTI::SK_Splice, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 5,.CodeSizeCost: 5,.SizeAndLatencyCost: 5}}, // 2*vperm2f128 + 2*vpalignr + vinsertf128
2604
2605 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4f64, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // vperm2f128 + vshufpd
2606 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2}}, // vperm2f128 + vshufpd
2607 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8f32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 4,.CodeSizeCost: 4,.SizeAndLatencyCost: 4}}, // 2*vperm2f128 + 2*vshufps
2608 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 4,.CodeSizeCost: 4,.SizeAndLatencyCost: 4}}, // 2*vperm2f128 + 2*vshufps
2609 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i16,.Cost: {.RecipThroughputCost: 8,.LatencyCost: 8,.CodeSizeCost: 8,.SizeAndLatencyCost: 8}}, // vextractf128 + 4*pshufb
2610 // + 2*por + vinsertf128
2611 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16f16,.Cost: {.RecipThroughputCost: 8,.LatencyCost: 8,.CodeSizeCost: 8,.SizeAndLatencyCost: 8}}, // vextractf128 + 4*pshufb
2612 // + 2*por + vinsertf128
2613 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 8,.LatencyCost: 8,.CodeSizeCost: 8,.SizeAndLatencyCost: 8}}, // vextractf128 + 4*pshufb
2614 // + 2*por + vinsertf128
2615
2616 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4f64, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 3}}, // 2*vperm2f128 + vshufpd
2617 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 3,.CodeSizeCost: 3,.SizeAndLatencyCost: 3}}, // 2*vperm2f128 + vshufpd
2618 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 4,.CodeSizeCost: 4,.SizeAndLatencyCost: 4}}, // 2*vperm2f128 + 2*vshufps
2619 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 4,.CodeSizeCost: 4,.SizeAndLatencyCost: 4}}, // 2*vperm2f128 + 2*vshufps
2620 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i16,.Cost: {.RecipThroughputCost: 15,.LatencyCost: 15,.CodeSizeCost: 15,.SizeAndLatencyCost: 15}}, // 2*vextractf128 + 8*pshufb
2621 // + 4*por + vinsertf128
2622 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16f16,.Cost: {.RecipThroughputCost: 15,.LatencyCost: 15,.CodeSizeCost: 15,.SizeAndLatencyCost: 15}}, // 2*vextractf128 + 8*pshufb
2623 // + 4*por + vinsertf128
2624 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 15,.LatencyCost: 15,.CodeSizeCost: 15,.SizeAndLatencyCost: 15}}, // 2*vextractf128 + 8*pshufb
2625 // + 4*por + vinsertf128
2626 };
2627
2628 if (ST->hasAVX())
2629 if (const auto *Entry = CostTableLookup(Table: AVX1ShuffleTbl, ISD: Kind, Ty: LT.second))
2630 if (auto KindCost = Entry->Cost[CostKind])
2631 return LT.first * *KindCost;
2632
2633 static const CostKindTblEntry SSE41ShuffleTbl[] = {
2634 {.ISD: TTI::SK_Select, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pblendw
2635 {.ISD: TTI::SK_Select, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // movsd
2636 {.ISD: TTI::SK_Select, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pblendw
2637 {.ISD: TTI::SK_Select, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // blendps
2638 {.ISD: TTI::SK_Select, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pblendw
2639 {.ISD: TTI::SK_Select, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}}, // pblendw
2640 {.ISD: TTI::SK_Select, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1}} // pblendvb
2641 };
2642
2643 if (ST->hasSSE41())
2644 if (const auto *Entry = CostTableLookup(Table: SSE41ShuffleTbl, ISD: Kind, Ty: LT.second))
2645 if (auto KindCost = Entry->Cost[CostKind])
2646 return LT.first * *KindCost;
2647
2648 static const CostKindTblEntry SSSE3ShuffleTbl[] = {
2649 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}}, // pshufb
2650 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}}, // pshufb
2651 {.ISD: TTI::SK_Broadcast, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}}, // pshufb
2652
2653 {.ISD: TTI::SK_Reverse, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2}}, // pshufb
2654 {.ISD: TTI::SK_Reverse, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2}}, // pshufb
2655 {.ISD: TTI::SK_Reverse, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2}}, // pshufb
2656
2657 {.ISD: TTI::SK_Splice, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // palignr
2658 {.ISD: TTI::SK_Splice, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // palignr
2659 {.ISD: TTI::SK_Splice, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // palignr
2660 {.ISD: TTI::SK_Splice, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // palignr
2661 {.ISD: TTI::SK_Splice, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // palignr
2662
2663 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufb
2664 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufb
2665 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufb
2666
2667 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // 2*pshufb + por
2668 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // 2*pshufb + por
2669 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // 2*pshufb + por
2670 };
2671
2672 if (ST->hasSSSE3())
2673 if (const auto *Entry = CostTableLookup(Table: SSSE3ShuffleTbl, ISD: Kind, Ty: LT.second))
2674 if (auto KindCost = Entry->Cost[CostKind])
2675 return LT.first * *KindCost;
2676
2677 static const CostKindTblEntry SSE2ShuffleTbl[] = {
2678 {.ISD: TTI::SK_Broadcast, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // shufpd
2679 {.ISD: TTI::SK_Broadcast, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufd
2680 {.ISD: TTI::SK_Broadcast, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufd
2681 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}}, // pshuflw + pshufd
2682 {.ISD: TTI::SK_Broadcast, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}}, // pshuflw + pshufd
2683 {.ISD: TTI::SK_Broadcast, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 4}}, // unpck + pshuflw + pshufd
2684
2685 {.ISD: TTI::SK_Reverse, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // shufpd
2686 {.ISD: TTI::SK_Reverse, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufd
2687 {.ISD: TTI::SK_Reverse, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufd
2688 {.ISD: TTI::SK_Reverse, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // pshuflw + pshufhw + pshufd
2689 {.ISD: TTI::SK_Reverse, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // pshuflw + pshufhw + pshufd
2690 {.ISD: TTI::SK_Reverse, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 5, .LatencyCost: 6,.CodeSizeCost: 11,.SizeAndLatencyCost: 11}}, // 2*pshuflw + 2*pshufhw
2691 // + 2*pshufd + 2*unpck + packus
2692
2693 {.ISD: TTI::SK_Select, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // movsd
2694 {.ISD: TTI::SK_Select, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // movsd
2695 {.ISD: TTI::SK_Select, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}}, // 2*shufps
2696 {.ISD: TTI::SK_Select, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // pand + pandn + por
2697 {.ISD: TTI::SK_Select, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // pand + pandn + por
2698 {.ISD: TTI::SK_Select, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // pand + pandn + por
2699
2700 {.ISD: TTI::SK_Splice, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // shufpd
2701 {.ISD: TTI::SK_Splice, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // shufpd
2702 {.ISD: TTI::SK_Splice, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}}, // 2*{unpck,movsd,pshufd}
2703 {.ISD: TTI::SK_Splice, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // psrldq + psrlldq + por
2704 {.ISD: TTI::SK_Splice, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // psrldq + psrlldq + por
2705 {.ISD: TTI::SK_Splice, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}}, // psrldq + psrlldq + por
2706
2707 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // shufpd
2708 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufd
2709 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // pshufd
2710 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}}, // 2*pshuflw + 2*pshufhw
2711 // + pshufd/unpck
2712 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}}, // 2*pshuflw + 2*pshufhw
2713 // + pshufd/unpck
2714 {.ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 8, .LatencyCost: 10, .CodeSizeCost: 10, .SizeAndLatencyCost: 10}}, // 2*pshuflw + 2*pshufhw
2715 // + 2*pshufd + 2*unpck + 2*packus
2716
2717 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // shufpd
2718 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1}}, // shufpd
2719 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}}, // 2*{unpck,movsd,pshufd}
2720 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 6, .LatencyCost: 8, .CodeSizeCost: 8, .SizeAndLatencyCost: 8}}, // blend+permute
2721 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v8f16, .Cost: {.RecipThroughputCost: 6, .LatencyCost: 8, .CodeSizeCost: 8, .SizeAndLatencyCost: 8}}, // blend+permute
2722 {.ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 11, .LatencyCost: 13, .CodeSizeCost: 13, .SizeAndLatencyCost: 13}}, // blend+permute
2723 };
2724
2725 static const CostTblEntry SSE3BroadcastLoadTbl[] = {
2726 {.ISD: TTI::SK_Broadcast, .Type: MVT::v2f64, .Cost: 0}, // broadcast handled by movddup
2727 };
2728
2729 if (ST->hasSSE2()) {
2730 bool IsLoad =
2731 llvm::any_of(Range&: Args, P: [](const auto &V) { return isa<LoadInst>(V); });
2732 if (ST->hasSSE3() && IsLoad)
2733 if (const auto *Entry =
2734 CostTableLookup(Table: SSE3BroadcastLoadTbl, ISD: Kind, Ty: LT.second)) {
2735 assert(isLegalBroadcastLoad(SrcTy->getElementType(),
2736 LT.second.getVectorElementCount()) &&
2737 "Table entry missing from isLegalBroadcastLoad()");
2738 return LT.first * Entry->Cost;
2739 }
2740
2741 if (const auto *Entry = CostTableLookup(Table: SSE2ShuffleTbl, ISD: Kind, Ty: LT.second))
2742 if (auto KindCost = Entry->Cost[CostKind])
2743 return LT.first * *KindCost;
2744 }
2745
2746 static const CostKindTblEntry SSE1ShuffleTbl[] = {
2747 { .ISD: TTI::SK_Broadcast, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1} }, // shufps
2748 { .ISD: TTI::SK_Reverse, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1} }, // shufps
2749 { .ISD: TTI::SK_Select, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2} }, // 2*shufps
2750 { .ISD: TTI::SK_Splice, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2} }, // 2*shufps
2751 { .ISD: TTI::SK_PermuteSingleSrc, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 1,.LatencyCost: 1,.CodeSizeCost: 1,.SizeAndLatencyCost: 1} }, // shufps
2752 { .ISD: TTI::SK_PermuteTwoSrc, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 2,.CodeSizeCost: 2,.SizeAndLatencyCost: 2} }, // 2*shufps
2753 };
2754
2755 if (ST->hasSSE1()) {
2756 if (LT.first == 1 && LT.second == MVT::v4f32 && Mask.size() == 4) {
2757 // SHUFPS: both pairs must come from the same source register.
2758 auto MatchSHUFPS = [](int X, int Y) {
2759 return X < 0 || Y < 0 || ((X & 4) == (Y & 4));
2760 };
2761 if (MatchSHUFPS(Mask[0], Mask[1]) && MatchSHUFPS(Mask[2], Mask[3]))
2762 return 1;
2763 }
2764 if (const auto *Entry = CostTableLookup(Table: SSE1ShuffleTbl, ISD: Kind, Ty: LT.second))
2765 if (auto KindCost = Entry->Cost[CostKind])
2766 return LT.first * *KindCost;
2767 }
2768
2769 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
2770 SubTp);
2771}
2772
2773InstructionCost X86TTIImpl::getCastInstrCost(unsigned Opcode, Type *Dst,
2774 Type *Src,
2775 TTI::CastContextHint CCH,
2776 TTI::TargetCostKind CostKind,
2777 const Instruction *I) const {
2778 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2779 assert(ISD && "Invalid opcode");
2780
2781 // A narrow (i8/i16) zero-extension used as a GEP *index* can be folded into
2782 // the addressing mode of the consuming memory op, but only if the source is
2783 // already materialised zero-extended in a full register. X86's SIB form
2784 // [base + index*scale + disp] reads the index at full width and does NOT
2785 // zero-extend a narrow index (unlike AArch64's uxtw-extended addressing), so
2786 // a "dirty" narrow source (e.g. an i16 add result used only as an index)
2787 // still needs a dedicated movzx and is not free. Price it as free only with
2788 // positive evidence that no movzx is required.
2789 if (ISD == ISD::ZERO_EXTEND && I && I->hasOneUse() && Src->isIntegerTy() &&
2790 Src->getScalarSizeInBits() < 32) {
2791 const Use &U = *I->use_begin();
2792 if (isa<GetElementPtrInst>(Val: U.getUser()) &&
2793 U.getOperandNo() != GetElementPtrInst::getPointerOperandIndex()) {
2794 const Value *Op = I->getOperand(i: 0);
2795 // Clean sources: an extending load, a zeroext argument, or a value whose
2796 // high bits are provably zero (e.g. from a shift/mask). These mirror the
2797 // proof-based reasoning the middle end uses elsewhere (ValueTracking and
2798 // InstCombine's canEvaluateZExtd); we intentionally do NOT treat a merely
2799 // multiply-used operand as clean, since that is a guess rather than
2800 // proof.
2801 if (isa<LoadInst>(Val: Op))
2802 return TTI::TCC_Free;
2803 if (const auto *A = dyn_cast<Argument>(Val: Op))
2804 if (A->hasAttribute(Kind: Attribute::ZExt))
2805 return TTI::TCC_Free;
2806 if (computeKnownBits(V: Op, DL: I->getDataLayout(), /*AC=*/nullptr, CtxI: I)
2807 .countMinLeadingZeros() > 0)
2808 return TTI::TCC_Free;
2809 }
2810 }
2811
2812 // The cost tables include both specific, custom (non-legal) src/dst type
2813 // conversions and generic, legalized types. We test for customs first, before
2814 // falling back to legalization.
2815 // FIXME: Need a better design of the cost table to handle non-simple types of
2816 // potential massive combinations (elem_num x src_type x dst_type).
2817 static const TypeConversionCostKindTblEntry AVX512BWConversionTbl[]{
2818 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v32i16, .Src: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2819 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v32i16, .Src: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2820
2821 // Mask sign extend has an instruction.
2822 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2823 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2824 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2825 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2826 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2827 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2828 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2829 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2830 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2831 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2832 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2833 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2834 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2835 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v32i8, .Src: MVT::v32i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2836 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v32i16, .Src: MVT::v32i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2837 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v64i8, .Src: MVT::v64i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2838 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v32i16, .Src: MVT::v64i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2839
2840 // Mask zero extend is a sext + shift.
2841 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2842 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2843 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2844 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2845 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2846 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2847 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2848 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2849 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2850 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2851 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2852 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2853 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2854 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v32i8, .Src: MVT::v32i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2855 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v32i16, .Src: MVT::v32i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2856 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v64i8, .Src: MVT::v64i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2857 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v32i16, .Src: MVT::v64i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2858
2859 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2860 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2861 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2862 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2863 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2864 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2865 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2866 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2867 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2868 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2869 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2870 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2871 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2872 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i1, .Src: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2873 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i1, .Src: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2874 { .ISD: ISD::TRUNCATE, .Dst: MVT::v64i1, .Src: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2875 { .ISD: ISD::TRUNCATE, .Dst: MVT::v64i1, .Src: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2876
2877 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i8, .Src: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2878 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // widen to zmm
2879 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i8, .Src: MVT::v2i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovwb
2880 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i8, .Src: MVT::v4i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovwb
2881 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i8, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovwb
2882 };
2883
2884 static const TypeConversionCostKindTblEntry AVX512DQConversionTbl[] = {
2885 // Mask sign extend has an instruction.
2886 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2887 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2888 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2889 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2890 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2891 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2892 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2893 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2894
2895 // Mask zero extend is a sext + shift.
2896 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1, } },
2897 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1, } },
2898 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1, } },
2899 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1, } },
2900 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1, } },
2901 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1, } },
2902 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1, } },
2903 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1, } },
2904
2905 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2906 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2907 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2908 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2909 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2910 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2911 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2912 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2913
2914 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2915 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2916
2917 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2918 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2919
2920 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i64, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2921 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i64, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2922
2923 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i64, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2924 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i64, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2925 };
2926
2927 // TODO: For AVX512DQ + AVX512VL, we also have cheap casts for 128-bit and
2928 // 256-bit wide vectors.
2929
2930 static const TypeConversionCostKindTblEntry AVX512FConversionTbl[] = {
2931 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v8f64, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2932 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v8f64, .Src: MVT::v16f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2933 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v16f64, .Src: MVT::v16f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // 2*vcvtps2pd+vextractf64x4
2934 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v16f32, .Src: MVT::v16f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vcvtph2ps
2935 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v8f64, .Src: MVT::v8f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vcvtph2ps+vcvtps2pd
2936 { .ISD: ISD::FP_ROUND, .Dst: MVT::v8f32, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2937 { .ISD: ISD::FP_ROUND, .Dst: MVT::v16f16, .Src: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vcvtps2ph
2938
2939 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpslld+vptestmd
2940 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpslld+vptestmd
2941 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpslld+vptestmd
2942 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpslld+vptestmd
2943 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpsllq+vptestmq
2944 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpsllq+vptestmq
2945 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpsllq+vptestmq
2946 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpslld+vptestmd
2947 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpslld+vptestmd
2948 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpslld+vptestmd
2949 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpslld+vptestmd
2950 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpslld+vptestmd
2951 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpsllq+vptestmq
2952 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpsllq+vptestmq
2953 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsllq+vptestmq
2954 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i8, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovdb
2955 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i8, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovdb
2956 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovdb
2957 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i8, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovdb
2958 { .ISD: ISD::TRUNCATE, .Dst: MVT::v64i8, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovdb
2959 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i16, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovdw
2960 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i16, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovdw
2961 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i8, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqb
2962 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i16, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpshufb
2963 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i8, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqb
2964 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqb
2965 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i8, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqb
2966 { .ISD: ISD::TRUNCATE, .Dst: MVT::v64i8, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqb
2967 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqw
2968 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i16, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqw
2969 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i16, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqw
2970 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i32, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqd
2971 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i32, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpmovqd
2972 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },// 2*vpmovqd+concat+vpmovdb
2973
2974 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // extend to v16i32
2975 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i8, .Src: MVT::v32i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2976 { .ISD: ISD::TRUNCATE, .Dst: MVT::v64i8, .Src: MVT::v32i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2977
2978 // Sign extend is zmm vpternlogd+vptruncdb.
2979 // Zero extend is zmm broadcast load+vptruncdw.
2980 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2981 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2982 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2983 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2984 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2985 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2986 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2987 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2988
2989 // Sign extend is zmm vpternlogd+vptruncdw.
2990 // Zero extend is zmm vpternlogd+vptruncdw+vpsrlw.
2991 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2992 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2993 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2994 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2995 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2996 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2997 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2998 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
2999
3000 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i32, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogd
3001 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i32, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogd+psrld
3002 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogd
3003 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogd+psrld
3004 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogd
3005 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogd+psrld
3006 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogq
3007 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogq+psrlq
3008 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogq
3009 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // zmm vpternlogq+psrlq
3010
3011 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd
3012 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd+psrld
3013 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq
3014 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq+psrlq
3015
3016 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3017 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3018 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3019 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3020 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3021 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3022 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3023 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3024 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3025 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i64, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3026
3027 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v32i16, .Src: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // FIXME: May not be right
3028 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v32i16, .Src: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // FIXME: May not be right
3029
3030 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3031 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v16f32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3032 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3033 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v16f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3034 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3035 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v16f32, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3036 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3037 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v16f32, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3038
3039 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3040 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v16f32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3041 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3042 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v16f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3043 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3044 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v16f32, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3045 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3046 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v16f32, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3047 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i64, .Cost: {.RecipThroughputCost: 26, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3048 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3049
3050 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v16f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3051 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v16f64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3052 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v32i8, .Src: MVT::v32f64, .Cost: {.RecipThroughputCost: 15, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3053 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v64i8, .Src: MVT::v64f32, .Cost: {.RecipThroughputCost: 11, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3054 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v64i8, .Src: MVT::v64f64, .Cost: {.RecipThroughputCost: 31, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3055 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i16, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3056 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i16, .Src: MVT::v16f64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3057 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v32i16, .Src: MVT::v32f32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3058 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v32i16, .Src: MVT::v32f64, .Cost: {.RecipThroughputCost: 15, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3059 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i32, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3060 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i32, .Src: MVT::v16f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3061
3062 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i32, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3063 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i16, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3064 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i8, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3065 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i32, .Src: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3066 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i16, .Src: MVT::v16f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3067 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i8, .Src: MVT::v16f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3068 };
3069
3070 static const TypeConversionCostKindTblEntry AVX512BWVLConversionTbl[] {
3071 // Mask sign extend has an instruction.
3072 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3073 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3074 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3075 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3076 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3077 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3078 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3079 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3080 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3081 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3082 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3083 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3084 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3085 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v32i8, .Src: MVT::v32i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3086 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v32i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3087 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v32i8, .Src: MVT::v64i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3088 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v64i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3089
3090 // Mask zero extend is a sext + shift.
3091 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3092 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3093 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3094 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3095 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3096 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3097 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3098 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3099 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3100 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3101 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3102 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3103 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3104 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v32i8, .Src: MVT::v32i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3105 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v32i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3106 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v32i8, .Src: MVT::v64i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3107 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v64i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3108
3109 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3110 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3111 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3112 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3113 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3114 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3115 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3116 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3117 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3118 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3119 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3120 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3121 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3122 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i1, .Src: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3123 { .ISD: ISD::TRUNCATE, .Dst: MVT::v32i1, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3124 { .ISD: ISD::TRUNCATE, .Dst: MVT::v64i1, .Src: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3125 { .ISD: ISD::TRUNCATE, .Dst: MVT::v64i1, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3126
3127 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3128 };
3129
3130 static const TypeConversionCostKindTblEntry AVX512DQVLConversionTbl[] = {
3131 // Mask sign extend has an instruction.
3132 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3133 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3134 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3135 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3136 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3137 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3138 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3139 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3140
3141 // Mask zero extend is a sext + shift.
3142 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3143 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3144 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3145 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3146 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3147 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3148 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3149 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3150
3151 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3152 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3153 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3154 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3155 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3156 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3157 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3158 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3159
3160 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3161 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3162 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3163 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3164
3165 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3166 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3167 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3168 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3169
3170 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v2i64, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3171 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v4i64, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3172 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v2i64, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3173 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v4i64, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3174
3175 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v2i64, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3176 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i64, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3177 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v2i64, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3178 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i64, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3179 };
3180
3181 static const TypeConversionCostKindTblEntry AVX512VLConversionTbl[] = {
3182 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpslld+vptestmd
3183 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpslld+vptestmd
3184 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpslld+vptestmd
3185 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // split+2*v8i8
3186 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpsllq+vptestmq
3187 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpsllq+vptestmq
3188 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sext+vpsllq+vptestmq
3189 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // split+2*v8i16
3190 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpslld+vptestmd
3191 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpslld+vptestmd
3192 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpslld+vptestmd
3193 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpslld+vptestmd
3194 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsllq+vptestmq
3195 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpsllq+vptestmq
3196 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i32, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqd
3197 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i8, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqb
3198 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i16, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovqw
3199 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i8, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpmovwb
3200
3201 // sign extend is vpcmpeq+maskedmove+vpmovdw+vpacksswb
3202 // zero extend is vpcmpeq+maskedmove+vpmovdw+vpsrlw+vpackuswb
3203 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3204 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i8, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3205 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3206 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i8, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3207 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3208 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i8, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3209 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v16i1, .Cost: {.RecipThroughputCost: 10, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3210 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i8, .Src: MVT::v16i1, .Cost: {.RecipThroughputCost: 12, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3211
3212 // sign extend is vpcmpeq+maskedmove+vpmovdw
3213 // zero extend is vpcmpeq+maskedmove+vpmovdw+vpsrlw
3214 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3215 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i16, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3216 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3217 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i16, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3218 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3219 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3220 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: {.RecipThroughputCost: 10, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3221 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: {.RecipThroughputCost: 12, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3222
3223 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i32, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd
3224 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i32, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd+psrld
3225 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd
3226 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd+psrld
3227 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd
3228 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd+psrld
3229 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd
3230 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogd+psrld
3231
3232 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq
3233 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v2i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq+psrlq
3234 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq
3235 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vpternlogq+psrlq
3236
3237 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3238 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3239 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3240 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3241 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3242 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3243 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3244 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3245 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3246 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3247 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3248 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3249
3250 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3251 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3252 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3253 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3254
3255 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3256 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3257 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3258 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3259 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3260 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3261 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3262 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3263 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3264 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3265 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3266 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3267 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3268
3269 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3270 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v16f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3271 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v32i8, .Src: MVT::v32f32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3272
3273 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i64, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3274 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i64, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3275 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3276 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3277 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3278 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i32, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3279 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i32, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3280 };
3281
3282 static const TypeConversionCostKindTblEntry AVX2ConversionTbl[] = {
3283 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3284 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3285 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3286 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3287 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3288 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3289
3290 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3291 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3292 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3293 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3294 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3295 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3296 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3297 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3298 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3299 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3300 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3301 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i32, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3302 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3303 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3304
3305 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3306
3307 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i16, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3308 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3309 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3310 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3311 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3312 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3313 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3314 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3315 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3316 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3317 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i32, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3318 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3319
3320 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v8f64, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3321 { .ISD: ISD::FP_ROUND, .Dst: MVT::v8f32, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3322
3323 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i16, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3324 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v4i32, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3325 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i32, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3326 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i32, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3327
3328 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i64, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3329 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i64, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3330 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i16, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3331 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3332 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3333 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3334 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i32, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3335 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i32, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3336
3337 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3338 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3339 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3340 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3341 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3342 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3343 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3344
3345 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3346 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3347 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3348 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3349 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3350 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3351 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3352 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3353 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3354 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3355 };
3356
3357 static const TypeConversionCostKindTblEntry AVXConversionTbl[] = {
3358 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3359 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3360 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3361 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3362 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3363 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i1, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3364
3365 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3366 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3367 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3368 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3369 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3370 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v16i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3371 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3372 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3373 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3374 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3375 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3376 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3377
3378 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3379 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3380 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3381 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i64, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3382 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i1, .Src: MVT::v16i64, .Cost: {.RecipThroughputCost: 11, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3383
3384 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i16, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3385 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3386 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // and+extract+packuswb
3387 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3388 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3389 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3390 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // and+extract+2*packusdw
3391 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i32, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3392
3393 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3394 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3395 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3396 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3397 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3398 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3399 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3400 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3401 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3402 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3403 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3404 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3405
3406 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3407 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i1, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3408 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i1, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3409 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3410 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3411 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3412 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3413 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3414 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3415 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3416 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3417 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f32, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3418 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v8f64, .Src: MVT::v8i32, .Cost: {.RecipThroughputCost: 10, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3419 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i64, .Cost: {.RecipThroughputCost: 11, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3420 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i64, .Cost: {.RecipThroughputCost: 18, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3421 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3422 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i64, .Cost: {.RecipThroughputCost: 10, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3423
3424 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3425 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3426 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v32i8, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3427 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v32i8, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3428 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i16, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3429 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i16, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3430 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i16, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3431 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i16, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3432 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v4i32, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3433 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i32, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3434 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i32, .Src: MVT::v8f64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3435
3436 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i8, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3437 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i8, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3438 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v32i8, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3439 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v32i8, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3440 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i16, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3441 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i16, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3442 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i16, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3443 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i16, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3444 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3445 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3446 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3447 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i32, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3448 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i32, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3449
3450 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v4f64, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3451 { .ISD: ISD::FP_ROUND, .Dst: MVT::v4f32, .Src: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3452 };
3453
3454 static const TypeConversionCostKindTblEntry SSE41ConversionTbl[] = {
3455 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3456 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3457 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3458 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3459 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3460 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3461 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3462 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3463 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3464 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3465 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3466 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3467
3468 // These truncates end up widening elements.
3469 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PMOVXZBQ
3470 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PMOVXZWQ
3471 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PMOVXZBD
3472
3473 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3474 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3475 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3476
3477 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3478 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3479 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3480 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3481 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3482 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3483 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3484 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3485 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3486 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3487 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3488
3489 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3490 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3491 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3492 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3493 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3494 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3495 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3496 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3497 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3498 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3499 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3500 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v2i64, .Cost: {.RecipThroughputCost: 12, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3501 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i64, .Cost: {.RecipThroughputCost: 22, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3502 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3503
3504 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::i32, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3505 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::i64, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3506 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::i32, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3507 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::i64, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3508 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3509 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3510 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i16, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3511 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i16, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3512 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v4i32, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3513 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v4i32, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3514
3515 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i32, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3516 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i64, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3517 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i32, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3518 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i64, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3519 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i8, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3520 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i8, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3521 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i16, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3522 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i16, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3523 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3524 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3525 };
3526
3527 static const TypeConversionCostKindTblEntry SSE2ConversionTbl[] = {
3528 // These are somewhat magic numbers justified by comparing the
3529 // output of llvm-mca for our various supported scheduler models
3530 // and basing it off the worst case scenario.
3531 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3532 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3533 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3534 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3535 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3536 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3537 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3538 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3539 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3540 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3541 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3542 { .ISD: ISD::SINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3543
3544 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3545 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3546 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f32, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3547 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::f64, .Src: MVT::i64, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3548 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3549 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3550 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3551 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3552 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f32, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3553 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3554 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3555 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v2f64, .Src: MVT::v2i64, .Cost: {.RecipThroughputCost: 15, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3556 { .ISD: ISD::UINT_TO_FP, .Dst: MVT::v4f32, .Src: MVT::v2i64, .Cost: {.RecipThroughputCost: 18, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3557
3558 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::i32, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3559 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::i64, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3560 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::i32, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3561 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::i64, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3562 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3563 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v16i8, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3564 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i16, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3565 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v8i16, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3566 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v4i32, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3567 { .ISD: ISD::FP_TO_SINT, .Dst: MVT::v4i32, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3568
3569 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i32, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3570 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i64, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3571 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i32, .Src: MVT::f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3572 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::i64, .Src: MVT::f64, .Cost: {.RecipThroughputCost: 15, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3573 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i8, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3574 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v16i8, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3575 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i16, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3576 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v8i16, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3577 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3578 { .ISD: ISD::FP_TO_UINT, .Dst: MVT::v4i32, .Src: MVT::v2f64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3579
3580 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3581 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3582 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3583 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3584 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3585 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v8i16, .Src: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3586 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3587 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3588 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3589 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v4i32, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3590 { .ISD: ISD::ZERO_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3591 { .ISD: ISD::SIGN_EXTEND, .Dst: MVT::v2i64, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3592
3593 // These truncates are really widening elements.
3594 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PSHUFD
3595 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PUNPCKLWD+DQ
3596 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i1, .Src: MVT::v2i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PUNPCKLBW+WD+PSHUFD
3597 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PUNPCKLWD
3598 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i1, .Src: MVT::v4i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PUNPCKLBW+WD
3599 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i1, .Src: MVT::v8i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PUNPCKLBW
3600
3601 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PAND+PACKUSWB
3602 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3603 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PAND+2*PACKUSWB
3604 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v16i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3605 { .ISD: ISD::TRUNCATE, .Dst: MVT::v2i16, .Src: MVT::v2i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3606 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3607 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v8i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3608 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i16, .Src: MVT::v16i32, .Cost: {.RecipThroughputCost: 10, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3609 { .ISD: ISD::TRUNCATE, .Dst: MVT::v16i8, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PAND+3*PACKUSWB
3610 { .ISD: ISD::TRUNCATE, .Dst: MVT::v8i16, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PSHUFD+PSHUFLW
3611 { .ISD: ISD::TRUNCATE, .Dst: MVT::v4i32, .Src: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // PSHUFD
3612 };
3613
3614 static const TypeConversionCostKindTblEntry F16ConversionTbl[] = {
3615 { .ISD: ISD::FP_ROUND, .Dst: MVT::f16, .Src: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3616 { .ISD: ISD::FP_ROUND, .Dst: MVT::v8f16, .Src: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3617 { .ISD: ISD::FP_ROUND, .Dst: MVT::v4f16, .Src: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3618 { .ISD: ISD::FP_EXTEND, .Dst: MVT::f32, .Src: MVT::f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3619 { .ISD: ISD::FP_EXTEND, .Dst: MVT::f64, .Src: MVT::f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vcvtph2ps+vcvtps2pd
3620 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v8f32, .Src: MVT::v8f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3621 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v4f32, .Src: MVT::v4f16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3622 { .ISD: ISD::FP_EXTEND, .Dst: MVT::v4f64, .Src: MVT::v4f16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vcvtph2ps+vcvtps2pd
3623 };
3624
3625 // Attempt to map directly to (simple) MVT types to let us match custom entries.
3626 EVT SrcTy = TLI->getValueType(DL, Ty: Src);
3627 EVT DstTy = TLI->getValueType(DL, Ty: Dst);
3628
3629 // If we're sign-extending a vector comparison result back to the comparison
3630 // width, this will be free without AVX512 (or for 8/16-bit types without
3631 // BWI).
3632 if (!ST->hasAVX512() || (!ST->hasBWI() && DstTy.getScalarSizeInBits() < 32)) {
3633 if (I && Opcode == Instruction::CastOps::SExt &&
3634 SrcTy.isFixedLengthVectorOf(EltVT: MVT::i1)) {
3635 if (auto *CmpI = dyn_cast<CmpInst>(Val: I->getOperand(i: 0))) {
3636 Type *CmpTy = CmpI->getOperand(i_nocapture: 0)->getType();
3637 if (CmpTy->getScalarSizeInBits() == DstTy.getScalarSizeInBits())
3638 return TTI::TCC_Free;
3639 }
3640 }
3641 }
3642
3643 // The function getSimpleVT only handles simple value types.
3644 if (SrcTy.isSimple() && DstTy.isSimple()) {
3645 MVT SimpleSrcTy = SrcTy.getSimpleVT();
3646 MVT SimpleDstTy = DstTy.getSimpleVT();
3647
3648 if (ST->useAVX512Regs()) {
3649 if (ST->hasBWI())
3650 if (const auto *Entry = ConvertCostTableLookup(
3651 Table: AVX512BWConversionTbl, ISD, Dst: SimpleDstTy, Src: SimpleSrcTy))
3652 if (auto KindCost = Entry->Cost[CostKind])
3653 return *KindCost;
3654
3655 if (ST->hasDQI())
3656 if (const auto *Entry = ConvertCostTableLookup(
3657 Table: AVX512DQConversionTbl, ISD, Dst: SimpleDstTy, Src: SimpleSrcTy))
3658 if (auto KindCost = Entry->Cost[CostKind])
3659 return *KindCost;
3660
3661 if (ST->hasAVX512())
3662 if (const auto *Entry = ConvertCostTableLookup(
3663 Table: AVX512FConversionTbl, ISD, Dst: SimpleDstTy, Src: SimpleSrcTy))
3664 if (auto KindCost = Entry->Cost[CostKind])
3665 return *KindCost;
3666 }
3667
3668 if (ST->hasBWI())
3669 if (const auto *Entry = ConvertCostTableLookup(
3670 Table: AVX512BWVLConversionTbl, ISD, Dst: SimpleDstTy, Src: SimpleSrcTy))
3671 if (auto KindCost = Entry->Cost[CostKind])
3672 return *KindCost;
3673
3674 if (ST->hasDQI())
3675 if (const auto *Entry = ConvertCostTableLookup(
3676 Table: AVX512DQVLConversionTbl, ISD, Dst: SimpleDstTy, Src: SimpleSrcTy))
3677 if (auto KindCost = Entry->Cost[CostKind])
3678 return *KindCost;
3679
3680 if (ST->hasAVX512())
3681 if (const auto *Entry = ConvertCostTableLookup(Table: AVX512VLConversionTbl, ISD,
3682 Dst: SimpleDstTy, Src: SimpleSrcTy))
3683 if (auto KindCost = Entry->Cost[CostKind])
3684 return *KindCost;
3685
3686 if (ST->hasAVX2()) {
3687 if (const auto *Entry = ConvertCostTableLookup(Table: AVX2ConversionTbl, ISD,
3688 Dst: SimpleDstTy, Src: SimpleSrcTy))
3689 if (auto KindCost = Entry->Cost[CostKind])
3690 return *KindCost;
3691 }
3692
3693 if (ST->hasAVX()) {
3694 if (const auto *Entry = ConvertCostTableLookup(Table: AVXConversionTbl, ISD,
3695 Dst: SimpleDstTy, Src: SimpleSrcTy))
3696 if (auto KindCost = Entry->Cost[CostKind])
3697 return *KindCost;
3698 }
3699
3700 if (ST->hasF16C()) {
3701 if (const auto *Entry = ConvertCostTableLookup(Table: F16ConversionTbl, ISD,
3702 Dst: SimpleDstTy, Src: SimpleSrcTy))
3703 if (auto KindCost = Entry->Cost[CostKind])
3704 return *KindCost;
3705 }
3706
3707 if (ST->hasSSE41()) {
3708 if (const auto *Entry = ConvertCostTableLookup(Table: SSE41ConversionTbl, ISD,
3709 Dst: SimpleDstTy, Src: SimpleSrcTy))
3710 if (auto KindCost = Entry->Cost[CostKind])
3711 return *KindCost;
3712 }
3713
3714 if (ST->hasSSE2()) {
3715 if (const auto *Entry = ConvertCostTableLookup(Table: SSE2ConversionTbl, ISD,
3716 Dst: SimpleDstTy, Src: SimpleSrcTy))
3717 if (auto KindCost = Entry->Cost[CostKind])
3718 return *KindCost;
3719 }
3720
3721 if ((ISD == ISD::FP_ROUND && SimpleDstTy == MVT::f16) ||
3722 (ISD == ISD::FP_EXTEND && SimpleSrcTy == MVT::f16)) {
3723 // fp16 conversions not covered by any table entries require a libcall.
3724 // Return a large (arbitrary) number to model this.
3725 return InstructionCost(64);
3726 }
3727 }
3728
3729 // Fall back to legalized types.
3730 std::pair<InstructionCost, MVT> LTSrc = getTypeLegalizationCost(Ty: Src);
3731 std::pair<InstructionCost, MVT> LTDest = getTypeLegalizationCost(Ty: Dst);
3732
3733 // If we're truncating to the same legalized type - just assume its free.
3734 if (ISD == ISD::TRUNCATE && LTSrc.second == LTDest.second)
3735 return TTI::TCC_Free;
3736
3737 if (ST->useAVX512Regs()) {
3738 if (ST->hasBWI())
3739 if (const auto *Entry = ConvertCostTableLookup(
3740 Table: AVX512BWConversionTbl, ISD, Dst: LTDest.second, Src: LTSrc.second))
3741 if (auto KindCost = Entry->Cost[CostKind])
3742 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3743
3744 if (ST->hasDQI())
3745 if (const auto *Entry = ConvertCostTableLookup(
3746 Table: AVX512DQConversionTbl, ISD, Dst: LTDest.second, Src: LTSrc.second))
3747 if (auto KindCost = Entry->Cost[CostKind])
3748 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3749
3750 if (ST->hasAVX512())
3751 if (const auto *Entry = ConvertCostTableLookup(
3752 Table: AVX512FConversionTbl, ISD, Dst: LTDest.second, Src: LTSrc.second))
3753 if (auto KindCost = Entry->Cost[CostKind])
3754 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3755 }
3756
3757 if (ST->hasBWI())
3758 if (const auto *Entry = ConvertCostTableLookup(Table: AVX512BWVLConversionTbl, ISD,
3759 Dst: LTDest.second, Src: LTSrc.second))
3760 if (auto KindCost = Entry->Cost[CostKind])
3761 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3762
3763 if (ST->hasDQI())
3764 if (const auto *Entry = ConvertCostTableLookup(Table: AVX512DQVLConversionTbl, ISD,
3765 Dst: LTDest.second, Src: LTSrc.second))
3766 if (auto KindCost = Entry->Cost[CostKind])
3767 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3768
3769 if (ST->hasAVX512())
3770 if (const auto *Entry = ConvertCostTableLookup(Table: AVX512VLConversionTbl, ISD,
3771 Dst: LTDest.second, Src: LTSrc.second))
3772 if (auto KindCost = Entry->Cost[CostKind])
3773 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3774
3775 if (ST->hasAVX2())
3776 if (const auto *Entry = ConvertCostTableLookup(Table: AVX2ConversionTbl, ISD,
3777 Dst: LTDest.second, Src: LTSrc.second))
3778 if (auto KindCost = Entry->Cost[CostKind])
3779 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3780
3781 if (ST->hasAVX())
3782 if (const auto *Entry = ConvertCostTableLookup(Table: AVXConversionTbl, ISD,
3783 Dst: LTDest.second, Src: LTSrc.second))
3784 if (auto KindCost = Entry->Cost[CostKind])
3785 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3786
3787 if (ST->hasF16C()) {
3788 if (const auto *Entry = ConvertCostTableLookup(Table: F16ConversionTbl, ISD,
3789 Dst: LTDest.second, Src: LTSrc.second))
3790 if (auto KindCost = Entry->Cost[CostKind])
3791 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3792 }
3793
3794 if (ST->hasSSE41())
3795 if (const auto *Entry = ConvertCostTableLookup(Table: SSE41ConversionTbl, ISD,
3796 Dst: LTDest.second, Src: LTSrc.second))
3797 if (auto KindCost = Entry->Cost[CostKind])
3798 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3799
3800 if (ST->hasSSE2())
3801 if (const auto *Entry = ConvertCostTableLookup(Table: SSE2ConversionTbl, ISD,
3802 Dst: LTDest.second, Src: LTSrc.second))
3803 if (auto KindCost = Entry->Cost[CostKind])
3804 return std::max(a: LTSrc.first, b: LTDest.first) * *KindCost;
3805
3806 // Fallback, for i8/i16 sitofp/uitofp cases we need to extend to i32 for
3807 // sitofp.
3808 if ((ISD == ISD::SINT_TO_FP || ISD == ISD::UINT_TO_FP) &&
3809 1 < Src->getScalarSizeInBits() && Src->getScalarSizeInBits() < 32) {
3810 Type *ExtSrc = Src->getWithNewBitWidth(NewBitWidth: 32);
3811 unsigned ExtOpc =
3812 (ISD == ISD::SINT_TO_FP) ? Instruction::SExt : Instruction::ZExt;
3813
3814 // For scalar loads the extend would be free.
3815 InstructionCost ExtCost = 0;
3816 if (!(Src->isIntegerTy() && I && isa<LoadInst>(Val: I->getOperand(i: 0))))
3817 ExtCost = getCastInstrCost(Opcode: ExtOpc, Dst: ExtSrc, Src, CCH, CostKind);
3818
3819 return ExtCost + getCastInstrCost(Opcode: Instruction::SIToFP, Dst, Src: ExtSrc,
3820 CCH: TTI::CastContextHint::None, CostKind);
3821 }
3822
3823 // Fallback for fptosi/fptoui i8/i16 cases we need to truncate from fptosi
3824 // i32.
3825 if ((ISD == ISD::FP_TO_SINT || ISD == ISD::FP_TO_UINT) &&
3826 1 < Dst->getScalarSizeInBits() && Dst->getScalarSizeInBits() < 32) {
3827 Type *TruncDst = Dst->getWithNewBitWidth(NewBitWidth: 32);
3828 return getCastInstrCost(Opcode: Instruction::FPToSI, Dst: TruncDst, Src, CCH, CostKind) +
3829 getCastInstrCost(Opcode: Instruction::Trunc, Dst, Src: TruncDst,
3830 CCH: TTI::CastContextHint::None, CostKind);
3831 }
3832
3833 // TODO: Allow non-throughput costs that aren't binary.
3834 auto AdjustCost = [&CostKind](InstructionCost Cost,
3835 InstructionCost N = 1) -> InstructionCost {
3836 if (CostKind != TTI::TCK_RecipThroughput)
3837 return Cost == 0 ? 0 : N;
3838 return Cost * N;
3839 };
3840 return AdjustCost(
3841 BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I));
3842}
3843
3844// Additive cost for a predicated op's "predicate fanout": a select, masked
3845// load/store or gather/scatter keeps a compact <N x i1> mask, but if its data
3846// legalizes into more parts than the mask, codegen builds a sub-mask per extra
3847// part (kshiftr on AVX-512, unpack/extend on SSE/AVX) the per-part tables miss.
3848// FanoutCostPerPart is a tie-break: it beats the vectorizer's widest-VF bias.
3849static InstructionCost getPredicateFanoutCost(const X86TTIImpl &TTI,
3850 Type *DataVTy, Type *MaskVTy) {
3851 constexpr unsigned FanoutCostPerPart = 4;
3852 unsigned DataParts = TTI.getNumberOfParts(Tp: DataVTy);
3853 unsigned MaskParts = TTI.getNumberOfParts(Tp: MaskVTy);
3854 if (DataParts > MaskParts && MaskParts > 0)
3855 return InstructionCost((DataParts - MaskParts) * FanoutCostPerPart);
3856 return InstructionCost(0);
3857}
3858
3859InstructionCost X86TTIImpl::getCmpSelInstrCost(
3860 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
3861 TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
3862 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
3863 // Early out if this type isn't scalar/vector integer/float.
3864 if (!(ValTy->isIntOrIntVectorTy() || ValTy->isFPOrFPVectorTy()))
3865 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
3866 Op1Info, Op2Info, I);
3867
3868 // Legalize the type.
3869 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: ValTy);
3870
3871 MVT MTy = LT.second;
3872
3873 int ISD = TLI->InstructionOpcodeToISD(Opcode);
3874 assert(ISD && "Invalid opcode");
3875
3876 // Predicate fanout (see getPredicateFanoutCost), AVX-512 only: on SSE/AVX a
3877 // select blend is already priced per part, so charging it there shrinks VF.
3878 InstructionCost MaskExpCost = 0;
3879 if (Opcode == Instruction::Select && ST->hasAVX512() &&
3880 isa<VectorType>(Val: ValTy) && isa_and_nonnull<VectorType>(Val: CondTy))
3881 MaskExpCost = getPredicateFanoutCost(TTI: *this, DataVTy: ValTy, MaskVTy: CondTy);
3882
3883 InstructionCost ExtraCost = 0;
3884 if (Opcode == Instruction::ICmp || Opcode == Instruction::FCmp) {
3885 // Some vector comparison predicates cost extra instructions.
3886 // TODO: Adjust ExtraCost based on CostKind?
3887 // TODO: Should we invert this and assume worst case cmp costs
3888 // and reduce for particular predicates?
3889 if (MTy.isVector() &&
3890 !((ST->hasXOP() && (!ST->hasAVX2() || MTy.is128BitVector())) ||
3891 (ST->hasAVX512() && 32 <= MTy.getScalarSizeInBits()) ||
3892 ST->hasBWI())) {
3893 // Fallback to I if a specific predicate wasn't specified.
3894 CmpInst::Predicate Pred = VecPred;
3895 if (I && (Pred == CmpInst::BAD_ICMP_PREDICATE ||
3896 Pred == CmpInst::BAD_FCMP_PREDICATE))
3897 Pred = cast<CmpInst>(Val: I)->getPredicate();
3898
3899 bool CmpWithConstant = false;
3900 if (auto *CmpInstr = dyn_cast_or_null<CmpInst>(Val: I))
3901 CmpWithConstant = isa<Constant>(Val: CmpInstr->getOperand(i_nocapture: 1));
3902
3903 switch (Pred) {
3904 case CmpInst::Predicate::ICMP_NE:
3905 // xor(cmpeq(x,y),-1)
3906 ExtraCost = CmpWithConstant ? 0 : 1;
3907 break;
3908 case CmpInst::Predicate::ICMP_SGE:
3909 case CmpInst::Predicate::ICMP_SLE:
3910 // xor(cmpgt(x,y),-1)
3911 ExtraCost = CmpWithConstant ? 0 : 1;
3912 break;
3913 case CmpInst::Predicate::ICMP_ULT:
3914 case CmpInst::Predicate::ICMP_UGT:
3915 // cmpgt(xor(x,signbit),xor(y,signbit))
3916 // xor(cmpeq(pmaxu(x,y),x),-1)
3917 ExtraCost = CmpWithConstant ? 1 : 2;
3918 break;
3919 case CmpInst::Predicate::ICMP_ULE:
3920 case CmpInst::Predicate::ICMP_UGE:
3921 if ((ST->hasSSE41() && MTy.getScalarSizeInBits() == 32) ||
3922 (ST->hasSSE2() && MTy.getScalarSizeInBits() < 32)) {
3923 // cmpeq(psubus(x,y),0)
3924 // cmpeq(pminu(x,y),x)
3925 ExtraCost = 1;
3926 } else {
3927 // xor(cmpgt(xor(x,signbit),xor(y,signbit)),-1)
3928 ExtraCost = CmpWithConstant ? 2 : 3;
3929 }
3930 break;
3931 case CmpInst::Predicate::FCMP_ONE:
3932 case CmpInst::Predicate::FCMP_UEQ:
3933 // Without AVX we need to expand FCMP_ONE/FCMP_UEQ cases.
3934 // Use FCMP_UEQ expansion - FCMP_ONE should be the same.
3935 if (CondTy && !ST->hasAVX())
3936 return getCmpSelInstrCost(Opcode, ValTy, CondTy,
3937 VecPred: CmpInst::Predicate::FCMP_UNO, CostKind,
3938 Op1Info, Op2Info) +
3939 getCmpSelInstrCost(Opcode, ValTy, CondTy,
3940 VecPred: CmpInst::Predicate::FCMP_OEQ, CostKind,
3941 Op1Info, Op2Info) +
3942 getArithmeticInstrCost(Opcode: Instruction::Or, Ty: CondTy, CostKind);
3943
3944 break;
3945 case CmpInst::Predicate::BAD_ICMP_PREDICATE:
3946 case CmpInst::Predicate::BAD_FCMP_PREDICATE:
3947 // Assume worst case scenario and add the maximum extra cost.
3948 ExtraCost = 3;
3949 break;
3950 default:
3951 break;
3952 }
3953 }
3954 }
3955
3956 static const CostKindTblEntry SLMCostTbl[] = {
3957 // slm pcmpeq/pcmpgt throughput is 2
3958 { .ISD: ISD::SETCC, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
3959 // slm pblendvb/blendvpd/blendvps throughput is 4
3960 { .ISD: ISD::SELECT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vblendvpd
3961 { .ISD: ISD::SELECT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vblendvps
3962 { .ISD: ISD::SELECT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // pblendvb
3963 { .ISD: ISD::SELECT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // pblendvb
3964 { .ISD: ISD::SELECT, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // pblendvb
3965 { .ISD: ISD::SELECT, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // pblendvb
3966 };
3967
3968 static const CostKindTblEntry AVX512BWCostTbl[] = {
3969 { .ISD: ISD::SETCC, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3970 { .ISD: ISD::SETCC, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3971 { .ISD: ISD::SETCC, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3972 { .ISD: ISD::SETCC, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3973
3974 { .ISD: ISD::SELECT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3975 { .ISD: ISD::SELECT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3976 };
3977
3978 static const CostKindTblEntry AVX512CostTbl[] = {
3979 { .ISD: ISD::SETCC, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3980 { .ISD: ISD::SETCC, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3981 { .ISD: ISD::SETCC, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3982 { .ISD: ISD::SETCC, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3983
3984 { .ISD: ISD::SETCC, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3985 { .ISD: ISD::SETCC, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3986 { .ISD: ISD::SETCC, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3987 { .ISD: ISD::SETCC, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3988 { .ISD: ISD::SETCC, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3989 { .ISD: ISD::SETCC, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
3990 { .ISD: ISD::SETCC, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
3991
3992 { .ISD: ISD::SELECT, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3993 { .ISD: ISD::SELECT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3994 { .ISD: ISD::SELECT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3995 { .ISD: ISD::SELECT, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3996 { .ISD: ISD::SELECT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3997 { .ISD: ISD::SELECT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3998 { .ISD: ISD::SELECT, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
3999 { .ISD: ISD::SELECT, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4000 { .ISD: ISD::SELECT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4001 { .ISD: ISD::SELECT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4002 { .ISD: ISD::SELECT, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4003 { .ISD: ISD::SELECT, .Type: MVT::v8f32 , .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4004 { .ISD: ISD::SELECT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4005 { .ISD: ISD::SELECT, .Type: MVT::f32 , .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4006
4007 { .ISD: ISD::SELECT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4008 { .ISD: ISD::SELECT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4009 { .ISD: ISD::SELECT, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4010 { .ISD: ISD::SELECT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4011 { .ISD: ISD::SELECT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4012 { .ISD: ISD::SELECT, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4013 };
4014
4015 static const CostKindTblEntry AVX2CostTbl[] = {
4016 { .ISD: ISD::SETCC, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4017 { .ISD: ISD::SETCC, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4018 { .ISD: ISD::SETCC, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4019 { .ISD: ISD::SETCC, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4020 { .ISD: ISD::SETCC, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4021 { .ISD: ISD::SETCC, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4022
4023 { .ISD: ISD::SETCC, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4024 { .ISD: ISD::SETCC, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4025 { .ISD: ISD::SETCC, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4026 { .ISD: ISD::SETCC, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4027
4028 { .ISD: ISD::SELECT, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vblendvpd
4029 { .ISD: ISD::SELECT, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vblendvps
4030 { .ISD: ISD::SELECT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pblendvb
4031 { .ISD: ISD::SELECT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pblendvb
4032 { .ISD: ISD::SELECT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pblendvb
4033 { .ISD: ISD::SELECT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pblendvb
4034 };
4035
4036 static const CostKindTblEntry XOPCostTbl[] = {
4037 { .ISD: ISD::SETCC, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4038 { .ISD: ISD::SETCC, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4039 };
4040
4041 static const CostKindTblEntry AVX1CostTbl[] = {
4042 { .ISD: ISD::SETCC, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4043 { .ISD: ISD::SETCC, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4044 { .ISD: ISD::SETCC, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4045 { .ISD: ISD::SETCC, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4046 { .ISD: ISD::SETCC, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4047 { .ISD: ISD::SETCC, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4048
4049 // AVX1 does not support 8-wide integer compare.
4050 { .ISD: ISD::SETCC, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4051 { .ISD: ISD::SETCC, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4052 { .ISD: ISD::SETCC, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4053 { .ISD: ISD::SETCC, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4054
4055 { .ISD: ISD::SELECT, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vblendvpd
4056 { .ISD: ISD::SELECT, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vblendvps
4057 { .ISD: ISD::SELECT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vblendvpd
4058 { .ISD: ISD::SELECT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // vblendvps
4059 { .ISD: ISD::SELECT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // vandps + vandnps + vorps
4060 { .ISD: ISD::SELECT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // vandps + vandnps + vorps
4061 };
4062
4063 static const CostKindTblEntry SSE42CostTbl[] = {
4064 { .ISD: ISD::SETCC, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4065 };
4066
4067 static const CostKindTblEntry SSE41CostTbl[] = {
4068 { .ISD: ISD::SETCC, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4069 { .ISD: ISD::SETCC, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4070
4071 { .ISD: ISD::SELECT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // blendvpd
4072 { .ISD: ISD::SELECT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // blendvpd
4073 { .ISD: ISD::SELECT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // blendvps
4074 { .ISD: ISD::SELECT, .Type: MVT::f32 , .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // blendvps
4075 { .ISD: ISD::SELECT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pblendvb
4076 { .ISD: ISD::SELECT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pblendvb
4077 { .ISD: ISD::SELECT, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pblendvb
4078 { .ISD: ISD::SELECT, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // pblendvb
4079 };
4080
4081 static const CostKindTblEntry SSE2CostTbl[] = {
4082 { .ISD: ISD::SETCC, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4083 { .ISD: ISD::SETCC, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4084
4085 { .ISD: ISD::SETCC, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } }, // pcmpeqd/pcmpgtd expansion
4086 { .ISD: ISD::SETCC, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4087 { .ISD: ISD::SETCC, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4088 { .ISD: ISD::SETCC, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4089
4090 { .ISD: ISD::SELECT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // andpd + andnpd + orpd
4091 { .ISD: ISD::SELECT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // andpd + andnpd + orpd
4092 { .ISD: ISD::SELECT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // pand + pandn + por
4093 { .ISD: ISD::SELECT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // pand + pandn + por
4094 { .ISD: ISD::SELECT, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // pand + pandn + por
4095 { .ISD: ISD::SELECT, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // pand + pandn + por
4096 };
4097
4098 static const CostKindTblEntry SSE1CostTbl[] = {
4099 { .ISD: ISD::SETCC, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4100 { .ISD: ISD::SETCC, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4101
4102 { .ISD: ISD::SELECT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // andps + andnps + orps
4103 { .ISD: ISD::SELECT, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // andps + andnps + orps
4104 };
4105
4106 if (ST->useSLMArithCosts())
4107 if (const auto *Entry = CostTableLookup(Table: SLMCostTbl, ISD, Ty: MTy))
4108 if (auto KindCost = Entry->Cost[CostKind])
4109 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4110
4111 if (ST->hasBWI())
4112 if (const auto *Entry = CostTableLookup(Table: AVX512BWCostTbl, ISD, Ty: MTy))
4113 if (auto KindCost = Entry->Cost[CostKind])
4114 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4115
4116 if (ST->hasAVX512())
4117 if (const auto *Entry = CostTableLookup(Table: AVX512CostTbl, ISD, Ty: MTy))
4118 if (auto KindCost = Entry->Cost[CostKind])
4119 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4120
4121 if (ST->hasAVX2())
4122 if (const auto *Entry = CostTableLookup(Table: AVX2CostTbl, ISD, Ty: MTy))
4123 if (auto KindCost = Entry->Cost[CostKind])
4124 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4125
4126 if (ST->hasXOP())
4127 if (const auto *Entry = CostTableLookup(Table: XOPCostTbl, ISD, Ty: MTy))
4128 if (auto KindCost = Entry->Cost[CostKind])
4129 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4130
4131 if (ST->hasAVX())
4132 if (const auto *Entry = CostTableLookup(Table: AVX1CostTbl, ISD, Ty: MTy))
4133 if (auto KindCost = Entry->Cost[CostKind])
4134 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4135
4136 if (ST->hasSSE42())
4137 if (const auto *Entry = CostTableLookup(Table: SSE42CostTbl, ISD, Ty: MTy))
4138 if (auto KindCost = Entry->Cost[CostKind])
4139 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4140
4141 if (ST->hasSSE41())
4142 if (const auto *Entry = CostTableLookup(Table: SSE41CostTbl, ISD, Ty: MTy))
4143 if (auto KindCost = Entry->Cost[CostKind])
4144 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4145
4146 if (ST->hasSSE2())
4147 if (const auto *Entry = CostTableLookup(Table: SSE2CostTbl, ISD, Ty: MTy))
4148 if (auto KindCost = Entry->Cost[CostKind])
4149 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4150
4151 if (ST->hasSSE1())
4152 if (const auto *Entry = CostTableLookup(Table: SSE1CostTbl, ISD, Ty: MTy))
4153 if (auto KindCost = Entry->Cost[CostKind])
4154 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4155
4156 // Assume a 3cy latency for fp select ops.
4157 if (CostKind == TTI::TCK_Latency && Opcode == Instruction::Select)
4158 if (ValTy->getScalarType()->isFloatingPointTy())
4159 return 3;
4160
4161 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
4162 Op1Info, Op2Info, I) +
4163 MaskExpCost;
4164}
4165
4166unsigned X86TTIImpl::getAtomicMemIntrinsicMaxElementSize() const { return 16; }
4167
4168/// Returns true if the square root is divided into with the reciprocal
4169/// allowed and the target replaces the pair with the rsqrt* based estimate.
4170static bool isFoldedIntoRsqrtEstimate(const IntrinsicCostAttributes &ICA,
4171 const X86TargetLowering &TLI,
4172 const DataLayout &DL) {
4173 using namespace PatternMatch;
4174 const IntrinsicInst *II = ICA.getInst();
4175 if (!II || !II->hasOneUse() ||
4176 !match(V: II->user_back(), P: m_FDiv(L: m_Value(), R: m_Specific(V: II))) ||
4177 !cast<FPMathOperator>(Val: II->user_back())->hasAllowReciprocal())
4178 return false;
4179 EVT VT = TLI.getValueType(DL, Ty: ICA.getReturnType());
4180 return TLI.hasSqrtEstimate(VT, /*Reciprocal=*/true) &&
4181 TLI.getRecipEstimateSqrtEnabled(VT, F: *II->getFunction()) !=
4182 TargetLoweringBase::ReciprocalEstimate::Disabled;
4183}
4184
4185InstructionCost
4186X86TTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
4187 TTI::TargetCostKind CostKind) const {
4188 // Costs should match the codegen from:
4189 // BITREVERSE: llvm\test\CodeGen\X86\vector-bitreverse.ll
4190 // BSWAP: llvm\test\CodeGen\X86\bswap-vector.ll
4191 // CTLZ: llvm\test\CodeGen\X86\vector-lzcnt-*.ll
4192 // CTPOP: llvm\test\CodeGen\X86\vector-popcnt-*.ll
4193 // CTTZ: llvm\test\CodeGen\X86\vector-tzcnt-*.ll
4194
4195 // TODO: Overflow intrinsics (*ADDO, *SUBO, *MULO) with vector types are not
4196 // specialized in these tables yet.
4197 static const CostKindTblEntry AVX512VBMI2CostTbl[] = {
4198 { .ISD: ISD::FSHL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4199 { .ISD: ISD::FSHL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4200 { .ISD: ISD::FSHL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4201 { .ISD: ISD::FSHL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4202 { .ISD: ISD::FSHL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4203 { .ISD: ISD::FSHL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4204 { .ISD: ISD::FSHL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4205 { .ISD: ISD::FSHL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4206 { .ISD: ISD::FSHL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4207 { .ISD: ISD::ROTL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4208 { .ISD: ISD::ROTL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4209 { .ISD: ISD::ROTL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4210 { .ISD: ISD::ROTR, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4211 { .ISD: ISD::ROTR, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4212 { .ISD: ISD::ROTR, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4213 { .ISD: X86ISD::VROTLI, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4214 { .ISD: X86ISD::VROTLI, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4215 { .ISD: X86ISD::VROTLI, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4216 };
4217 static const CostKindTblEntry AVX512BITALGCostTbl[] = {
4218 { .ISD: ISD::CTPOP, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4219 { .ISD: ISD::CTPOP, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4220 { .ISD: ISD::CTPOP, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4221 { .ISD: ISD::CTPOP, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4222 { .ISD: ISD::CTPOP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4223 { .ISD: ISD::CTPOP, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4224 };
4225 static const CostKindTblEntry AVX512VPOPCNTDQCostTbl[] = {
4226 { .ISD: ISD::CTPOP, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4227 { .ISD: ISD::CTPOP, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4228 { .ISD: ISD::CTPOP, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4229 { .ISD: ISD::CTPOP, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4230 { .ISD: ISD::CTPOP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4231 { .ISD: ISD::CTPOP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4232 };
4233 static const CostKindTblEntry AVX512CDCostTbl[] = {
4234 { .ISD: ISD::CTLZ, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4235 { .ISD: ISD::CTLZ, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4236 { .ISD: ISD::CTLZ, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 27, .CodeSizeCost: 23, .SizeAndLatencyCost: 27 } },
4237 { .ISD: ISD::CTLZ, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 16, .CodeSizeCost: 9, .SizeAndLatencyCost: 11 } },
4238 { .ISD: ISD::CTLZ, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4239 { .ISD: ISD::CTLZ, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4240 { .ISD: ISD::CTLZ, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 19, .CodeSizeCost: 11, .SizeAndLatencyCost: 13 } },
4241 { .ISD: ISD::CTLZ, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 11, .CodeSizeCost: 9, .SizeAndLatencyCost: 10 } },
4242 { .ISD: ISD::CTLZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4243 { .ISD: ISD::CTLZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4244 { .ISD: ISD::CTLZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 15, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
4245 { .ISD: ISD::CTLZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 10, .CodeSizeCost: 9, .SizeAndLatencyCost: 10 } },
4246
4247 { .ISD: ISD::CTTZ, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4248 { .ISD: ISD::CTTZ, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4249 { .ISD: ISD::CTTZ, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4250 { .ISD: ISD::CTTZ, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4251 { .ISD: ISD::CTTZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4252 { .ISD: ISD::CTTZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4253 };
4254 static const CostKindTblEntry AVX512BWCostTbl[] = {
4255 { .ISD: ISD::ABS, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4256 { .ISD: ISD::ABS, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4257 { .ISD: ISD::BITREVERSE, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4258 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4259 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 10, .SizeAndLatencyCost: 14 } },
4260 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4261 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4262 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 10, .SizeAndLatencyCost: 14 } },
4263 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4264 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4265 { .ISD: ISD::BITREVERSE, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 10, .SizeAndLatencyCost: 14 } },
4266 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } },
4267 { .ISD: ISD::BITREVERSE, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } },
4268 { .ISD: ISD::BITREVERSE, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 9, .SizeAndLatencyCost: 12 } },
4269 { .ISD: ISD::BSWAP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4270 { .ISD: ISD::BSWAP, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4271 { .ISD: ISD::BSWAP, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4272 { .ISD: ISD::BSWAP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4273 { .ISD: ISD::BSWAP, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4274 { .ISD: ISD::BSWAP, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4275 { .ISD: ISD::BSWAP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4276 { .ISD: ISD::BSWAP, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4277 { .ISD: ISD::BSWAP, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4278 { .ISD: ISD::CTLZ, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 22, .CodeSizeCost: 23, .SizeAndLatencyCost: 23 } },
4279 { .ISD: ISD::CTLZ, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 23, .CodeSizeCost: 25, .SizeAndLatencyCost: 25 } },
4280 { .ISD: ISD::CTLZ, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 15, .CodeSizeCost: 15, .SizeAndLatencyCost: 16 } },
4281 { .ISD: ISD::CTLZ, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 10, .SizeAndLatencyCost: 9 } },
4282 { .ISD: ISD::CTPOP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 10, .SizeAndLatencyCost: 10 } },
4283 { .ISD: ISD::CTPOP, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 10, .SizeAndLatencyCost: 10 } },
4284 { .ISD: ISD::CTPOP, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 10, .SizeAndLatencyCost: 12 } },
4285 { .ISD: ISD::CTPOP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 11, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4286 { .ISD: ISD::CTPOP, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 11, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4287 { .ISD: ISD::CTPOP, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 12, .CodeSizeCost: 14, .SizeAndLatencyCost: 16 } },
4288 { .ISD: ISD::CTPOP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 11, .SizeAndLatencyCost: 11 } },
4289 { .ISD: ISD::CTPOP, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 11, .SizeAndLatencyCost: 11 } },
4290 { .ISD: ISD::CTPOP, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 11, .SizeAndLatencyCost: 13 } },
4291 { .ISD: ISD::CTPOP, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } },
4292 { .ISD: ISD::CTPOP, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } },
4293 { .ISD: ISD::CTPOP, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 10 } },
4294 { .ISD: ISD::CTTZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4295 { .ISD: ISD::CTTZ, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4296 { .ISD: ISD::CTTZ, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 14, .SizeAndLatencyCost: 16 } },
4297 { .ISD: ISD::CTTZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 11, .SizeAndLatencyCost: 11 } },
4298 { .ISD: ISD::CTTZ, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 11, .SizeAndLatencyCost: 11 } },
4299 { .ISD: ISD::CTTZ, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 11, .SizeAndLatencyCost: 13 } },
4300 { .ISD: ISD::MULHS, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4301 { .ISD: ISD::MULHU, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4302 { .ISD: ISD::ROTL, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 8 } },
4303 { .ISD: ISD::ROTL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4304 { .ISD: ISD::ROTL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4305 { .ISD: ISD::ROTL, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 6, .CodeSizeCost: 11, .SizeAndLatencyCost: 12 } },
4306 { .ISD: ISD::ROTL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 15, .CodeSizeCost: 7, .SizeAndLatencyCost: 10 } },
4307 { .ISD: ISD::ROTL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 15, .CodeSizeCost: 7, .SizeAndLatencyCost: 10 } },
4308 { .ISD: ISD::ROTR, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 8 } },
4309 { .ISD: ISD::ROTR, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4310 { .ISD: ISD::ROTR, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4311 { .ISD: ISD::ROTR, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 6, .CodeSizeCost: 12, .SizeAndLatencyCost: 14 } },
4312 { .ISD: ISD::ROTR, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 14, .CodeSizeCost: 6, .SizeAndLatencyCost: 9 } },
4313 { .ISD: ISD::ROTR, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 14, .CodeSizeCost: 6, .SizeAndLatencyCost: 9 } },
4314 { .ISD: X86ISD::VROTLI, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4315 { .ISD: X86ISD::VROTLI, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4316 { .ISD: X86ISD::VROTLI, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4317 { .ISD: X86ISD::VROTLI, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 3, .SizeAndLatencyCost: 4 } },
4318 { .ISD: X86ISD::VROTLI, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 3, .SizeAndLatencyCost: 4 } },
4319 { .ISD: X86ISD::VROTLI, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 3, .SizeAndLatencyCost: 4 } },
4320 { .ISD: ISD::SADDSAT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4321 { .ISD: ISD::SADDSAT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4322 { .ISD: ISD::SMAX, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4323 { .ISD: ISD::SMAX, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4324 { .ISD: ISD::SMIN, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4325 { .ISD: ISD::SMIN, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4326 { .ISD: ISD::SMULO, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4327 { .ISD: ISD::SMULO, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 21, .CodeSizeCost: 17, .SizeAndLatencyCost: 18 } },
4328 { .ISD: ISD::UMULO, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4329 { .ISD: ISD::UMULO, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 15, .CodeSizeCost: 15, .SizeAndLatencyCost: 16 } },
4330 { .ISD: ISD::SSUBSAT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4331 { .ISD: ISD::SSUBSAT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4332 { .ISD: ISD::UADDSAT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4333 { .ISD: ISD::UADDSAT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4334 { .ISD: ISD::UMAX, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4335 { .ISD: ISD::UMAX, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4336 { .ISD: ISD::UMIN, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4337 { .ISD: ISD::UMIN, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4338 { .ISD: ISD::USUBSAT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4339 { .ISD: ISD::USUBSAT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4340 };
4341 static const CostKindTblEntry AVX512CostTbl[] = {
4342 { .ISD: ISD::ABS, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4343 { .ISD: ISD::ABS, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4344 { .ISD: ISD::ABS, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4345 { .ISD: ISD::ABS, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4346 { .ISD: ISD::ABS, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4347 { .ISD: ISD::ABS, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4348 { .ISD: ISD::ABS, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4349 { .ISD: ISD::ABS, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4350 { .ISD: ISD::ABS, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4351 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 13, .CodeSizeCost: 20, .SizeAndLatencyCost: 20 } },
4352 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 13, .CodeSizeCost: 20, .SizeAndLatencyCost: 20 } },
4353 { .ISD: ISD::BITREVERSE, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 13, .CodeSizeCost: 20, .SizeAndLatencyCost: 20 } },
4354 { .ISD: ISD::BITREVERSE, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 11, .CodeSizeCost: 17, .SizeAndLatencyCost: 17 } },
4355 { .ISD: ISD::BSWAP, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4356 { .ISD: ISD::BSWAP, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4357 { .ISD: ISD::BSWAP, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4358 { .ISD: ISD::CTLZ, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 28, .CodeSizeCost: 32, .SizeAndLatencyCost: 32 } },
4359 { .ISD: ISD::CTLZ, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 30, .CodeSizeCost: 38, .SizeAndLatencyCost: 38 } },
4360 { .ISD: ISD::CTLZ, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 15, .CodeSizeCost: 29, .SizeAndLatencyCost: 29 } },
4361 { .ISD: ISD::CTLZ, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 11, .CodeSizeCost: 19, .SizeAndLatencyCost: 19 } },
4362 { .ISD: ISD::CTPOP, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 16, .CodeSizeCost: 19, .SizeAndLatencyCost: 19 } },
4363 { .ISD: ISD::CTPOP, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 24, .LatencyCost: 19, .CodeSizeCost: 27, .SizeAndLatencyCost: 27 } },
4364 { .ISD: ISD::CTPOP, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 15, .CodeSizeCost: 22, .SizeAndLatencyCost: 22 } },
4365 { .ISD: ISD::CTPOP, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 11, .CodeSizeCost: 16, .SizeAndLatencyCost: 16 } },
4366 { .ISD: ISD::CTTZ, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4367 { .ISD: ISD::CTTZ, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4368 { .ISD: ISD::CTTZ, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 17, .CodeSizeCost: 27, .SizeAndLatencyCost: 27 } },
4369 { .ISD: ISD::CTTZ, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 13, .CodeSizeCost: 21, .SizeAndLatencyCost: 21 } },
4370 { .ISD: ISD::MULHS, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4371 { .ISD: ISD::MULHS, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4372 { .ISD: ISD::MULHS, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4373 { .ISD: ISD::MULHS, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4374 { .ISD: ISD::MULHU, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4375 { .ISD: ISD::MULHU, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4376 { .ISD: ISD::MULHU, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4377 { .ISD: ISD::MULHU, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4378 { .ISD: ISD::ROTL, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4379 { .ISD: ISD::ROTL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4380 { .ISD: ISD::ROTL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4381 { .ISD: ISD::ROTL, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4382 { .ISD: ISD::ROTL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4383 { .ISD: ISD::ROTL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4384 { .ISD: ISD::ROTR, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4385 { .ISD: ISD::ROTR, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4386 { .ISD: ISD::ROTR, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4387 { .ISD: ISD::ROTR, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4388 { .ISD: ISD::ROTR, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4389 { .ISD: ISD::ROTR, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4390 { .ISD: X86ISD::VROTLI, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4391 { .ISD: X86ISD::VROTLI, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4392 { .ISD: X86ISD::VROTLI, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4393 { .ISD: X86ISD::VROTLI, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4394 { .ISD: X86ISD::VROTLI, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4395 { .ISD: X86ISD::VROTLI, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4396 { .ISD: ISD::SADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 8, .SizeAndLatencyCost: 9 } },
4397 { .ISD: ISD::SADDSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4398 { .ISD: ISD::SADDSAT, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4399 { .ISD: ISD::SADDSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4400 { .ISD: ISD::SADDSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4401 { .ISD: ISD::SADDSAT, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4402 { .ISD: ISD::SADDSAT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4403 { .ISD: ISD::SADDSAT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4404 { .ISD: ISD::SMAX, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4405 { .ISD: ISD::SMAX, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4406 { .ISD: ISD::SMAX, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4407 { .ISD: ISD::SMAX, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4408 { .ISD: ISD::SMAX, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4409 { .ISD: ISD::SMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4410 { .ISD: ISD::SMIN, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4411 { .ISD: ISD::SMIN, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4412 { .ISD: ISD::SMIN, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4413 { .ISD: ISD::SMIN, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4414 { .ISD: ISD::SMIN, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4415 { .ISD: ISD::SMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4416 { .ISD: ISD::SMULO, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 44, .LatencyCost: 44, .CodeSizeCost: 81, .SizeAndLatencyCost: 93 } },
4417 { .ISD: ISD::SMULO, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 11 } },
4418 { .ISD: ISD::SMULO, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 12, .CodeSizeCost: 17, .SizeAndLatencyCost: 17 } },
4419 { .ISD: ISD::SMULO, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 22, .LatencyCost: 28, .CodeSizeCost: 42, .SizeAndLatencyCost: 42 } },
4420 { .ISD: ISD::SSUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 13, .CodeSizeCost: 9, .SizeAndLatencyCost: 10 } },
4421 { .ISD: ISD::SSUBSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 15, .CodeSizeCost: 7, .SizeAndLatencyCost: 8 } },
4422 { .ISD: ISD::SSUBSAT, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 14, .CodeSizeCost: 7, .SizeAndLatencyCost: 8 } },
4423 { .ISD: ISD::SSUBSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 14, .CodeSizeCost: 7, .SizeAndLatencyCost: 8 } },
4424 { .ISD: ISD::SSUBSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 15, .CodeSizeCost: 7, .SizeAndLatencyCost: 8 } },
4425 { .ISD: ISD::SSUBSAT, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 14, .CodeSizeCost: 7, .SizeAndLatencyCost: 8 } },
4426 { .ISD: ISD::SSUBSAT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4427 { .ISD: ISD::SSUBSAT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4428 { .ISD: ISD::UMAX, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4429 { .ISD: ISD::UMAX, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4430 { .ISD: ISD::UMAX, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4431 { .ISD: ISD::UMAX, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4432 { .ISD: ISD::UMAX, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4433 { .ISD: ISD::UMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4434 { .ISD: ISD::UMIN, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4435 { .ISD: ISD::UMIN, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4436 { .ISD: ISD::UMIN, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4437 { .ISD: ISD::UMIN, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4438 { .ISD: ISD::UMIN, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4439 { .ISD: ISD::UMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4440 { .ISD: ISD::UMULO, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 52, .LatencyCost: 52, .CodeSizeCost: 95, .SizeAndLatencyCost: 104} },
4441 { .ISD: ISD::UMULO, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 12, .CodeSizeCost: 8, .SizeAndLatencyCost: 10 } },
4442 { .ISD: ISD::UMULO, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 13, .CodeSizeCost: 16, .SizeAndLatencyCost: 16 } },
4443 { .ISD: ISD::UMULO, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 24, .CodeSizeCost: 30, .SizeAndLatencyCost: 30 } },
4444 { .ISD: ISD::UADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4445 { .ISD: ISD::UADDSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4446 { .ISD: ISD::UADDSAT, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4447 { .ISD: ISD::UADDSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4448 { .ISD: ISD::UADDSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4449 { .ISD: ISD::UADDSAT, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4450 { .ISD: ISD::UADDSAT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4451 { .ISD: ISD::UADDSAT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4452 { .ISD: ISD::USUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4453 { .ISD: ISD::USUBSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4454 { .ISD: ISD::USUBSAT, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4455 { .ISD: ISD::USUBSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4456 { .ISD: ISD::USUBSAT, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4457 { .ISD: ISD::USUBSAT, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4458 { .ISD: ISD::USUBSAT, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4459 { .ISD: ISD::FMAXNUM, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4460 { .ISD: ISD::FMAXNUM, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4461 { .ISD: ISD::FMAXNUM, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4462 { .ISD: ISD::FMAXNUM, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4463 { .ISD: ISD::FMAXNUM, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4464 { .ISD: ISD::FMAXNUM, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4465 { .ISD: ISD::FMAXNUM, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4466 { .ISD: ISD::FMAXNUM, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4467 { .ISD: ISD::FSQRT, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
4468 { .ISD: ISD::FSQRT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
4469 { .ISD: ISD::FSQRT, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
4470 { .ISD: ISD::FSQRT, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 20, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // Skylake from http://www.agner.org/
4471 { .ISD: ISD::FSQRT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 18, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
4472 { .ISD: ISD::FSQRT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 18, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
4473 { .ISD: ISD::FSQRT, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 18, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Skylake from http://www.agner.org/
4474 { .ISD: ISD::FSQRT, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 24, .LatencyCost: 32, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // Skylake from http://www.agner.org/
4475 };
4476 static const CostKindTblEntry XOPCostTbl[] = {
4477 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4478 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4479 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4480 { .ISD: ISD::BITREVERSE, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4481 { .ISD: ISD::BITREVERSE, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4482 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4483 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4484 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4485 { .ISD: ISD::BITREVERSE, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 4 } },
4486 { .ISD: ISD::BITREVERSE, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 4 } },
4487 { .ISD: ISD::BITREVERSE, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 4 } },
4488 { .ISD: ISD::BITREVERSE, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 4 } },
4489 // XOP: ROTL = VPROT(X,Y), ROTR = VPROT(X,SUB(0,Y))
4490 { .ISD: ISD::ROTL, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4491 { .ISD: ISD::ROTL, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4492 { .ISD: ISD::ROTL, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4493 { .ISD: ISD::ROTL, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4494 { .ISD: ISD::ROTL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4495 { .ISD: ISD::ROTL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4496 { .ISD: ISD::ROTL, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4497 { .ISD: ISD::ROTL, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4498 { .ISD: ISD::ROTR, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 8, .SizeAndLatencyCost: 9 } },
4499 { .ISD: ISD::ROTR, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 8, .SizeAndLatencyCost: 9 } },
4500 { .ISD: ISD::ROTR, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 8, .SizeAndLatencyCost: 9 } },
4501 { .ISD: ISD::ROTR, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 8, .SizeAndLatencyCost: 9 } },
4502 { .ISD: ISD::ROTR, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4503 { .ISD: ISD::ROTR, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4504 { .ISD: ISD::ROTR, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4505 { .ISD: ISD::ROTR, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4506 { .ISD: X86ISD::VROTLI, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4507 { .ISD: X86ISD::VROTLI, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4508 { .ISD: X86ISD::VROTLI, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4509 { .ISD: X86ISD::VROTLI, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4510 { .ISD: X86ISD::VROTLI, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4511 { .ISD: X86ISD::VROTLI, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4512 { .ISD: X86ISD::VROTLI, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4513 { .ISD: X86ISD::VROTLI, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4514 };
4515 static const CostKindTblEntry AVX2CostTbl[] = {
4516 { .ISD: ISD::ABS, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // VBLENDVPD(X,VPSUBQ(0,X),X)
4517 { .ISD: ISD::ABS, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // VBLENDVPD(X,VPSUBQ(0,X),X)
4518 { .ISD: ISD::ABS, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4519 { .ISD: ISD::ABS, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4520 { .ISD: ISD::ABS, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4521 { .ISD: ISD::ABS, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4522 { .ISD: ISD::ABS, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4523 { .ISD: ISD::ABS, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4524 { .ISD: ISD::BITREVERSE, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4525 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 17 } },
4526 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4527 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 17 } },
4528 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } },
4529 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 10, .SizeAndLatencyCost: 17 } },
4530 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } },
4531 { .ISD: ISD::BITREVERSE, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 5, .CodeSizeCost: 9, .SizeAndLatencyCost: 15 } },
4532 { .ISD: ISD::BSWAP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4533 { .ISD: ISD::BSWAP, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4534 { .ISD: ISD::BSWAP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4535 { .ISD: ISD::BSWAP, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4536 { .ISD: ISD::BSWAP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4537 { .ISD: ISD::BSWAP, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4538 { .ISD: ISD::CTLZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 18, .CodeSizeCost: 24, .SizeAndLatencyCost: 25 } },
4539 { .ISD: ISD::CTLZ, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 18, .CodeSizeCost: 24, .SizeAndLatencyCost: 44 } },
4540 { .ISD: ISD::CTLZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 16, .CodeSizeCost: 19, .SizeAndLatencyCost: 20 } },
4541 { .ISD: ISD::CTLZ, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 16, .CodeSizeCost: 19, .SizeAndLatencyCost: 34 } },
4542 { .ISD: ISD::CTLZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 13, .CodeSizeCost: 14, .SizeAndLatencyCost: 15 } },
4543 { .ISD: ISD::CTLZ, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 14, .CodeSizeCost: 14, .SizeAndLatencyCost: 24 } },
4544 { .ISD: ISD::CTLZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 10 } },
4545 { .ISD: ISD::CTLZ, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 14 } },
4546 { .ISD: ISD::CTPOP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 10, .SizeAndLatencyCost: 10 } },
4547 { .ISD: ISD::CTPOP, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 10, .SizeAndLatencyCost: 14 } },
4548 { .ISD: ISD::CTPOP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 12, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4549 { .ISD: ISD::CTPOP, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 12, .CodeSizeCost: 14, .SizeAndLatencyCost: 18 } },
4550 { .ISD: ISD::CTPOP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 11, .SizeAndLatencyCost: 11 } },
4551 { .ISD: ISD::CTPOP, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 8, .CodeSizeCost: 11, .SizeAndLatencyCost: 18 } },
4552 { .ISD: ISD::CTPOP, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } },
4553 { .ISD: ISD::CTPOP, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 12 } },
4554 { .ISD: ISD::CTTZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 11, .CodeSizeCost: 13, .SizeAndLatencyCost: 13 } },
4555 { .ISD: ISD::CTTZ, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 13, .SizeAndLatencyCost: 20 } },
4556 { .ISD: ISD::CTTZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 14, .CodeSizeCost: 17, .SizeAndLatencyCost: 17 } },
4557 { .ISD: ISD::CTTZ, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 15, .CodeSizeCost: 17, .SizeAndLatencyCost: 24 } },
4558 { .ISD: ISD::CTTZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4559 { .ISD: ISD::CTTZ, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 9, .CodeSizeCost: 14, .SizeAndLatencyCost: 24 } },
4560 { .ISD: ISD::CTTZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 11, .SizeAndLatencyCost: 11 } },
4561 { .ISD: ISD::CTTZ, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 7, .CodeSizeCost: 11, .SizeAndLatencyCost: 18 } },
4562 { .ISD: ISD::MULHS, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 12 } },
4563 { .ISD: ISD::MULHS, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4564 { .ISD: ISD::MULHU, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 12 } },
4565 { .ISD: ISD::MULHU, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4566 { .ISD: ISD::SADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 13, .CodeSizeCost: 8, .SizeAndLatencyCost: 11 } },
4567 { .ISD: ISD::SADDSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 10, .CodeSizeCost: 8, .SizeAndLatencyCost: 12 } },
4568 { .ISD: ISD::SADDSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 9 } },
4569 { .ISD: ISD::SADDSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 13 } },
4570 { .ISD: ISD::SADDSAT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4571 { .ISD: ISD::SADDSAT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4572 { .ISD: ISD::SMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4573 { .ISD: ISD::SMAX, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4574 { .ISD: ISD::SMAX, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4575 { .ISD: ISD::SMAX, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4576 { .ISD: ISD::SMAX, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4577 { .ISD: ISD::SMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4578 { .ISD: ISD::SMIN, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4579 { .ISD: ISD::SMIN, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4580 { .ISD: ISD::SMIN, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4581 { .ISD: ISD::SMIN, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4582 { .ISD: ISD::SMULO, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 20, .LatencyCost: 20, .CodeSizeCost: 33, .SizeAndLatencyCost: 37 } },
4583 { .ISD: ISD::SMULO, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 8, .CodeSizeCost: 13, .SizeAndLatencyCost: 15 } },
4584 { .ISD: ISD::SMULO, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 20, .CodeSizeCost: 13, .SizeAndLatencyCost: 24 } },
4585 { .ISD: ISD::SMULO, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 15, .CodeSizeCost: 11, .SizeAndLatencyCost: 12 } },
4586 { .ISD: ISD::SMULO, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 14, .CodeSizeCost: 8, .SizeAndLatencyCost: 14 } },
4587 { .ISD: ISD::SMULO, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4588 { .ISD: ISD::SMULO, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 15, .CodeSizeCost: 18, .SizeAndLatencyCost: 35 } },
4589 { .ISD: ISD::SMULO, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 22, .CodeSizeCost: 14, .SizeAndLatencyCost: 21 } },
4590 { .ISD: ISD::SSUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 13, .CodeSizeCost: 9, .SizeAndLatencyCost: 13 } },
4591 { .ISD: ISD::SSUBSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 15, .CodeSizeCost: 9, .SizeAndLatencyCost: 13 } },
4592 { .ISD: ISD::SSUBSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 14, .CodeSizeCost: 9, .SizeAndLatencyCost: 11 } },
4593 { .ISD: ISD::SSUBSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 15, .CodeSizeCost: 9, .SizeAndLatencyCost: 16 } },
4594 { .ISD: ISD::SSUBSAT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4595 { .ISD: ISD::SSUBSAT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4596 { .ISD: ISD::UADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4597 { .ISD: ISD::UADDSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 10 } },
4598 { .ISD: ISD::UADDSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 8 } },
4599 { .ISD: ISD::UADDSAT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4600 { .ISD: ISD::UADDSAT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4601 { .ISD: ISD::UMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4602 { .ISD: ISD::UMAX, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 8 } },
4603 { .ISD: ISD::UMAX, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4604 { .ISD: ISD::UMAX, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4605 { .ISD: ISD::UMAX, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4606 { .ISD: ISD::UMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4607 { .ISD: ISD::UMIN, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 8 } },
4608 { .ISD: ISD::UMIN, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4609 { .ISD: ISD::UMIN, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4610 { .ISD: ISD::UMIN, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4611 { .ISD: ISD::UMULO, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 24, .LatencyCost: 24, .CodeSizeCost: 39, .SizeAndLatencyCost: 43 } },
4612 { .ISD: ISD::UMULO, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 10, .CodeSizeCost: 15, .SizeAndLatencyCost: 19 } },
4613 { .ISD: ISD::UMULO, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 11, .CodeSizeCost: 13, .SizeAndLatencyCost: 23 } },
4614 { .ISD: ISD::UMULO, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 12, .CodeSizeCost: 11, .SizeAndLatencyCost: 12 } },
4615 { .ISD: ISD::UMULO, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 8, .SizeAndLatencyCost: 13 } },
4616 { .ISD: ISD::UMULO, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4617 { .ISD: ISD::UMULO, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 13, .CodeSizeCost: 17, .SizeAndLatencyCost: 33 } },
4618 { .ISD: ISD::UMULO, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 19, .CodeSizeCost: 13, .SizeAndLatencyCost: 20 } },
4619 { .ISD: ISD::USUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4620 { .ISD: ISD::USUBSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 6, .SizeAndLatencyCost: 10 } },
4621 { .ISD: ISD::USUBSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
4622 { .ISD: ISD::USUBSAT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4623 { .ISD: ISD::USUBSAT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4624 { .ISD: ISD::FMAXNUM, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // MAXSS + CMPUNORDSS + BLENDVPS
4625 { .ISD: ISD::FMAXNUM, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4626 { .ISD: ISD::FMAXNUM, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 3, .SizeAndLatencyCost: 6 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4627 { .ISD: ISD::FMAXNUM, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // MAXSD + CMPUNORDSD + BLENDVPD
4628 { .ISD: ISD::FMAXNUM, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4629 { .ISD: ISD::FMAXNUM, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 3, .SizeAndLatencyCost: 6 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4630 { .ISD: ISD::FSQRT, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 15, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsqrtss
4631 { .ISD: ISD::FSQRT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 15, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsqrtps
4632 { .ISD: ISD::FSQRT, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 21, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vsqrtps
4633 { .ISD: ISD::FSQRT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 21, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsqrtsd
4634 { .ISD: ISD::FSQRT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 21, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsqrtpd
4635 { .ISD: ISD::FSQRT, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 28, .LatencyCost: 35, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vsqrtpd
4636 };
4637 static const CostKindTblEntry AVX1CostTbl[] = {
4638 { .ISD: ISD::ABS, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 12 } }, // VBLENDVPD(X,VPSUBQ(0,X),X)
4639 { .ISD: ISD::ABS, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } },
4640 { .ISD: ISD::ABS, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } },
4641 { .ISD: ISD::ABS, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } },
4642 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 20, .CodeSizeCost: 20, .SizeAndLatencyCost: 33 } }, // 2 x 128-bit Op + extract/insert
4643 { .ISD: ISD::BITREVERSE, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 13, .CodeSizeCost: 10, .SizeAndLatencyCost: 16 } },
4644 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 20, .CodeSizeCost: 20, .SizeAndLatencyCost: 33 } }, // 2 x 128-bit Op + extract/insert
4645 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 13, .CodeSizeCost: 10, .SizeAndLatencyCost: 16 } },
4646 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 20, .CodeSizeCost: 20, .SizeAndLatencyCost: 33 } }, // 2 x 128-bit Op + extract/insert
4647 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 13, .CodeSizeCost: 10, .SizeAndLatencyCost: 16 } },
4648 { .ISD: ISD::BITREVERSE, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 15, .CodeSizeCost: 17, .SizeAndLatencyCost: 26 } }, // 2 x 128-bit Op + extract/insert
4649 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 9, .SizeAndLatencyCost: 13 } },
4650 { .ISD: ISD::BSWAP, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 10 } },
4651 { .ISD: ISD::BSWAP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
4652 { .ISD: ISD::BSWAP, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 10 } },
4653 { .ISD: ISD::BSWAP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
4654 { .ISD: ISD::BSWAP, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 10 } },
4655 { .ISD: ISD::BSWAP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
4656 { .ISD: ISD::CTLZ, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 29, .LatencyCost: 33, .CodeSizeCost: 49, .SizeAndLatencyCost: 58 } }, // 2 x 128-bit Op + extract/insert
4657 { .ISD: ISD::CTLZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 24, .CodeSizeCost: 24, .SizeAndLatencyCost: 28 } },
4658 { .ISD: ISD::CTLZ, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 24, .LatencyCost: 28, .CodeSizeCost: 39, .SizeAndLatencyCost: 48 } }, // 2 x 128-bit Op + extract/insert
4659 { .ISD: ISD::CTLZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 20, .CodeSizeCost: 19, .SizeAndLatencyCost: 23 } },
4660 { .ISD: ISD::CTLZ, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 19, .LatencyCost: 22, .CodeSizeCost: 29, .SizeAndLatencyCost: 38 } }, // 2 x 128-bit Op + extract/insert
4661 { .ISD: ISD::CTLZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 16, .CodeSizeCost: 14, .SizeAndLatencyCost: 18 } },
4662 { .ISD: ISD::CTLZ, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 15, .CodeSizeCost: 19, .SizeAndLatencyCost: 28 } }, // 2 x 128-bit Op + extract/insert
4663 { .ISD: ISD::CTLZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 13 } },
4664 { .ISD: ISD::CTPOP, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 18, .CodeSizeCost: 19, .SizeAndLatencyCost: 28 } }, // 2 x 128-bit Op + extract/insert
4665 { .ISD: ISD::CTPOP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 14, .CodeSizeCost: 10, .SizeAndLatencyCost: 14 } },
4666 { .ISD: ISD::CTPOP, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 24, .CodeSizeCost: 27, .SizeAndLatencyCost: 36 } }, // 2 x 128-bit Op + extract/insert
4667 { .ISD: ISD::CTPOP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 20, .CodeSizeCost: 14, .SizeAndLatencyCost: 18 } },
4668 { .ISD: ISD::CTPOP, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 21, .CodeSizeCost: 22, .SizeAndLatencyCost: 31 } }, // 2 x 128-bit Op + extract/insert
4669 { .ISD: ISD::CTPOP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 18, .CodeSizeCost: 11, .SizeAndLatencyCost: 15 } },
4670 { .ISD: ISD::CTPOP, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 15, .CodeSizeCost: 16, .SizeAndLatencyCost: 25 } }, // 2 x 128-bit Op + extract/insert
4671 { .ISD: ISD::CTPOP, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 12, .CodeSizeCost: 8, .SizeAndLatencyCost: 12 } },
4672 { .ISD: ISD::CTTZ, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 17, .LatencyCost: 22, .CodeSizeCost: 24, .SizeAndLatencyCost: 33 } }, // 2 x 128-bit Op + extract/insert
4673 { .ISD: ISD::CTTZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 19, .CodeSizeCost: 13, .SizeAndLatencyCost: 17 } },
4674 { .ISD: ISD::CTTZ, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 21, .LatencyCost: 27, .CodeSizeCost: 32, .SizeAndLatencyCost: 41 } }, // 2 x 128-bit Op + extract/insert
4675 { .ISD: ISD::CTTZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 24, .CodeSizeCost: 17, .SizeAndLatencyCost: 21 } },
4676 { .ISD: ISD::CTTZ, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 24, .CodeSizeCost: 27, .SizeAndLatencyCost: 36 } }, // 2 x 128-bit Op + extract/insert
4677 { .ISD: ISD::CTTZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 21, .CodeSizeCost: 14, .SizeAndLatencyCost: 18 } },
4678 { .ISD: ISD::CTTZ, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 18, .CodeSizeCost: 21, .SizeAndLatencyCost: 30 } }, // 2 x 128-bit Op + extract/insert
4679 { .ISD: ISD::CTTZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 16, .CodeSizeCost: 11, .SizeAndLatencyCost: 15 } },
4680 { .ISD: ISD::MULHS, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 11, .CodeSizeCost: 14, .SizeAndLatencyCost: 18 } },
4681 { .ISD: ISD::MULHS, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4682 { .ISD: ISD::MULHU, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 11, .CodeSizeCost: 14, .SizeAndLatencyCost: 18 } },
4683 { .ISD: ISD::MULHU, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } },
4684 { .ISD: ISD::SADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 13, .CodeSizeCost: 8, .SizeAndLatencyCost: 11 } },
4685 { .ISD: ISD::SADDSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 20, .CodeSizeCost: 15, .SizeAndLatencyCost: 25 } }, // 2 x 128-bit Op + extract/insert
4686 { .ISD: ISD::SADDSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 18, .CodeSizeCost: 14, .SizeAndLatencyCost: 24 } }, // 2 x 128-bit Op + extract/insert
4687 { .ISD: ISD::SADDSAT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4688 { .ISD: ISD::SADDSAT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4689 { .ISD: ISD::SMAX, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 12 } }, // 2 x 128-bit Op + extract/insert
4690 { .ISD: ISD::SMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
4691 { .ISD: ISD::SMAX, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4692 { .ISD: ISD::SMAX, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4693 { .ISD: ISD::SMAX, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4694 { .ISD: ISD::SMIN, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 12 } }, // 2 x 128-bit Op + extract/insert
4695 { .ISD: ISD::SMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4696 { .ISD: ISD::SMIN, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4697 { .ISD: ISD::SMIN, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4698 { .ISD: ISD::SMIN, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4699 { .ISD: ISD::SMULO, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 20, .LatencyCost: 20, .CodeSizeCost: 33, .SizeAndLatencyCost: 37 } },
4700 { .ISD: ISD::SMULO, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 9, .CodeSizeCost: 13, .SizeAndLatencyCost: 17 } },
4701 { .ISD: ISD::SMULO, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 20, .CodeSizeCost: 24, .SizeAndLatencyCost: 29 } },
4702 { .ISD: ISD::SMULO, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 15, .CodeSizeCost: 11, .SizeAndLatencyCost: 13 } },
4703 { .ISD: ISD::SMULO, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 14, .CodeSizeCost: 14, .SizeAndLatencyCost: 15 } },
4704 { .ISD: ISD::SMULO, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4705 { .ISD: ISD::SMULO, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 20, .LatencyCost: 20, .CodeSizeCost: 37, .SizeAndLatencyCost: 39 } },
4706 { .ISD: ISD::SMULO, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 22, .CodeSizeCost: 18, .SizeAndLatencyCost: 21 } },
4707 { .ISD: ISD::SSUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 13, .CodeSizeCost: 9, .SizeAndLatencyCost: 13 } },
4708 { .ISD: ISD::SSUBSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 21, .CodeSizeCost: 18, .SizeAndLatencyCost: 29 } }, // 2 x 128-bit Op + extract/insert
4709 { .ISD: ISD::SSUBSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 19, .CodeSizeCost: 18, .SizeAndLatencyCost: 29 } }, // 2 x 128-bit Op + extract/insert
4710 { .ISD: ISD::SSUBSAT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4711 { .ISD: ISD::SSUBSAT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4712 { .ISD: ISD::UADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4713 { .ISD: ISD::UADDSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 11, .CodeSizeCost: 14, .SizeAndLatencyCost: 15 } }, // 2 x 128-bit Op + extract/insert
4714 { .ISD: ISD::UADDSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 6, .CodeSizeCost: 10, .SizeAndLatencyCost: 11 } }, // 2 x 128-bit Op + extract/insert
4715 { .ISD: ISD::UADDSAT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4716 { .ISD: ISD::UADDSAT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4717 { .ISD: ISD::UMAX, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 10, .CodeSizeCost: 11, .SizeAndLatencyCost: 17 } }, // 2 x 128-bit Op + extract/insert
4718 { .ISD: ISD::UMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } },
4719 { .ISD: ISD::UMAX, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4720 { .ISD: ISD::UMAX, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4721 { .ISD: ISD::UMAX, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4722 { .ISD: ISD::UMIN, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 10, .CodeSizeCost: 11, .SizeAndLatencyCost: 17 } }, // 2 x 128-bit Op + extract/insert
4723 { .ISD: ISD::UMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 7 } },
4724 { .ISD: ISD::UMIN, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4725 { .ISD: ISD::UMIN, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4726 { .ISD: ISD::UMIN, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4727 { .ISD: ISD::UMULO, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 24, .LatencyCost: 26, .CodeSizeCost: 39, .SizeAndLatencyCost: 45 } },
4728 { .ISD: ISD::UMULO, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 12, .CodeSizeCost: 15, .SizeAndLatencyCost: 20 } },
4729 { .ISD: ISD::UMULO, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 15, .CodeSizeCost: 23, .SizeAndLatencyCost: 28 } },
4730 { .ISD: ISD::UMULO, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 12, .CodeSizeCost: 11, .SizeAndLatencyCost: 13 } },
4731 { .ISD: ISD::UMULO, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 11, .CodeSizeCost: 13, .SizeAndLatencyCost: 14 } },
4732 { .ISD: ISD::UMULO, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4733 { .ISD: ISD::UMULO, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 19, .LatencyCost: 19, .CodeSizeCost: 35, .SizeAndLatencyCost: 37 } },
4734 { .ISD: ISD::UMULO, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 19, .CodeSizeCost: 17, .SizeAndLatencyCost: 20 } },
4735 { .ISD: ISD::USUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4736 { .ISD: ISD::USUBSAT, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 10, .CodeSizeCost: 14, .SizeAndLatencyCost: 15 } }, // 2 x 128-bit Op + extract/insert
4737 { .ISD: ISD::USUBSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 7, .SizeAndLatencyCost: 8 } }, // 2 x 128-bit Op + extract/insert
4738 { .ISD: ISD::USUBSAT, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4739 { .ISD: ISD::USUBSAT, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4740 { .ISD: ISD::USUBSAT, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // 2 x 128-bit Op + extract/insert
4741 { .ISD: ISD::FMAXNUM, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // MAXSS + CMPUNORDSS + BLENDVPS
4742 { .ISD: ISD::FMAXNUM, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4743 { .ISD: ISD::FMAXNUM, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 7, .CodeSizeCost: 3, .SizeAndLatencyCost: 10 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4744 { .ISD: ISD::FMAXNUM, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // MAXSD + CMPUNORDSD + BLENDVPD
4745 { .ISD: ISD::FMAXNUM, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4746 { .ISD: ISD::FMAXNUM, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 7, .CodeSizeCost: 3, .SizeAndLatencyCost: 10 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4747 { .ISD: ISD::FSQRT, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 21, .LatencyCost: 21, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsqrtss
4748 { .ISD: ISD::FSQRT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 21, .LatencyCost: 21, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsqrtps
4749 { .ISD: ISD::FSQRT, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 42, .LatencyCost: 42, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vsqrtps
4750 { .ISD: ISD::FSQRT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 27, .LatencyCost: 27, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsqrtsd
4751 { .ISD: ISD::FSQRT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 27, .LatencyCost: 27, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // vsqrtpd
4752 { .ISD: ISD::FSQRT, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 54, .LatencyCost: 54, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } }, // vsqrtpd
4753 };
4754 static const CostKindTblEntry GFNICostTbl[] = {
4755 { .ISD: ISD::BITREVERSE, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4756 { .ISD: ISD::BITREVERSE, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } }, // gf2p8affineqb
4757 { .ISD: ISD::BITREVERSE, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // gf2p8affineqb
4758 { .ISD: ISD::BITREVERSE, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } }, // gf2p8affineqb
4759 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
4760 { .ISD: ISD::BITREVERSE, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
4761 { .ISD: ISD::BITREVERSE, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
4762 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4763 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4764 { .ISD: ISD::BITREVERSE, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4765 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4766 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4767 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4768 { .ISD: ISD::BITREVERSE, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 8, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4769 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4770 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 9, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } }, // gf2p8affineqb
4771 { .ISD: X86ISD::VROTLI, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
4772 { .ISD: X86ISD::VROTLI, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
4773 { .ISD: X86ISD::VROTLI, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 6, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // gf2p8affineqb
4774 };
4775 static const CostKindTblEntry GLMCostTbl[] = {
4776 { .ISD: ISD::FSQRT, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 19, .LatencyCost: 20, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sqrtss
4777 { .ISD: ISD::FSQRT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 37, .LatencyCost: 41, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } }, // sqrtps
4778 { .ISD: ISD::FSQRT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 34, .LatencyCost: 35, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sqrtsd
4779 { .ISD: ISD::FSQRT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 67, .LatencyCost: 71, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } }, // sqrtpd
4780 };
4781 static const CostKindTblEntry SLMCostTbl[] = {
4782 { .ISD: ISD::BSWAP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } },
4783 { .ISD: ISD::BSWAP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } },
4784 { .ISD: ISD::BSWAP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } },
4785 { .ISD: ISD::FSQRT, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 20, .LatencyCost: 20, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sqrtss
4786 { .ISD: ISD::FSQRT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 40, .LatencyCost: 41, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } }, // sqrtps
4787 { .ISD: ISD::FSQRT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 35, .LatencyCost: 35, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // sqrtsd
4788 { .ISD: ISD::FSQRT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 70, .LatencyCost: 71, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } }, // sqrtpd
4789 };
4790 static const CostKindTblEntry SSE42CostTbl[] = {
4791 { .ISD: ISD::FMAXNUM, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } }, // MAXSS + CMPUNORDSS + BLENDVPS
4792 { .ISD: ISD::FMAXNUM, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4793 { .ISD: ISD::FMAXNUM, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } }, // MAXSD + CMPUNORDSD + BLENDVPD
4794 { .ISD: ISD::FMAXNUM, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4795 { .ISD: ISD::FSQRT, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 18, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
4796 { .ISD: ISD::FSQRT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 18, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
4797 };
4798 static const CostKindTblEntry SSE41CostTbl[] = {
4799 { .ISD: ISD::ABS, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 3, .SizeAndLatencyCost: 5 } }, // BLENDVPD(X,PSUBQ(0,X),X)
4800 { .ISD: ISD::MULHS, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4801 { .ISD: ISD::MULHU, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4802 { .ISD: ISD::SADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 14, .CodeSizeCost: 17, .SizeAndLatencyCost: 21 } },
4803 { .ISD: ISD::SADDSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 8, .SizeAndLatencyCost: 10 } },
4804 { .ISD: ISD::SSUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 19, .CodeSizeCost: 25, .SizeAndLatencyCost: 29 } },
4805 { .ISD: ISD::SSUBSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 14, .CodeSizeCost: 10, .SizeAndLatencyCost: 12 } },
4806 { .ISD: ISD::SMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4807 { .ISD: ISD::SMAX, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4808 { .ISD: ISD::SMAX, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4809 { .ISD: ISD::SMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4810 { .ISD: ISD::SMIN, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4811 { .ISD: ISD::SMIN, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4812 { .ISD: ISD::SMULO, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 11, .CodeSizeCost: 13, .SizeAndLatencyCost: 17 } },
4813 { .ISD: ISD::SMULO, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 20, .LatencyCost: 24, .CodeSizeCost: 13, .SizeAndLatencyCost: 19 } },
4814 { .ISD: ISD::SMULO, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 9, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } },
4815 { .ISD: ISD::SMULO, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 22, .CodeSizeCost: 24, .SizeAndLatencyCost: 25 } },
4816 { .ISD: ISD::UADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 13, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4817 { .ISD: ISD::UADDSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4818 { .ISD: ISD::USUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 10, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4819 { .ISD: ISD::USUBSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } },
4820 { .ISD: ISD::UMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 11, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4821 { .ISD: ISD::UMAX, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4822 { .ISD: ISD::UMAX, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4823 { .ISD: ISD::UMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 11, .CodeSizeCost: 6, .SizeAndLatencyCost: 7 } },
4824 { .ISD: ISD::UMIN, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4825 { .ISD: ISD::UMIN, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4826 { .ISD: ISD::UMULO, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 20, .CodeSizeCost: 15, .SizeAndLatencyCost: 20 } },
4827 { .ISD: ISD::UMULO, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 19, .LatencyCost: 22, .CodeSizeCost: 12, .SizeAndLatencyCost: 18 } },
4828 { .ISD: ISD::UMULO, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
4829 { .ISD: ISD::UMULO, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 19, .CodeSizeCost: 18, .SizeAndLatencyCost: 20 } },
4830 };
4831 static const CostKindTblEntry SSSE3CostTbl[] = {
4832 { .ISD: ISD::ABS, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4833 { .ISD: ISD::ABS, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4834 { .ISD: ISD::ABS, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4835 { .ISD: ISD::BITREVERSE, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 20, .CodeSizeCost: 11, .SizeAndLatencyCost: 21 } },
4836 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 20, .CodeSizeCost: 11, .SizeAndLatencyCost: 21 } },
4837 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 20, .CodeSizeCost: 11, .SizeAndLatencyCost: 21 } },
4838 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 12, .CodeSizeCost: 10, .SizeAndLatencyCost: 16 } },
4839 { .ISD: ISD::BSWAP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } },
4840 { .ISD: ISD::BSWAP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } },
4841 { .ISD: ISD::BSWAP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 5 } },
4842 { .ISD: ISD::CTLZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 28, .CodeSizeCost: 28, .SizeAndLatencyCost: 35 } },
4843 { .ISD: ISD::CTLZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 20, .CodeSizeCost: 22, .SizeAndLatencyCost: 28 } },
4844 { .ISD: ISD::CTLZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 17, .CodeSizeCost: 16, .SizeAndLatencyCost: 22 } },
4845 { .ISD: ISD::CTLZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 15, .CodeSizeCost: 10, .SizeAndLatencyCost: 16 } },
4846 { .ISD: ISD::CTPOP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 19, .CodeSizeCost: 12, .SizeAndLatencyCost: 18 } },
4847 { .ISD: ISD::CTPOP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 24, .CodeSizeCost: 16, .SizeAndLatencyCost: 22 } },
4848 { .ISD: ISD::CTPOP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 18, .CodeSizeCost: 14, .SizeAndLatencyCost: 20 } },
4849 { .ISD: ISD::CTPOP, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 12, .CodeSizeCost: 10, .SizeAndLatencyCost: 16 } },
4850 { .ISD: ISD::CTTZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 25, .CodeSizeCost: 15, .SizeAndLatencyCost: 22 } },
4851 { .ISD: ISD::CTTZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 26, .CodeSizeCost: 19, .SizeAndLatencyCost: 25 } },
4852 { .ISD: ISD::CTTZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 20, .CodeSizeCost: 17, .SizeAndLatencyCost: 23 } },
4853 { .ISD: ISD::CTTZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 16, .CodeSizeCost: 13, .SizeAndLatencyCost: 19 } }
4854 };
4855 static const CostKindTblEntry SSE2CostTbl[] = {
4856 { .ISD: ISD::ABS, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4857 { .ISD: ISD::ABS, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4 } },
4858 { .ISD: ISD::ABS, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4859 { .ISD: ISD::ABS, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4860 { .ISD: ISD::BITREVERSE, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 20, .CodeSizeCost: 32, .SizeAndLatencyCost: 32 } },
4861 { .ISD: ISD::BITREVERSE, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 20, .CodeSizeCost: 30, .SizeAndLatencyCost: 30 } },
4862 { .ISD: ISD::BITREVERSE, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 20, .CodeSizeCost: 25, .SizeAndLatencyCost: 25 } },
4863 { .ISD: ISD::BITREVERSE, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 11, .LatencyCost: 12, .CodeSizeCost: 21, .SizeAndLatencyCost: 21 } },
4864 { .ISD: ISD::BSWAP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 6, .CodeSizeCost: 11, .SizeAndLatencyCost: 11 } },
4865 { .ISD: ISD::BSWAP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 9, .SizeAndLatencyCost: 9 } },
4866 { .ISD: ISD::BSWAP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } },
4867 { .ISD: ISD::CTLZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 45, .CodeSizeCost: 36, .SizeAndLatencyCost: 38 } },
4868 { .ISD: ISD::CTLZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 45, .CodeSizeCost: 38, .SizeAndLatencyCost: 40 } },
4869 { .ISD: ISD::CTLZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 38, .CodeSizeCost: 32, .SizeAndLatencyCost: 34 } },
4870 { .ISD: ISD::CTLZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 39, .CodeSizeCost: 29, .SizeAndLatencyCost: 32 } },
4871 { .ISD: ISD::CTPOP, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 26, .CodeSizeCost: 16, .SizeAndLatencyCost: 18 } },
4872 { .ISD: ISD::CTPOP, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 15, .LatencyCost: 29, .CodeSizeCost: 21, .SizeAndLatencyCost: 23 } },
4873 { .ISD: ISD::CTPOP, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 25, .CodeSizeCost: 18, .SizeAndLatencyCost: 20 } },
4874 { .ISD: ISD::CTPOP, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 21, .CodeSizeCost: 14, .SizeAndLatencyCost: 16 } },
4875 { .ISD: ISD::CTTZ, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 14, .LatencyCost: 28, .CodeSizeCost: 19, .SizeAndLatencyCost: 21 } },
4876 { .ISD: ISD::CTTZ, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 18, .LatencyCost: 31, .CodeSizeCost: 24, .SizeAndLatencyCost: 26 } },
4877 { .ISD: ISD::CTTZ, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 27, .CodeSizeCost: 21, .SizeAndLatencyCost: 23 } },
4878 { .ISD: ISD::CTTZ, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 23, .CodeSizeCost: 17, .SizeAndLatencyCost: 19 } },
4879 { .ISD: ISD::MULHS, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 11, .CodeSizeCost: 15, .SizeAndLatencyCost: 15 } },
4880 { .ISD: ISD::MULHS, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4881 { .ISD: ISD::MULHU, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
4882 { .ISD: ISD::MULHU, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 5, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4883 { .ISD: ISD::SADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 12, .LatencyCost: 14, .CodeSizeCost: 24, .SizeAndLatencyCost: 24 } },
4884 { .ISD: ISD::SADDSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 11, .CodeSizeCost: 11, .SizeAndLatencyCost: 12 } },
4885 { .ISD: ISD::SADDSAT, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4886 { .ISD: ISD::SADDSAT, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4887 { .ISD: ISD::SMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 15, .SizeAndLatencyCost: 15 } },
4888 { .ISD: ISD::SMAX, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4889 { .ISD: ISD::SMAX, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4890 { .ISD: ISD::SMAX, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4891 { .ISD: ISD::SMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 15, .SizeAndLatencyCost: 15 } },
4892 { .ISD: ISD::SMIN, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4893 { .ISD: ISD::SMIN, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4894 { .ISD: ISD::SMIN, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5 } },
4895 { .ISD: ISD::SMULO, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 30, .LatencyCost: 33, .CodeSizeCost: 13, .SizeAndLatencyCost: 23 } },
4896 { .ISD: ISD::SMULO, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 20, .LatencyCost: 24, .CodeSizeCost: 23, .SizeAndLatencyCost: 23 } },
4897 { .ISD: ISD::SMULO, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 10, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } },
4898 { .ISD: ISD::SMULO, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 23, .CodeSizeCost: 24, .SizeAndLatencyCost: 25 } },
4899 { .ISD: ISD::SSUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 16, .LatencyCost: 19, .CodeSizeCost: 31, .SizeAndLatencyCost: 31 } },
4900 { .ISD: ISD::SSUBSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 14, .CodeSizeCost: 12, .SizeAndLatencyCost: 13 } },
4901 { .ISD: ISD::SSUBSAT, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4902 { .ISD: ISD::SSUBSAT, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4903 { .ISD: ISD::UADDSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 13, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4904 { .ISD: ISD::UADDSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 5, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
4905 { .ISD: ISD::UADDSAT, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4906 { .ISD: ISD::UADDSAT, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4907 { .ISD: ISD::UMAX, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 15, .SizeAndLatencyCost: 15 } },
4908 { .ISD: ISD::UMAX, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } },
4909 { .ISD: ISD::UMAX, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4910 { .ISD: ISD::UMAX, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4911 { .ISD: ISD::UMIN, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 15, .SizeAndLatencyCost: 15 } },
4912 { .ISD: ISD::UMIN, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 8 } },
4913 { .ISD: ISD::UMIN, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } },
4914 { .ISD: ISD::UMIN, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4915 { .ISD: ISD::UMULO, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 30, .LatencyCost: 33, .CodeSizeCost: 15, .SizeAndLatencyCost: 29 } },
4916 { .ISD: ISD::UMULO, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 19, .LatencyCost: 22, .CodeSizeCost: 14, .SizeAndLatencyCost: 18 } },
4917 { .ISD: ISD::UMULO, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
4918 { .ISD: ISD::UMULO, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 13, .LatencyCost: 19, .CodeSizeCost: 20, .SizeAndLatencyCost: 20 } },
4919 { .ISD: ISD::USUBSAT, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 10, .CodeSizeCost: 14, .SizeAndLatencyCost: 14 } },
4920 { .ISD: ISD::USUBSAT, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
4921 { .ISD: ISD::USUBSAT, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4922 { .ISD: ISD::USUBSAT, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4923 { .ISD: ISD::FMAXNUM, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
4924 { .ISD: ISD::FMAXNUM, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4925 { .ISD: ISD::FSQRT, .Type: MVT::f64, .Cost: { .RecipThroughputCost: 32, .LatencyCost: 32, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
4926 { .ISD: ISD::FSQRT, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 32, .LatencyCost: 32, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // Nehalem from http://www.agner.org/
4927 };
4928 static const CostKindTblEntry SSE1CostTbl[] = {
4929 { .ISD: ISD::FMAXNUM, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 7, .SizeAndLatencyCost: 7 } },
4930 { .ISD: ISD::FMAXNUM, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 6, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
4931 { .ISD: ISD::FSQRT, .Type: MVT::f32, .Cost: { .RecipThroughputCost: 28, .LatencyCost: 30, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // Pentium III from http://www.agner.org/
4932 { .ISD: ISD::FSQRT, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 56, .LatencyCost: 56, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // Pentium III from http://www.agner.org/
4933 };
4934 static const CostKindTblEntry BMI64CostTbl[] = { // 64-bit targets
4935 { .ISD: ISD::CTTZ, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4936 };
4937 static const CostKindTblEntry BMI32CostTbl[] = { // 32 or 64-bit targets
4938 { .ISD: ISD::CTTZ, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4939 { .ISD: ISD::CTTZ, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4940 { .ISD: ISD::CTTZ, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4941 };
4942 static const CostKindTblEntry LZCNT64CostTbl[] = { // 64-bit targets
4943 { .ISD: ISD::CTLZ, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4944 };
4945 static const CostKindTblEntry LZCNT32CostTbl[] = { // 32 or 64-bit targets
4946 { .ISD: ISD::CTLZ, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4947 { .ISD: ISD::CTLZ, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4948 { .ISD: ISD::CTLZ, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4949 };
4950 static const CostKindTblEntry POPCNT64CostTbl[] = { // 64-bit targets
4951 { .ISD: ISD::CTPOP, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // popcnt
4952 };
4953 static const CostKindTblEntry POPCNT32CostTbl[] = { // 32 or 64-bit targets
4954 { .ISD: ISD::CTPOP, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } }, // popcnt
4955 { .ISD: ISD::CTPOP, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // popcnt(zext())
4956 { .ISD: ISD::CTPOP, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // popcnt(zext())
4957 };
4958 static const CostKindTblEntry PCLMULCostTbl[] = {
4959 { .ISD: ISD::CLMUL, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 4, .SizeAndLatencyCost: 8 } }, // MOV+2xPCLMUL+unpack
4960 { .ISD: ISD::CLMUL, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 18, .CodeSizeCost: 12, .SizeAndLatencyCost: 16 } }, // MOV+4xPCLMUL+unpack
4961 { .ISD: ISD::CLMUL, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 4, .SizeAndLatencyCost: 8 } }, // MOV+PCLMUL+MOV
4962 { .ISD: ISD::CLMUL, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 4, .SizeAndLatencyCost: 8 } }, // MOV+PCLMUL+MOV
4963 { .ISD: ISD::CLMUL, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 4, .SizeAndLatencyCost: 8 } }, // MOV+PCLMUL+MOV
4964 { .ISD: ISD::CLMUL, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 12, .CodeSizeCost: 4, .SizeAndLatencyCost: 8 } }, // MOV+PCLMUL+MOV
4965 };
4966 static const CostKindTblEntry X64CostTbl[] = { // 64-bit targets
4967 { .ISD: ISD::ABS, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // SUB+CMOV
4968 { .ISD: ISD::BITREVERSE, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 12, .CodeSizeCost: 20, .SizeAndLatencyCost: 22 } },
4969 { .ISD: ISD::BSWAP, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } },
4970 { .ISD: ISD::CTLZ, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // MOV+BSR+XOR
4971 { .ISD: ISD::CTLZ, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // MOV+BSR+XOR
4972 { .ISD: ISD::CTLZ, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // MOV+BSR+XOR
4973 { .ISD: ISD::CTLZ, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 3 } }, // MOV+BSR+XOR
4974 { .ISD: ISD::CTLZ_ZERO_POISON,.Type: MVT::i64,.Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // BSR+XOR
4975 { .ISD: ISD::CTTZ, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // MOV+BSF
4976 { .ISD: ISD::CTTZ, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // MOV+BSF
4977 { .ISD: ISD::CTTZ, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // MOV+BSF
4978 { .ISD: ISD::CTTZ, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // MOV+BSF
4979 { .ISD: ISD::CTTZ_ZERO_POISON,.Type: MVT::i64,.Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BSF
4980 { .ISD: ISD::CTPOP, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 10, .LatencyCost: 6, .CodeSizeCost: 19, .SizeAndLatencyCost: 19 } },
4981 { .ISD: ISD::ROTL, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
4982 { .ISD: ISD::ROTR, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
4983 { .ISD: X86ISD::VROTLI, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
4984 { .ISD: ISD::FSHL, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 4 } },
4985 { .ISD: ISD::SADDSAT, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 7, .SizeAndLatencyCost: 10 } },
4986 { .ISD: ISD::SSUBSAT, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 11 } },
4987 { .ISD: ISD::UADDSAT, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 7 } },
4988 { .ISD: ISD::USUBSAT, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 7 } },
4989 { .ISD: ISD::SMAX, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4990 { .ISD: ISD::SMIN, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4991 { .ISD: ISD::UMAX, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4992 { .ISD: ISD::UMIN, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 3, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
4993 { .ISD: ISD::SADDO, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
4994 { .ISD: ISD::UADDO, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
4995 { .ISD: ISD::SMULO, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
4996 { .ISD: ISD::UMULO, .Type: MVT::i64, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 8, .CodeSizeCost: 4, .SizeAndLatencyCost: 7 } },
4997 };
4998 static const CostKindTblEntry X86CostTbl[] = { // 32 or 64-bit targets
4999 { .ISD: ISD::ABS, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // SUB+XOR+SRA or SUB+CMOV
5000 { .ISD: ISD::ABS, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // SUB+XOR+SRA or SUB+CMOV
5001 { .ISD: ISD::ABS, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 3 } }, // SUB+XOR+SRA
5002 { .ISD: ISD::BITREVERSE, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 12, .CodeSizeCost: 17, .SizeAndLatencyCost: 19 } },
5003 { .ISD: ISD::BITREVERSE, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 12, .CodeSizeCost: 17, .SizeAndLatencyCost: 19 } },
5004 { .ISD: ISD::BITREVERSE, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 9, .CodeSizeCost: 13, .SizeAndLatencyCost: 14 } },
5005 { .ISD: ISD::BSWAP, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
5006 { .ISD: ISD::BSWAP, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // ROL
5007 { .ISD: ISD::CTLZ, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // BSR+XOR or BSR+XOR+CMOV
5008 { .ISD: ISD::CTLZ, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 5 } }, // BSR+XOR or BSR+XOR+CMOV
5009 { .ISD: ISD::CTLZ, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6 } }, // BSR+XOR or BSR+XOR+CMOV
5010 { .ISD: ISD::CTLZ_ZERO_POISON,.Type: MVT::i32,.Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // BSR+XOR
5011 { .ISD: ISD::CTLZ_ZERO_POISON,.Type: MVT::i16,.Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2 } }, // BSR+XOR
5012 { .ISD: ISD::CTLZ_ZERO_POISON,.Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // BSR+XOR
5013 { .ISD: ISD::CTTZ, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3 } }, // TEST+BSF+CMOV/BRANCH
5014 { .ISD: ISD::CTTZ, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // TEST+BSF+CMOV/BRANCH
5015 { .ISD: ISD::CTTZ, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } }, // TEST+BSF+CMOV/BRANCH
5016 { .ISD: ISD::CTTZ_ZERO_POISON,.Type: MVT::i32,.Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BSF
5017 { .ISD: ISD::CTTZ_ZERO_POISON,.Type: MVT::i16,.Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BSF
5018 { .ISD: ISD::CTTZ_ZERO_POISON,.Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 1, .SizeAndLatencyCost: 2 } }, // BSF
5019 { .ISD: ISD::CTPOP, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 8, .LatencyCost: 7, .CodeSizeCost: 15, .SizeAndLatencyCost: 15 } },
5020 { .ISD: ISD::CTPOP, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 8, .CodeSizeCost: 17, .SizeAndLatencyCost: 17 } },
5021 { .ISD: ISD::CTPOP, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 6, .CodeSizeCost: 6, .SizeAndLatencyCost: 6 } },
5022 { .ISD: ISD::ROTL, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
5023 { .ISD: ISD::ROTL, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
5024 { .ISD: ISD::ROTL, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
5025 { .ISD: ISD::ROTR, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
5026 { .ISD: ISD::ROTR, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
5027 { .ISD: ISD::ROTR, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
5028 { .ISD: X86ISD::VROTLI, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
5029 { .ISD: X86ISD::VROTLI, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
5030 { .ISD: X86ISD::VROTLI, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1 } },
5031 { .ISD: ISD::FSHL, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 1, .SizeAndLatencyCost: 4 } },
5032 { .ISD: ISD::FSHL, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 5 } },
5033 { .ISD: ISD::FSHL, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 5 } },
5034 { .ISD: ISD::SADDSAT, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 6, .SizeAndLatencyCost: 9 } },
5035 { .ISD: ISD::SADDSAT, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 7, .SizeAndLatencyCost: 10 } },
5036 { .ISD: ISD::SADDSAT, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 11 } },
5037 { .ISD: ISD::SSUBSAT, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 7, .SizeAndLatencyCost: 10 } },
5038 { .ISD: ISD::SSUBSAT, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 7, .SizeAndLatencyCost: 10 } },
5039 { .ISD: ISD::SSUBSAT, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 4, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 11 } },
5040 { .ISD: ISD::UADDSAT, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 7 } },
5041 { .ISD: ISD::UADDSAT, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 7 } },
5042 { .ISD: ISD::UADDSAT, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 8 } },
5043 { .ISD: ISD::USUBSAT, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 7 } },
5044 { .ISD: ISD::USUBSAT, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 7 } },
5045 { .ISD: ISD::USUBSAT, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 8 } },
5046 { .ISD: ISD::SMAX, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
5047 { .ISD: ISD::SMAX, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
5048 { .ISD: ISD::SMAX, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
5049 { .ISD: ISD::SMIN, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
5050 { .ISD: ISD::SMIN, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
5051 { .ISD: ISD::SMIN, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
5052 { .ISD: ISD::UMAX, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
5053 { .ISD: ISD::UMAX, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
5054 { .ISD: ISD::UMAX, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
5055 { .ISD: ISD::UMIN, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 3 } },
5056 { .ISD: ISD::UMIN, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
5057 { .ISD: ISD::UMIN, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 1, .LatencyCost: 4, .CodeSizeCost: 2, .SizeAndLatencyCost: 4 } },
5058 { .ISD: ISD::SADDO, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5059 { .ISD: ISD::SADDO, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5060 { .ISD: ISD::SADDO, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5061 { .ISD: ISD::UADDO, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5062 { .ISD: ISD::UADDO, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5063 { .ISD: ISD::UADDO, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5064 { .ISD: ISD::SMULO, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5065 { .ISD: ISD::SMULO, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5066 { .ISD: ISD::SMULO, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5067 { .ISD: ISD::UMULO, .Type: MVT::i32, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 8 } },
5068 { .ISD: ISD::UMULO, .Type: MVT::i16, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 9 } },
5069 { .ISD: ISD::UMULO, .Type: MVT::i8, .Cost: { .RecipThroughputCost: 6, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 6 } },
5070 };
5071
5072 Type *RetTy = ICA.getReturnType();
5073 Type *OpTy = RetTy;
5074 Intrinsic::ID IID = ICA.getID();
5075 unsigned ISD = ISD::DELETED_NODE;
5076 switch (IID) {
5077 default:
5078 break;
5079 case Intrinsic::abs:
5080 ISD = ISD::ABS;
5081 break;
5082 case Intrinsic::bitreverse:
5083 ISD = ISD::BITREVERSE;
5084 break;
5085 case Intrinsic::bswap:
5086 ISD = ISD::BSWAP;
5087 break;
5088 case Intrinsic::ctlz:
5089 ISD = ISD::CTLZ;
5090 break;
5091 case Intrinsic::ctpop:
5092 ISD = ISD::CTPOP;
5093 break;
5094 case Intrinsic::cttz:
5095 ISD = ISD::CTTZ;
5096 break;
5097 case Intrinsic::fshl:
5098 ISD = ISD::FSHL;
5099 if (!ICA.isTypeBasedOnly()) {
5100 const SmallVectorImpl<const Value *> &Args = ICA.getArgs();
5101 if (Args[0] == Args[1]) {
5102 ISD = ISD::ROTL;
5103 // Handle uniform constant rotation amounts.
5104 // TODO: Handle funnel-shift cases.
5105 const APInt *Amt;
5106 if (Args[2] &&
5107 PatternMatch::match(V: Args[2], P: PatternMatch::m_APIntAllowPoison(Res&: Amt)))
5108 ISD = X86ISD::VROTLI;
5109 }
5110 }
5111 break;
5112 case Intrinsic::fshr:
5113 // FSHR has same costs so don't duplicate.
5114 ISD = ISD::FSHL;
5115 if (!ICA.isTypeBasedOnly()) {
5116 const SmallVectorImpl<const Value *> &Args = ICA.getArgs();
5117 if (Args[0] == Args[1]) {
5118 ISD = ISD::ROTR;
5119 // Handle uniform constant rotation amount.
5120 // TODO: Handle funnel-shift cases.
5121 const APInt *Amt;
5122 if (Args[2] &&
5123 PatternMatch::match(V: Args[2], P: PatternMatch::m_APIntAllowPoison(Res&: Amt)))
5124 ISD = X86ISD::VROTLI;
5125 }
5126 }
5127 break;
5128 case Intrinsic::lrint:
5129 case Intrinsic::llrint: {
5130 // X86 can use the CVTP2SI instructions to lower lrint/llrint calls, which
5131 // have the same costs as the CVTTP2SI (fptosi) instructions
5132 const SmallVectorImpl<Type *> &ArgTys = ICA.getArgTypes();
5133 return getCastInstrCost(Opcode: Instruction::FPToSI, Dst: RetTy, Src: ArgTys[0],
5134 CCH: TTI::CastContextHint::None, CostKind);
5135 }
5136 case Intrinsic::maxnum:
5137 case Intrinsic::minnum:
5138 // FMINNUM has same costs so don't duplicate.
5139 ISD = ISD::FMAXNUM;
5140 break;
5141 case Intrinsic::sadd_sat:
5142 ISD = ISD::SADDSAT;
5143 break;
5144 case Intrinsic::smax:
5145 ISD = ISD::SMAX;
5146 break;
5147 case Intrinsic::smin:
5148 ISD = ISD::SMIN;
5149 break;
5150 case Intrinsic::smulh:
5151 ISD = ISD::MULHS;
5152 break;
5153 case Intrinsic::ssub_sat:
5154 ISD = ISD::SSUBSAT;
5155 break;
5156 case Intrinsic::uadd_sat:
5157 ISD = ISD::UADDSAT;
5158 break;
5159 case Intrinsic::umax:
5160 ISD = ISD::UMAX;
5161 break;
5162 case Intrinsic::umin:
5163 ISD = ISD::UMIN;
5164 break;
5165 case Intrinsic::usub_sat:
5166 ISD = ISD::USUBSAT;
5167 break;
5168 case Intrinsic::umulh:
5169 ISD = ISD::MULHU;
5170 break;
5171 case Intrinsic::sqrt:
5172 // The estimate sequence is costed with the division it is folded into.
5173 if ((CostKind == TTI::TCK_RecipThroughput ||
5174 CostKind == TTI::TCK_Latency) &&
5175 isFoldedIntoRsqrtEstimate(ICA, TLI: *TLI, DL))
5176 return TTI::TCC_Free;
5177 ISD = ISD::FSQRT;
5178 break;
5179 case Intrinsic::sadd_with_overflow:
5180 case Intrinsic::ssub_with_overflow:
5181 // SSUBO has same costs so don't duplicate.
5182 ISD = ISD::SADDO;
5183 OpTy = RetTy->getContainedType(i: 0);
5184 break;
5185 case Intrinsic::uadd_with_overflow:
5186 case Intrinsic::usub_with_overflow:
5187 // USUBO has same costs so don't duplicate.
5188 ISD = ISD::UADDO;
5189 OpTy = RetTy->getContainedType(i: 0);
5190 break;
5191 case Intrinsic::smul_with_overflow:
5192 ISD = ISD::SMULO;
5193 OpTy = RetTy->getContainedType(i: 0);
5194 break;
5195 case Intrinsic::umul_with_overflow:
5196 ISD = ISD::UMULO;
5197 OpTy = RetTy->getContainedType(i: 0);
5198 break;
5199 case Intrinsic::clmul:
5200 ISD = ISD::CLMUL;
5201 break;
5202 }
5203
5204 if (ISD != ISD::DELETED_NODE) {
5205 auto adjustTableCost = [&](int ISD, unsigned Cost,
5206 std::pair<InstructionCost, MVT> LT,
5207 FastMathFlags FMF) -> InstructionCost {
5208 InstructionCost LegalizationCost = LT.first;
5209 MVT MTy = LT.second;
5210
5211 // If there are no NANs to deal with, then these are reduced to a
5212 // single MIN** or MAX** instruction instead of the MIN/CMP/SELECT that we
5213 // assume is used in the non-fast case.
5214 if (ISD == ISD::FMAXNUM || ISD == ISD::FMINNUM) {
5215 if (FMF.noNaNs())
5216 return LegalizationCost * 1;
5217 }
5218
5219 // For cases where some ops can be folded into a load/store, assume free.
5220 if (MTy.isScalarInteger()) {
5221 if (ISD == ISD::BSWAP && ST->hasMOVBE() && ST->hasFastMOVBE()) {
5222 if (const Instruction *II = ICA.getInst()) {
5223 if (II->hasOneUse() && isa<StoreInst>(Val: II->user_back()))
5224 return TTI::TCC_Free;
5225 if (auto *LI = dyn_cast<LoadInst>(Val: II->getOperand(i: 0))) {
5226 if (LI->hasOneUse())
5227 return TTI::TCC_Free;
5228 }
5229 }
5230 }
5231 }
5232
5233 return LegalizationCost * (int)Cost;
5234 };
5235
5236 // Legalize the type.
5237 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: OpTy);
5238 MVT MTy = LT.second;
5239
5240 // Without BMI/LZCNT see if we're only looking for a *_ZERO_POISON cost.
5241 if (((ISD == ISD::CTTZ && !ST->hasBMI()) ||
5242 (ISD == ISD::CTLZ && !ST->hasLZCNT())) &&
5243 !MTy.isVector() && !ICA.isTypeBasedOnly()) {
5244 const SmallVectorImpl<const Value *> &Args = ICA.getArgs();
5245 if (auto *Cst = dyn_cast<ConstantInt>(Val: Args[1]))
5246 if (Cst->isAllOnesValue())
5247 ISD =
5248 ISD == ISD::CTTZ ? ISD::CTTZ_ZERO_POISON : ISD::CTLZ_ZERO_POISON;
5249 }
5250
5251 // FSQRT is a single instruction.
5252 if (ISD == ISD::FSQRT && CostKind == TTI::TCK_CodeSize)
5253 return LT.first;
5254
5255 if (ST->useGLMDivSqrtCosts())
5256 if (const auto *Entry = CostTableLookup(Table: GLMCostTbl, ISD, Ty: MTy))
5257 if (auto KindCost = Entry->Cost[CostKind])
5258 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5259
5260 if (ST->useSLMArithCosts())
5261 if (const auto *Entry = CostTableLookup(Table: SLMCostTbl, ISD, Ty: MTy))
5262 if (auto KindCost = Entry->Cost[CostKind])
5263 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5264
5265 if (ST->hasVBMI2())
5266 if (const auto *Entry = CostTableLookup(Table: AVX512VBMI2CostTbl, ISD, Ty: MTy))
5267 if (auto KindCost = Entry->Cost[CostKind])
5268 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5269
5270 if (ST->hasBITALG())
5271 if (const auto *Entry = CostTableLookup(Table: AVX512BITALGCostTbl, ISD, Ty: MTy))
5272 if (auto KindCost = Entry->Cost[CostKind])
5273 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5274
5275 if (ST->hasVPOPCNTDQ())
5276 if (const auto *Entry = CostTableLookup(Table: AVX512VPOPCNTDQCostTbl, ISD, Ty: MTy))
5277 if (auto KindCost = Entry->Cost[CostKind])
5278 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5279
5280 if (ST->hasGFNI())
5281 if (const auto *Entry = CostTableLookup(Table: GFNICostTbl, ISD, Ty: MTy))
5282 if (auto KindCost = Entry->Cost[CostKind])
5283 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5284
5285 if (ST->hasCDI())
5286 if (const auto *Entry = CostTableLookup(Table: AVX512CDCostTbl, ISD, Ty: MTy))
5287 if (auto KindCost = Entry->Cost[CostKind])
5288 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5289
5290 if (ST->hasBWI())
5291 if (const auto *Entry = CostTableLookup(Table: AVX512BWCostTbl, ISD, Ty: MTy))
5292 if (auto KindCost = Entry->Cost[CostKind])
5293 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5294
5295 if (ST->hasAVX512())
5296 if (const auto *Entry = CostTableLookup(Table: AVX512CostTbl, ISD, Ty: MTy))
5297 if (auto KindCost = Entry->Cost[CostKind])
5298 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5299
5300 if (ST->hasXOP())
5301 if (const auto *Entry = CostTableLookup(Table: XOPCostTbl, ISD, Ty: MTy))
5302 if (auto KindCost = Entry->Cost[CostKind])
5303 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5304
5305 if (ST->hasAVX2())
5306 if (const auto *Entry = CostTableLookup(Table: AVX2CostTbl, ISD, Ty: MTy))
5307 if (auto KindCost = Entry->Cost[CostKind])
5308 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5309
5310 if (ST->hasAVX())
5311 if (const auto *Entry = CostTableLookup(Table: AVX1CostTbl, ISD, Ty: MTy))
5312 if (auto KindCost = Entry->Cost[CostKind])
5313 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5314
5315 if (ST->hasSSE42())
5316 if (const auto *Entry = CostTableLookup(Table: SSE42CostTbl, ISD, Ty: MTy))
5317 if (auto KindCost = Entry->Cost[CostKind])
5318 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5319
5320 if (ST->hasSSE41())
5321 if (const auto *Entry = CostTableLookup(Table: SSE41CostTbl, ISD, Ty: MTy))
5322 if (auto KindCost = Entry->Cost[CostKind])
5323 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5324
5325 if (ST->hasSSSE3())
5326 if (const auto *Entry = CostTableLookup(Table: SSSE3CostTbl, ISD, Ty: MTy))
5327 if (auto KindCost = Entry->Cost[CostKind])
5328 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5329
5330 if (ST->hasSSE2())
5331 if (const auto *Entry = CostTableLookup(Table: SSE2CostTbl, ISD, Ty: MTy))
5332 if (auto KindCost = Entry->Cost[CostKind])
5333 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5334
5335 if (ST->hasSSE1())
5336 if (const auto *Entry = CostTableLookup(Table: SSE1CostTbl, ISD, Ty: MTy))
5337 if (auto KindCost = Entry->Cost[CostKind])
5338 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5339
5340 if (ST->hasBMI()) {
5341 if (ST->is64Bit())
5342 if (const auto *Entry = CostTableLookup(Table: BMI64CostTbl, ISD, Ty: MTy))
5343 if (auto KindCost = Entry->Cost[CostKind])
5344 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5345
5346 if (const auto *Entry = CostTableLookup(Table: BMI32CostTbl, ISD, Ty: MTy))
5347 if (auto KindCost = Entry->Cost[CostKind])
5348 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5349 }
5350
5351 if (ST->hasLZCNT()) {
5352 if (ST->is64Bit())
5353 if (const auto *Entry = CostTableLookup(Table: LZCNT64CostTbl, ISD, Ty: MTy))
5354 if (auto KindCost = Entry->Cost[CostKind])
5355 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5356
5357 if (const auto *Entry = CostTableLookup(Table: LZCNT32CostTbl, ISD, Ty: MTy))
5358 if (auto KindCost = Entry->Cost[CostKind])
5359 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5360 }
5361
5362 if (ST->hasPOPCNT()) {
5363 if (ST->is64Bit())
5364 if (const auto *Entry = CostTableLookup(Table: POPCNT64CostTbl, ISD, Ty: MTy))
5365 if (auto KindCost = Entry->Cost[CostKind])
5366 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5367
5368 if (const auto *Entry = CostTableLookup(Table: POPCNT32CostTbl, ISD, Ty: MTy))
5369 if (auto KindCost = Entry->Cost[CostKind])
5370 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5371 }
5372
5373 // FIXME: PCLMUL w/ AVX/AVX512 and VPCLMULQDQ are not handled properly.
5374 if (ST->hasPCLMUL())
5375 if (const auto *Entry = CostTableLookup(Table: PCLMULCostTbl, ISD, Ty: MTy))
5376 if (auto KindCost = Entry->Cost[CostKind])
5377 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5378
5379 if (ST->is64Bit())
5380 if (const auto *Entry = CostTableLookup(Table: X64CostTbl, ISD, Ty: MTy))
5381 if (auto KindCost = Entry->Cost[CostKind])
5382 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5383
5384 if (const auto *Entry = CostTableLookup(Table: X86CostTbl, ISD, Ty: MTy))
5385 if (auto KindCost = Entry->Cost[CostKind])
5386 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5387
5388 // Without arg data, we need to compute the expanded costs of custom lowered
5389 // intrinsics to prevent use of the (very low) default costs.
5390 if (ICA.isTypeBasedOnly() &&
5391 (IID == Intrinsic::fshl || IID == Intrinsic::fshr)) {
5392 Type *CondTy = RetTy->getWithNewBitWidth(NewBitWidth: 1);
5393 InstructionCost Cost = 0;
5394 Cost += getArithmeticInstrCost(Opcode: BinaryOperator::Or, Ty: RetTy, CostKind);
5395 Cost += getArithmeticInstrCost(Opcode: BinaryOperator::Sub, Ty: RetTy, CostKind);
5396 Cost += getArithmeticInstrCost(Opcode: BinaryOperator::Shl, Ty: RetTy, CostKind);
5397 Cost += getArithmeticInstrCost(Opcode: BinaryOperator::LShr, Ty: RetTy, CostKind);
5398 Cost += getArithmeticInstrCost(Opcode: BinaryOperator::And, Ty: RetTy, CostKind);
5399 Cost += getCmpSelInstrCost(Opcode: BinaryOperator::ICmp, ValTy: RetTy, CondTy,
5400 VecPred: CmpInst::ICMP_EQ, CostKind);
5401 Cost += getCmpSelInstrCost(Opcode: BinaryOperator::Select, ValTy: RetTy, CondTy,
5402 VecPred: CmpInst::ICMP_EQ, CostKind);
5403 return Cost;
5404 }
5405 }
5406
5407 return BaseT::getIntrinsicInstrCost(ICA, CostKind);
5408}
5409
5410InstructionCost X86TTIImpl::getVectorInstrCost(
5411 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
5412 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
5413 static const CostTblEntry SLMCostTbl[] = {
5414 { .ISD: ISD::EXTRACT_VECTOR_ELT, .Type: MVT::i8, .Cost: 4 },
5415 { .ISD: ISD::EXTRACT_VECTOR_ELT, .Type: MVT::i16, .Cost: 4 },
5416 { .ISD: ISD::EXTRACT_VECTOR_ELT, .Type: MVT::i32, .Cost: 4 },
5417 { .ISD: ISD::EXTRACT_VECTOR_ELT, .Type: MVT::i64, .Cost: 7 }
5418 };
5419
5420 assert(Val->isVectorTy() && "This must be a vector type");
5421 auto *VT = cast<VectorType>(Val);
5422 if (VT->isScalableTy())
5423 return InstructionCost::getInvalid();
5424
5425 Type *ScalarType = Val->getScalarType();
5426 InstructionCost RegisterFileMoveCost = 0;
5427
5428 // Non-immediate extraction/insertion can be handled as a sequence of
5429 // aliased loads+stores via the stack.
5430 if (Index == -1U && (Opcode == Instruction::ExtractElement ||
5431 Opcode == Instruction::InsertElement)) {
5432 // TODO: On some SSE41+ targets, we expand to cmp+splat+select patterns:
5433 // inselt N0, N1, N2 --> select (SplatN2 == {0,1,2...}) ? SplatN1 : N0.
5434
5435 // TODO: Move this to BasicTTIImpl.h? We'd need better gep + index handling.
5436 assert(isa<FixedVectorType>(Val) && "Fixed vector type expected");
5437 Align VecAlign = DL.getPrefTypeAlign(Ty: Val);
5438 Align SclAlign = DL.getPrefTypeAlign(Ty: ScalarType);
5439
5440 // Extract - store vector to stack, load scalar.
5441 if (Opcode == Instruction::ExtractElement) {
5442 return getMemoryOpCost(Opcode: Instruction::Store, Src: Val, Alignment: VecAlign, AddressSpace: 0, CostKind) +
5443 getMemoryOpCost(Opcode: Instruction::Load, Src: ScalarType, Alignment: SclAlign, AddressSpace: 0,
5444 CostKind);
5445 }
5446 // Insert - store vector to stack, store scalar, load vector.
5447 if (Opcode == Instruction::InsertElement) {
5448 return getMemoryOpCost(Opcode: Instruction::Store, Src: Val, Alignment: VecAlign, AddressSpace: 0, CostKind) +
5449 getMemoryOpCost(Opcode: Instruction::Store, Src: ScalarType, Alignment: SclAlign, AddressSpace: 0,
5450 CostKind) +
5451 getMemoryOpCost(Opcode: Instruction::Load, Src: Val, Alignment: VecAlign, AddressSpace: 0, CostKind);
5452 }
5453 }
5454
5455 if (Index != -1U && (Opcode == Instruction::ExtractElement ||
5456 Opcode == Instruction::InsertElement)) {
5457 // Extraction of vXi1 elements are now efficiently handled by MOVMSK.
5458 if (Opcode == Instruction::ExtractElement &&
5459 ScalarType->getScalarSizeInBits() == 1 &&
5460 cast<FixedVectorType>(Val)->getNumElements() > 1)
5461 return 1;
5462
5463 // Legalize the type.
5464 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: Val);
5465
5466 // This type is legalized to a scalar type.
5467 if (!LT.second.isVector())
5468 return TTI::TCC_Free;
5469
5470 // The type may be split. Normalize the index to the new type.
5471 unsigned SizeInBits = LT.second.getSizeInBits();
5472 unsigned NumElts = LT.second.getVectorNumElements();
5473 unsigned SubNumElts = NumElts;
5474 Index = Index % NumElts;
5475
5476 // For >128-bit vectors, we need to extract higher 128-bit subvectors.
5477 // For inserts, we also need to insert the subvector back.
5478 if (SizeInBits > 128) {
5479 assert((SizeInBits % 128) == 0 && "Illegal vector");
5480 unsigned NumSubVecs = SizeInBits / 128;
5481 SubNumElts = NumElts / NumSubVecs;
5482 if (SubNumElts <= Index) {
5483 RegisterFileMoveCost += (Opcode == Instruction::InsertElement ? 2 : 1);
5484 Index %= SubNumElts;
5485 }
5486 }
5487
5488 MVT MScalarTy = LT.second.getScalarType();
5489 auto IsCheapPInsrPExtrInsertPS = [&]() {
5490 // Assume pinsr/pextr XMM <-> GPR is relatively cheap on all targets.
5491 // Inserting f32 into index0 is just movss.
5492 // Also, assume insertps is relatively cheap on all >= SSE41 targets.
5493 return (MScalarTy == MVT::i16 && ST->hasSSE2()) ||
5494 (MScalarTy.isInteger() && ST->hasSSE41()) ||
5495 (MScalarTy == MVT::f32 && ST->hasSSE1() && Index == 0 &&
5496 Opcode == Instruction::InsertElement) ||
5497 (MScalarTy == MVT::f32 && ST->hasSSE41() &&
5498 Opcode == Instruction::InsertElement);
5499 };
5500
5501 if (Index == 0) {
5502 // Floating point scalars are already located in index #0.
5503 // Many insertions to #0 can fold away for scalar fp-ops, so let's assume
5504 // true for all.
5505 if (ScalarType->isFloatingPointTy() &&
5506 (Opcode != Instruction::InsertElement || !Op0 ||
5507 isa<UndefValue>(Val: Op0)))
5508 return RegisterFileMoveCost;
5509
5510 if (Opcode == Instruction::InsertElement &&
5511 isa_and_nonnull<UndefValue>(Val: Op0)) {
5512 // Consider the gather cost to be cheap.
5513 if (isa_and_nonnull<LoadInst>(Val: Op1))
5514 return RegisterFileMoveCost;
5515 if (!IsCheapPInsrPExtrInsertPS()) {
5516 // mov constant-to-GPR + movd/movq GPR -> XMM.
5517 if (isa_and_nonnull<Constant>(Val: Op1) && Op1->getType()->isIntegerTy())
5518 return 2 + RegisterFileMoveCost;
5519 // Assume movd/movq GPR -> XMM is relatively cheap on all targets.
5520 return 1 + RegisterFileMoveCost;
5521 }
5522 }
5523
5524 // Assume movd/movq XMM -> GPR is relatively cheap on all targets.
5525 if (ScalarType->isIntegerTy() && Opcode == Instruction::ExtractElement)
5526 return 1 + RegisterFileMoveCost;
5527 }
5528
5529 int ISD = TLI->InstructionOpcodeToISD(Opcode);
5530 assert(ISD && "Unexpected vector opcode");
5531 if (ST->useSLMArithCosts())
5532 if (auto *Entry = CostTableLookup(Table: SLMCostTbl, ISD, Ty: MScalarTy))
5533 return Entry->Cost + RegisterFileMoveCost;
5534
5535 // Consider cheap cases.
5536 if (IsCheapPInsrPExtrInsertPS())
5537 return 1 + RegisterFileMoveCost;
5538
5539 // For extractions we just need to shuffle the element to index 0, which
5540 // should be very cheap (assume cost = 1). For insertions we need to shuffle
5541 // the elements to its destination. In both cases we must handle the
5542 // subvector move(s).
5543 // If the vector type is already less than 128-bits then don't reduce it.
5544 // TODO: Under what circumstances should we shuffle using the full width?
5545 InstructionCost ShuffleCost = 1;
5546 if (Opcode == Instruction::InsertElement) {
5547 auto *SubTy = cast<VectorType>(Val);
5548 EVT VT = TLI->getValueType(DL, Ty: Val);
5549 if (VT.getScalarType() != MScalarTy || VT.getSizeInBits() >= 128)
5550 SubTy = FixedVectorType::get(ElementType: ScalarType, NumElts: SubNumElts);
5551 ShuffleCost = getShuffleCost(Kind: TTI::SK_PermuteTwoSrc, DstTy: SubTy, SrcTy: SubTy,
5552 CostKind, Mask: {}, Index: 0, SubTp: SubTy);
5553 }
5554 int IntOrFpCost = ScalarType->isFloatingPointTy() ? 0 : 1;
5555 return ShuffleCost + IntOrFpCost + RegisterFileMoveCost;
5556 }
5557
5558 return BaseT::getVectorInstrCost(Opcode, Val, CostKind, Index, Op0, Op1,
5559 VIC) +
5560 RegisterFileMoveCost;
5561}
5562
5563InstructionCost X86TTIImpl::getScalarizationOverhead(
5564 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
5565 TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef<Value *> VL,
5566 TTI::VectorInstrContext VIC) const {
5567 assert(DemandedElts.getBitWidth() ==
5568 cast<FixedVectorType>(Ty)->getNumElements() &&
5569 "Vector size mismatch");
5570
5571 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
5572 MVT MScalarTy = LT.second.getScalarType();
5573 unsigned LegalVectorBitWidth = LT.second.getSizeInBits();
5574 InstructionCost Cost = 0;
5575
5576 constexpr unsigned LaneBitWidth = 128;
5577 assert((LegalVectorBitWidth < LaneBitWidth ||
5578 (LegalVectorBitWidth % LaneBitWidth) == 0) &&
5579 "Illegal vector");
5580
5581 const int NumLegalVectors = LT.first.getValue();
5582 assert(NumLegalVectors >= 0 && "Negative cost!");
5583
5584 // For insertions, a ISD::BUILD_VECTOR style vector initialization can be much
5585 // cheaper than an accumulation of ISD::INSERT_VECTOR_ELT. SLPVectorizer has
5586 // a special heuristic regarding poison input which is passed here in
5587 // ForPoisonSrc.
5588 if (Insert && !ForPoisonSrc) {
5589 // This is nearly identical to BaseT::getScalarizationOverhead(), except
5590 // it is passing nullptr to getVectorInstrCost() for Op0 (instead of
5591 // Constant::getNullValue()), which makes the X86TTIImpl
5592 // getVectorInstrCost() return 0 instead of 1.
5593 for (unsigned I : seq(Size: DemandedElts.getBitWidth())) {
5594 if (!DemandedElts[I])
5595 continue;
5596 Cost += getVectorInstrCost(Opcode: Instruction::InsertElement, Val: Ty, CostKind, Index: I,
5597 Op0: Constant::getNullValue(Ty),
5598 Op1: VL.empty() ? nullptr : VL[I],
5599 VIC: TTI::VectorInstrContext::None);
5600 }
5601 return Cost;
5602 }
5603
5604 if (Insert) {
5605 if ((MScalarTy == MVT::i16 && ST->hasSSE2()) ||
5606 (MScalarTy.isInteger() && ST->hasSSE41()) ||
5607 (MScalarTy == MVT::f32 && ST->hasSSE41())) {
5608 // For types we can insert directly, insertion into 128-bit sub vectors is
5609 // cheap, followed by a cheap chain of concatenations.
5610 if (LegalVectorBitWidth <= LaneBitWidth) {
5611 Cost += BaseT::getScalarizationOverhead(InTy: Ty, DemandedElts, Insert,
5612 /*Extract*/ false, CostKind);
5613 } else {
5614 // In each 128-lane, if at least one index is demanded but not all
5615 // indices are demanded and this 128-lane is not the first 128-lane of
5616 // the legalized-vector, then this 128-lane needs a extracti128; If in
5617 // each 128-lane, there is at least one demanded index, this 128-lane
5618 // needs a inserti128.
5619
5620 // The following cases will help you build a better understanding:
5621 // Assume we insert several elements into a v8i32 vector in avx2,
5622 // Case#1: inserting into 1th index needs vpinsrd + inserti128.
5623 // Case#2: inserting into 5th index needs extracti128 + vpinsrd +
5624 // inserti128.
5625 // Case#3: inserting into 4,5,6,7 index needs 4*vpinsrd + inserti128.
5626 assert((LegalVectorBitWidth % LaneBitWidth) == 0 && "Illegal vector");
5627 unsigned NumLegalLanes = LegalVectorBitWidth / LaneBitWidth;
5628 unsigned NumLanesTotal = NumLegalLanes * NumLegalVectors;
5629 unsigned NumLegalElts =
5630 LT.second.getVectorNumElements() * NumLegalVectors;
5631 assert(NumLegalElts >= DemandedElts.getBitWidth() &&
5632 "Vector has been legalized to smaller element count");
5633 assert((NumLegalElts % NumLanesTotal) == 0 &&
5634 "Unexpected elts per lane");
5635 unsigned NumEltsPerLane = NumLegalElts / NumLanesTotal;
5636
5637 APInt WidenedDemandedElts = DemandedElts.zext(width: NumLegalElts);
5638 auto *LaneTy =
5639 FixedVectorType::get(ElementType: Ty->getElementType(), NumElts: NumEltsPerLane);
5640
5641 for (unsigned I = 0; I != NumLanesTotal; ++I) {
5642 APInt LaneEltMask = WidenedDemandedElts.extractBits(
5643 numBits: NumEltsPerLane, bitPosition: NumEltsPerLane * I);
5644 if (LaneEltMask.isZero())
5645 continue;
5646 // FIXME: we don't need to extract if all non-demanded elements
5647 // are legalization-inserted padding.
5648 if (!LaneEltMask.isAllOnes())
5649 Cost += getShuffleCost(Kind: TTI::SK_ExtractSubvector, DstTy: Ty, SrcTy: Ty, CostKind,
5650 Mask: {}, Index: I * NumEltsPerLane, SubTp: LaneTy);
5651 Cost += BaseT::getScalarizationOverhead(InTy: LaneTy, DemandedElts: LaneEltMask, Insert,
5652 /*Extract*/ false, CostKind);
5653 }
5654
5655 APInt AffectedLanes =
5656 APIntOps::ScaleBitMask(A: WidenedDemandedElts, NewBitWidth: NumLanesTotal);
5657 APInt FullyAffectedLegalVectors = APIntOps::ScaleBitMask(
5658 A: AffectedLanes, NewBitWidth: NumLegalVectors, /*MatchAllBits=*/true);
5659 for (int LegalVec = 0; LegalVec != NumLegalVectors; ++LegalVec) {
5660 for (unsigned Lane = 0; Lane != NumLegalLanes; ++Lane) {
5661 unsigned I = NumLegalLanes * LegalVec + Lane;
5662 // No need to insert unaffected lane; or lane 0 of each legal vector
5663 // iff ALL lanes of that vector were affected and will be inserted.
5664 if (!AffectedLanes[I] ||
5665 (Lane == 0 && FullyAffectedLegalVectors[LegalVec]))
5666 continue;
5667 Cost += getShuffleCost(Kind: TTI::SK_InsertSubvector, DstTy: Ty, SrcTy: Ty, CostKind,
5668 Mask: {}, Index: I * NumEltsPerLane, SubTp: LaneTy);
5669 }
5670 }
5671 }
5672 } else if (LT.second.isVector()) {
5673 // Without fast insertion, we need to use MOVD/MOVQ to pass each demanded
5674 // integer element as a SCALAR_TO_VECTOR, then we build the vector as a
5675 // series of UNPCK followed by CONCAT_VECTORS - all of these can be
5676 // considered cheap.
5677 if (Ty->isIntOrIntVectorTy())
5678 Cost += DemandedElts.popcount();
5679
5680 // Get the smaller of the legalized or original pow2-extended number of
5681 // vector elements, which represents the number of unpacks we'll end up
5682 // performing.
5683 unsigned NumElts = LT.second.getVectorNumElements();
5684 unsigned Pow2Elts =
5685 PowerOf2Ceil(A: cast<FixedVectorType>(Val: Ty)->getNumElements());
5686 Cost += (std::min<unsigned>(a: NumElts, b: Pow2Elts) - 1) * LT.first;
5687 }
5688 }
5689
5690 if (Extract) {
5691 // vXi1 can be efficiently extracted with MOVMSK.
5692 // TODO: AVX512 predicate mask handling.
5693 // NOTE: This doesn't work well for roundtrip scalarization.
5694 if (!Insert && Ty->getScalarSizeInBits() == 1 && !ST->hasAVX512()) {
5695 unsigned NumElts = cast<FixedVectorType>(Val: Ty)->getNumElements();
5696 unsigned MaxElts = ST->hasAVX2() ? 32 : 16;
5697 unsigned MOVMSKCost = (NumElts + MaxElts - 1) / MaxElts;
5698 return MOVMSKCost;
5699 }
5700
5701 if (LT.second.isVector()) {
5702 unsigned NumLegalElts =
5703 LT.second.getVectorNumElements() * NumLegalVectors;
5704 assert(NumLegalElts >= DemandedElts.getBitWidth() &&
5705 "Vector has been legalized to smaller element count");
5706
5707 // If we're extracting elements from a 128-bit subvector lane,
5708 // we only need to extract each lane once, not for every element.
5709 if (LegalVectorBitWidth > LaneBitWidth) {
5710 unsigned NumLegalLanes = LegalVectorBitWidth / LaneBitWidth;
5711 unsigned NumLanesTotal = NumLegalLanes * NumLegalVectors;
5712 assert((NumLegalElts % NumLanesTotal) == 0 &&
5713 "Unexpected elts per lane");
5714 unsigned NumEltsPerLane = NumLegalElts / NumLanesTotal;
5715
5716 // Add cost for each demanded 128-bit subvector extraction.
5717 // Luckily this is a lot easier than for insertion.
5718 APInt WidenedDemandedElts = DemandedElts.zext(width: NumLegalElts);
5719 auto *LaneTy =
5720 FixedVectorType::get(ElementType: Ty->getElementType(), NumElts: NumEltsPerLane);
5721
5722 for (unsigned I = 0; I != NumLanesTotal; ++I) {
5723 APInt LaneEltMask = WidenedDemandedElts.extractBits(
5724 numBits: NumEltsPerLane, bitPosition: I * NumEltsPerLane);
5725 if (LaneEltMask.isZero())
5726 continue;
5727 Cost += getShuffleCost(Kind: TTI::SK_ExtractSubvector, DstTy: Ty, SrcTy: Ty, CostKind, Mask: {},
5728 Index: I * NumEltsPerLane, SubTp: LaneTy);
5729 Cost += BaseT::getScalarizationOverhead(
5730 InTy: LaneTy, DemandedElts: LaneEltMask, /*Insert*/ false, Extract, CostKind);
5731 }
5732
5733 return Cost;
5734 }
5735 }
5736
5737 // Fallback to default extraction.
5738 Cost += BaseT::getScalarizationOverhead(InTy: Ty, DemandedElts, /*Insert*/ false,
5739 Extract, CostKind);
5740 }
5741
5742 return Cost;
5743}
5744
5745InstructionCost
5746X86TTIImpl::getReplicationShuffleCost(Type *EltTy, int ReplicationFactor,
5747 int VF, const APInt &DemandedDstElts,
5748 TTI::TargetCostKind CostKind) const {
5749 const unsigned EltTyBits = DL.getTypeSizeInBits(Ty: EltTy);
5750 // We don't differentiate element types here, only element bit width.
5751 EltTy = IntegerType::getIntNTy(C&: EltTy->getContext(), N: EltTyBits);
5752
5753 auto bailout = [&]() {
5754 return BaseT::getReplicationShuffleCost(EltTy, ReplicationFactor, VF,
5755 DemandedDstElts, CostKind);
5756 };
5757
5758 // For now, only deal with AVX512 cases.
5759 if (!ST->hasAVX512())
5760 return bailout();
5761
5762 // Do we have a native shuffle for this element type, or should we promote?
5763 unsigned PromEltTyBits = EltTyBits;
5764 switch (EltTyBits) {
5765 case 32:
5766 case 64:
5767 break; // AVX512F.
5768 case 16:
5769 if (!ST->hasBWI())
5770 PromEltTyBits = 32; // promote to i32, AVX512F.
5771 break; // AVX512BW
5772 case 8:
5773 if (!ST->hasVBMI())
5774 PromEltTyBits = 32; // promote to i32, AVX512F.
5775 break; // AVX512VBMI
5776 case 1:
5777 // There is no support for shuffling i1 elements. We *must* promote.
5778 if (ST->hasBWI()) {
5779 if (ST->hasVBMI())
5780 PromEltTyBits = 8; // promote to i8, AVX512VBMI.
5781 else
5782 PromEltTyBits = 16; // promote to i16, AVX512BW.
5783 break;
5784 }
5785 PromEltTyBits = 32; // promote to i32, AVX512F.
5786 break;
5787 default:
5788 return bailout();
5789 }
5790 auto *PromEltTy = IntegerType::getIntNTy(C&: EltTy->getContext(), N: PromEltTyBits);
5791
5792 auto *SrcVecTy = FixedVectorType::get(ElementType: EltTy, NumElts: VF);
5793 auto *PromSrcVecTy = FixedVectorType::get(ElementType: PromEltTy, NumElts: VF);
5794
5795 int NumDstElements = VF * ReplicationFactor;
5796 auto *PromDstVecTy = FixedVectorType::get(ElementType: PromEltTy, NumElts: NumDstElements);
5797 auto *DstVecTy = FixedVectorType::get(ElementType: EltTy, NumElts: NumDstElements);
5798
5799 // Legalize the types.
5800 MVT LegalSrcVecTy = getTypeLegalizationCost(Ty: SrcVecTy).second;
5801 MVT LegalPromSrcVecTy = getTypeLegalizationCost(Ty: PromSrcVecTy).second;
5802 MVT LegalPromDstVecTy = getTypeLegalizationCost(Ty: PromDstVecTy).second;
5803 MVT LegalDstVecTy = getTypeLegalizationCost(Ty: DstVecTy).second;
5804 // They should have legalized into vector types.
5805 if (!LegalSrcVecTy.isVector() || !LegalPromSrcVecTy.isVector() ||
5806 !LegalPromDstVecTy.isVector() || !LegalDstVecTy.isVector())
5807 return bailout();
5808
5809 if (PromEltTyBits != EltTyBits) {
5810 // If we have to perform the shuffle with wider elt type than our data type,
5811 // then we will first need to anyext (we don't care about the new bits)
5812 // the source elements, and then truncate Dst elements.
5813 InstructionCost PromotionCost;
5814 PromotionCost += getCastInstrCost(
5815 Opcode: Instruction::SExt, /*Dst=*/PromSrcVecTy, /*Src=*/SrcVecTy,
5816 CCH: TargetTransformInfo::CastContextHint::None, CostKind);
5817 PromotionCost +=
5818 getCastInstrCost(Opcode: Instruction::Trunc, /*Dst=*/DstVecTy,
5819 /*Src=*/PromDstVecTy,
5820 CCH: TargetTransformInfo::CastContextHint::None, CostKind);
5821 return PromotionCost + getReplicationShuffleCost(EltTy: PromEltTy,
5822 ReplicationFactor, VF,
5823 DemandedDstElts, CostKind);
5824 }
5825
5826 assert(LegalSrcVecTy.getScalarSizeInBits() == EltTyBits &&
5827 LegalSrcVecTy.getScalarType() == LegalDstVecTy.getScalarType() &&
5828 "We expect that the legalization doesn't affect the element width, "
5829 "doesn't coalesce/split elements.");
5830
5831 unsigned NumEltsPerDstVec = LegalDstVecTy.getVectorNumElements();
5832 unsigned NumDstVectors =
5833 divideCeil(Numerator: DstVecTy->getNumElements(), Denominator: NumEltsPerDstVec);
5834
5835 auto *SingleDstVecTy = FixedVectorType::get(ElementType: EltTy, NumElts: NumEltsPerDstVec);
5836
5837 // Not all the produced Dst elements may be demanded. In our case,
5838 // given that a single Dst vector is formed by a single shuffle,
5839 // if all elements that will form a single Dst vector aren't demanded,
5840 // then we won't need to do that shuffle, so adjust the cost accordingly.
5841 APInt DemandedDstVectors = APIntOps::ScaleBitMask(
5842 A: DemandedDstElts.zext(width: NumDstVectors * NumEltsPerDstVec), NewBitWidth: NumDstVectors);
5843 unsigned NumDstVectorsDemanded = DemandedDstVectors.popcount();
5844
5845 InstructionCost SingleShuffleCost =
5846 getShuffleCost(Kind: TTI::SK_PermuteSingleSrc, DstTy: SingleDstVecTy, SrcTy: SingleDstVecTy,
5847 CostKind, /*Mask=*/{},
5848 /*Index=*/0, /*SubTp=*/nullptr);
5849 return NumDstVectorsDemanded * SingleShuffleCost;
5850}
5851
5852InstructionCost X86TTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src,
5853 Align Alignment,
5854 unsigned AddressSpace,
5855 TTI::TargetCostKind CostKind,
5856 TTI::OperandValueInfo OpInfo,
5857 const Instruction *I) const {
5858 // FIXME: Load latency isn't handled here
5859 if (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency)
5860 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
5861 CostKind, OpInfo, I);
5862
5863 // TODO: Handle other cost kinds.
5864 if (CostKind != TTI::TCK_RecipThroughput) {
5865 if (auto *SI = dyn_cast_or_null<StoreInst>(Val: I)) {
5866 // Store instruction with index and scale costs 2 Uops.
5867 // Check the preceding GEP to identify non-const indices.
5868 if (auto *GEP = dyn_cast<GetElementPtrInst>(Val: SI->getPointerOperand())) {
5869 if (!all_of(Range: GEP->indices(), P: [](Value *V) { return isa<Constant>(Val: V); }))
5870 return TTI::TCC_Basic * 2;
5871 }
5872 }
5873 return TTI::TCC_Basic;
5874 }
5875
5876 assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
5877 "Invalid Opcode");
5878 // Type legalization can't handle structs
5879 if (TLI->getValueType(DL, Ty: Src, AllowUnknown: true) == MVT::Other)
5880 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
5881 CostKind, OpInfo, I);
5882
5883 // Legalize the type.
5884 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: Src);
5885
5886 auto *VTy = dyn_cast<FixedVectorType>(Val: Src);
5887
5888 InstructionCost Cost = 0;
5889
5890 // Add a cost for constant load to vector.
5891 if (Opcode == Instruction::Store && OpInfo.isConstant())
5892 Cost += getMemoryOpCost(Opcode: Instruction::Load, Src, Alignment: DL.getABITypeAlign(Ty: Src),
5893 /*AddressSpace=*/0, CostKind, OpInfo);
5894
5895 // Handle the simple case of non-vectors.
5896 // NOTE: this assumes that legalization never creates vector from scalars!
5897 if (!VTy || !LT.second.isVector()) {
5898 // Each load/store unit costs 1.
5899 return (LT.second.isFloatingPoint() ? Cost : 0) + LT.first * 1;
5900 }
5901
5902 bool IsLoad = Opcode == Instruction::Load;
5903
5904 Type *EltTy = VTy->getElementType();
5905
5906 const int EltTyBits = DL.getTypeSizeInBits(Ty: EltTy);
5907
5908 // Source of truth: how many elements were there in the original IR vector?
5909 const unsigned SrcNumElt = VTy->getNumElements();
5910
5911 // How far have we gotten?
5912 int NumEltRemaining = SrcNumElt;
5913 // Note that we intentionally capture by-reference, NumEltRemaining changes.
5914 auto NumEltDone = [&]() { return SrcNumElt - NumEltRemaining; };
5915
5916 const int MaxLegalOpSizeBytes = divideCeil(Numerator: LT.second.getSizeInBits(), Denominator: 8);
5917
5918 // Note that even if we can store 64 bits of an XMM, we still operate on XMM.
5919 const unsigned XMMBits = 128;
5920 if (XMMBits % EltTyBits != 0)
5921 // Vector size must be a multiple of the element size. I.e. no padding.
5922 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
5923 CostKind, OpInfo, I);
5924 const int NumEltPerXMM = XMMBits / EltTyBits;
5925
5926 auto *XMMVecTy = FixedVectorType::get(ElementType: EltTy, NumElts: NumEltPerXMM);
5927
5928 for (int CurrOpSizeBytes = MaxLegalOpSizeBytes, SubVecEltsLeft = 0;
5929 NumEltRemaining > 0; CurrOpSizeBytes /= 2) {
5930 // How many elements would a single op deal with at once?
5931 if ((8 * CurrOpSizeBytes) % EltTyBits != 0)
5932 // Vector size must be a multiple of the element size. I.e. no padding.
5933 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
5934 CostKind, OpInfo, I);
5935 int CurrNumEltPerOp = (8 * CurrOpSizeBytes) / EltTyBits;
5936
5937 assert(CurrOpSizeBytes > 0 && CurrNumEltPerOp > 0 && "How'd we get here?");
5938 assert((((NumEltRemaining * EltTyBits) < (2 * 8 * CurrOpSizeBytes)) ||
5939 (CurrOpSizeBytes == MaxLegalOpSizeBytes)) &&
5940 "Unless we haven't halved the op size yet, "
5941 "we have less than two op's sized units of work left.");
5942
5943 auto *CurrVecTy = CurrNumEltPerOp > NumEltPerXMM
5944 ? FixedVectorType::get(ElementType: EltTy, NumElts: CurrNumEltPerOp)
5945 : XMMVecTy;
5946
5947 assert(CurrVecTy->getNumElements() % CurrNumEltPerOp == 0 &&
5948 "After halving sizes, the vector elt count is no longer a multiple "
5949 "of number of elements per operation?");
5950 auto *CoalescedVecTy =
5951 CurrNumEltPerOp == 1
5952 ? CurrVecTy
5953 : FixedVectorType::get(
5954 ElementType: IntegerType::get(C&: Src->getContext(),
5955 NumBits: EltTyBits * CurrNumEltPerOp),
5956 NumElts: CurrVecTy->getNumElements() / CurrNumEltPerOp);
5957 assert(DL.getTypeSizeInBits(CoalescedVecTy) ==
5958 DL.getTypeSizeInBits(CurrVecTy) &&
5959 "coalesciing elements doesn't change vector width.");
5960
5961 while (NumEltRemaining > 0) {
5962 assert(SubVecEltsLeft >= 0 && "Subreg element count overconsumtion?");
5963
5964 // Can we use this vector size, as per the remaining element count?
5965 // Iff the vector is naturally aligned, we can do a wide load regardless.
5966 if (NumEltRemaining < CurrNumEltPerOp &&
5967 (!IsLoad || Alignment < CurrOpSizeBytes) && CurrOpSizeBytes != 1)
5968 break; // Try smalled vector size.
5969
5970 // This isn't exactly right. We're using slow unaligned 32-byte accesses
5971 // as a proxy for a double-pumped AVX memory interface such as on
5972 // Sandybridge.
5973 // Sub-32-bit loads/stores will be slower either with PINSR*/PEXTR* or
5974 // will be scalarized.
5975 //
5976 // For a vector load, each non-0th 1/2/4-byte in-lane remainder chunk is
5977 // materialized by a *single* folded PINSR*(mem) that both loads and
5978 // inserts the lane (1B->PINSRB, 2B->PINSRW, 4B->PINSRD; the byte and
5979 // dword folds need SSE4.1, the word fold only SSE2).
5980 // When that fold is available the chunk is one instruction, so it must be
5981 // priced once here (as a plain load) and the separate lane-insert charge
5982 // below must be skipped - otherwise the folded insert is double-counted.
5983 // Stores (the symmetric PEXTR*(mem) fold) are left unchanged here.
5984 bool Is0thSubVec = (NumEltDone() % LT.second.getVectorNumElements()) == 0;
5985 bool FoldedInLaneInsert = IsLoad && !Is0thSubVec &&
5986 ((CurrOpSizeBytes == 1 && ST->hasSSE41()) ||
5987 (CurrOpSizeBytes == 2 && ST->hasSSE2()) ||
5988 (CurrOpSizeBytes == 4 && ST->hasSSE41()));
5989 if (CurrOpSizeBytes == 32 && ST->isUnalignedMem32Slow())
5990 Cost += 2;
5991 else if (CurrOpSizeBytes < 4 && !FoldedInLaneInsert)
5992 Cost += 2;
5993 else
5994 Cost += 1;
5995
5996 // If we're loading a uniform value, then we don't need to split the load,
5997 // loading just a single (widest) vector can be reused by all splits.
5998 if (IsLoad && OpInfo.isUniform())
5999 return Cost;
6000
6001 // If we have fully processed the previous reg, we need to replenish it.
6002 if (SubVecEltsLeft == 0) {
6003 SubVecEltsLeft += CurrVecTy->getNumElements();
6004 // And that's free only for the 0'th subvector of a legalized vector.
6005 if (!Is0thSubVec)
6006 Cost +=
6007 getShuffleCost(Kind: IsLoad ? TTI::ShuffleKind::SK_InsertSubvector
6008 : TTI::ShuffleKind::SK_ExtractSubvector,
6009 DstTy: VTy, SrcTy: VTy, CostKind, Mask: {}, Index: NumEltDone(), SubTp: CurrVecTy);
6010 }
6011
6012 // While we can directly load/store ZMM, YMM, and 64-bit halves of XMM,
6013 // for smaller widths (32/16/8) we have to insert/extract them separately.
6014 // Again, it's free for the 0'th subreg (if op is 32/64 bit wide,
6015 // but let's pretend that it is also true for 16/8 bit wide ops...)
6016 if (CurrOpSizeBytes <= 32 / 8 && !Is0thSubVec && !FoldedInLaneInsert) {
6017 int NumEltDoneInCurrXMM = NumEltDone() % NumEltPerXMM;
6018 assert(NumEltDoneInCurrXMM % CurrNumEltPerOp == 0 && "");
6019 int CoalescedVecEltIdx = NumEltDoneInCurrXMM / CurrNumEltPerOp;
6020 APInt DemandedElts =
6021 APInt::getBitsSet(numBits: CoalescedVecTy->getNumElements(),
6022 loBit: CoalescedVecEltIdx, hiBit: CoalescedVecEltIdx + 1);
6023 assert(DemandedElts.popcount() == 1 && "Inserting single value");
6024 Cost += getScalarizationOverhead(Ty: CoalescedVecTy, DemandedElts, Insert: IsLoad,
6025 Extract: !IsLoad, CostKind);
6026 }
6027
6028 SubVecEltsLeft -= CurrNumEltPerOp;
6029 NumEltRemaining -= CurrNumEltPerOp;
6030 Alignment = commonAlignment(A: Alignment, Offset: CurrOpSizeBytes);
6031 }
6032 }
6033
6034 assert(NumEltRemaining <= 0 && "Should have processed all the elements.");
6035
6036 return Cost;
6037}
6038
6039InstructionCost
6040X86TTIImpl::getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA,
6041 TTI::TargetCostKind CostKind) const {
6042 switch (MICA.getID()) {
6043 case Intrinsic::masked_scatter:
6044 case Intrinsic::masked_gather:
6045 return getGatherScatterOpCost(MICA, CostKind);
6046 case Intrinsic::masked_load:
6047 case Intrinsic::masked_store:
6048 return getMaskedMemoryOpCost(MICA, CostKind);
6049 case Intrinsic::masked_compressstore:
6050 // Fallback to scalarization for slow compressstore targets (e.g. znver4).
6051 if (ST->isVecCompressStoreSlow())
6052 return BaseT::getMemIntrinsicInstrCost(MICA, CostKind);
6053 break;
6054 }
6055
6056 static const CostKindTblEntry AVX512VBMI2CostTable[] = {
6057 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6058 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6059 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6060
6061 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6062 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6063 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6064
6065 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v16i8, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6066 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6067 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v64i8, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 8 } },
6068
6069 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v8i16, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6070 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6071 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v32i16, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 8 } },
6072 };
6073
6074 static const CostKindTblEntry AVX512CostTable[] = {
6075 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6076 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6077 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6078 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6079 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6080 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6081
6082 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6083 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6084 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6085 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6086 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6087 { .ISD: Intrinsic::masked_expandload, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 1, .SizeAndLatencyCost: 3 } },
6088
6089 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v4i32, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6090 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v4f32, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6091 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v8i32, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6092 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v8f32, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6093 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v16i32, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 8 } },
6094 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v16f32, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 8 } },
6095
6096 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v2i64, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6097 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v2f64, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6098 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v4i64, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6099 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v4f64, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 1, .SizeAndLatencyCost: 7 } },
6100 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v8i64, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 8 } },
6101 { .ISD: Intrinsic::masked_compressstore, .Type: MVT::v8f64, .Cost: { .RecipThroughputCost: 3,.LatencyCost: 12, .CodeSizeCost: 1, .SizeAndLatencyCost: 8 } },
6102 };
6103
6104 std::pair<InstructionCost, MVT> LT =
6105 getTypeLegalizationCost(Ty: MICA.getDataType());
6106
6107 if (ST->hasVBMI2())
6108 if (const auto *Entry =
6109 CostTableLookup(Table: AVX512VBMI2CostTable, ISD: MICA.getID(), Ty: LT.second))
6110 if (auto KindCost = Entry->Cost[CostKind])
6111 return LT.first * *KindCost;
6112
6113 if (ST->hasAVX512())
6114 if (const auto *Entry =
6115 CostTableLookup(Table: AVX512CostTable, ISD: MICA.getID(), Ty: LT.second))
6116 if (auto KindCost = Entry->Cost[CostKind])
6117 return LT.first * *KindCost;
6118
6119 return BaseT::getMemIntrinsicInstrCost(MICA, CostKind);
6120}
6121
6122InstructionCost
6123X86TTIImpl::getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA,
6124 TTI::TargetCostKind CostKind) const {
6125 unsigned Opcode = MICA.getID() == Intrinsic::masked_load ? Instruction::Load
6126 : Instruction::Store;
6127 Type *SrcTy = MICA.getDataType();
6128 Align Alignment = MICA.getAlignment();
6129 unsigned AddressSpace = MICA.getAddressSpace();
6130
6131 bool IsLoad = (Instruction::Load == Opcode);
6132 bool IsStore = (Instruction::Store == Opcode);
6133
6134 auto *SrcVTy = dyn_cast<FixedVectorType>(Val: SrcTy);
6135 if (!SrcVTy)
6136 // To calculate scalar take the regular cost, without mask
6137 return getMemoryOpCost(Opcode, Src: SrcTy, Alignment, AddressSpace, CostKind);
6138
6139 unsigned NumElem = SrcVTy->getNumElements();
6140 auto *MaskTy =
6141 FixedVectorType::get(ElementType: Type::getInt8Ty(C&: SrcVTy->getContext()), NumElts: NumElem);
6142 if ((IsLoad && !isLegalMaskedLoad(DataType: SrcVTy, Alignment, AddressSpace)) ||
6143 (IsStore && !isLegalMaskedStore(DataType: SrcVTy, Alignment, AddressSpace))) {
6144 // Scalarization
6145 APInt DemandedElts = APInt::getAllOnes(numBits: NumElem);
6146 InstructionCost MaskSplitCost = getScalarizationOverhead(
6147 Ty: MaskTy, DemandedElts, /*Insert*/ false, /*Extract*/ true, CostKind);
6148 InstructionCost ScalarCompareCost = getCmpSelInstrCost(
6149 Opcode: Instruction::ICmp, ValTy: Type::getInt8Ty(C&: SrcVTy->getContext()), CondTy: nullptr,
6150 VecPred: CmpInst::BAD_ICMP_PREDICATE, CostKind);
6151 InstructionCost BranchCost = getCFInstrCost(Opcode: Instruction::CondBr, CostKind);
6152 InstructionCost MaskCmpCost = NumElem * (BranchCost + ScalarCompareCost);
6153 InstructionCost ValueSplitCost = getScalarizationOverhead(
6154 Ty: SrcVTy, DemandedElts, Insert: IsLoad, Extract: IsStore, CostKind);
6155 InstructionCost MemopCost =
6156 NumElem * BaseT::getMemoryOpCost(Opcode, Src: SrcVTy->getScalarType(),
6157 Alignment, AddressSpace, CostKind);
6158 return MemopCost + ValueSplitCost + MaskSplitCost + MaskCmpCost;
6159 }
6160
6161 // Legalize the type.
6162 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: SrcVTy);
6163 auto VT = TLI->getValueType(DL, Ty: SrcVTy);
6164 InstructionCost Cost = 0;
6165 MVT Ty = LT.second;
6166 if (Ty == MVT::i16 || Ty == MVT::i32 || Ty == MVT::i64)
6167 // APX masked load/store for scalar is cheap.
6168 return Cost + LT.first;
6169
6170 if (VT.isSimple() && Ty != VT.getSimpleVT() &&
6171 LT.second.getVectorNumElements() == NumElem)
6172 // Promotion requires extend/truncate for data and a shuffle for mask.
6173 Cost += getShuffleCost(Kind: TTI::SK_PermuteTwoSrc, DstTy: SrcVTy, SrcTy: SrcVTy, CostKind, Mask: {},
6174 Index: 0, SubTp: nullptr) +
6175 getShuffleCost(Kind: TTI::SK_PermuteTwoSrc, DstTy: MaskTy, SrcTy: MaskTy, CostKind, Mask: {},
6176 Index: 0, SubTp: nullptr);
6177
6178 else if (LT.first * Ty.getVectorNumElements() > NumElem) {
6179 auto *NewMaskTy = FixedVectorType::get(ElementType: MaskTy->getElementType(),
6180 NumElts: (unsigned)LT.first.getValue() *
6181 Ty.getVectorNumElements());
6182 // Expanding requires fill mask with zeroes
6183 Cost += getShuffleCost(Kind: TTI::SK_InsertSubvector, DstTy: NewMaskTy, SrcTy: NewMaskTy,
6184 CostKind, Mask: {}, Index: 0, SubTp: MaskTy);
6185 }
6186
6187 // Predicate fanout (see getPredicateFanoutCost): only a variable mask needs
6188 // a per-part sub-mask; a constant mask is folded by codegen and pays nothing.
6189 InstructionCost MaskExpCost = 0;
6190 if (MICA.getVariableMask()) {
6191 auto *MaskVecTy =
6192 FixedVectorType::get(ElementType: Type::getInt1Ty(C&: SrcVTy->getContext()), NumElts: NumElem);
6193 MaskExpCost = getPredicateFanoutCost(TTI: *this, DataVTy: SrcVTy, MaskVTy: MaskVecTy);
6194 }
6195
6196 // Pre-AVX512 - each maskmov load costs 2 + store costs ~8.
6197 if (!ST->hasAVX512())
6198 return Cost + LT.first * (IsLoad ? 2 : 8) + MaskExpCost;
6199
6200 // AVX-512 masked load/store is cheaper.
6201 return Cost + LT.first + MaskExpCost;
6202}
6203
6204InstructionCost X86TTIImpl::getPointersChainCost(
6205 ArrayRef<const Value *> Ptrs, const Value *Base,
6206 const TTI::PointersChainInfo &Info, Type *AccessTy,
6207 const TTI::TargetCostKind CostKind) const {
6208 if (Info.isSameBase() && Info.isKnownStride()) {
6209 // If all the pointers have known stride all the differences are translated
6210 // into constants. X86 memory addressing allows encoding it into
6211 // displacement. So we just need to take the base GEP cost.
6212 if (const auto *BaseGEP = dyn_cast<GetElementPtrInst>(Val: Base)) {
6213 SmallVector<const Value *> Indices(BaseGEP->indices());
6214 return getGEPCost(PointeeType: BaseGEP->getSourceElementType(),
6215 Ptr: BaseGEP->getPointerOperand(), Operands: Indices, CostKind,
6216 AccessType: nullptr);
6217 }
6218 return TTI::TCC_Free;
6219 }
6220 return BaseT::getPointersChainCost(Ptrs, Base, Info, AccessTy, CostKind);
6221}
6222
6223InstructionCost
6224X86TTIImpl::getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE,
6225 const SCEV *Ptr,
6226 TTI::TargetCostKind CostKind) const {
6227 // Address computations in vectorized code with non-consecutive addresses will
6228 // likely result in more instructions compared to scalar code where the
6229 // computation can more often be merged into the index mode. The resulting
6230 // extra micro-ops can significantly decrease throughput.
6231 const unsigned NumVectorInstToHideOverhead = 10;
6232
6233 // Cost modeling of Strided Access Computation is hidden by the indexing
6234 // modes of X86 regardless of the stride value. We dont believe that there
6235 // is a difference between constant strided access in gerenal and constant
6236 // strided value which is less than or equal to 64.
6237 // Even in the case of (loop invariant) stride whose value is not known at
6238 // compile time, the address computation will not incur more than one extra
6239 // ADD instruction.
6240 if (PtrTy->isVectorTy() && SE) {
6241 if (BaseT::isStridedAccess(Ptr) && !BaseT::getConstantStrideStep(SE, Ptr))
6242 return 1;
6243 if (!ST->hasAVX2()) {
6244 // TODO: AVX2 is the current cut-off because we don't have correct
6245 // interleaving costs for prior ISA's.
6246 if (!BaseT::isStridedAccess(Ptr))
6247 return NumVectorInstToHideOverhead;
6248 }
6249 }
6250
6251 return BaseT::getAddressComputationCost(PtrTy, SE, Ptr, CostKind);
6252}
6253
6254InstructionCost X86TTIImpl::getPartialReductionCost(
6255 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
6256 ElementCount VF, TTI::PartialReductionExtendKind OpAExtend,
6257 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
6258 TTI::TargetCostKind CostKind, std::optional<FastMathFlags> FMF) const {
6259 auto ExpandCost = [&]() {
6260 return BaseT::getPartialReductionCost(Opcode, InputTypeA, InputTypeB,
6261 AccumType, VF, OpAExtend, OpBExtend,
6262 BinOp, CostKind, FMF);
6263 };
6264
6265 // The dot product instructions multiply-accumulate i8 x i8 -> i32,
6266 // i16 x i16 -> i32, bf16 x bf16 -> f32 or f16 x f16 -> f32. Partial
6267 // reductions may also multiply inputs extended from different types, which
6268 // they can't handle.
6269 if (VF.isScalable() || !BinOp || OpAExtend == TTI::PR_None ||
6270 OpBExtend == TTI::PR_None || InputTypeA != InputTypeB)
6271 return ExpandCost();
6272
6273 unsigned Opc;
6274 if (Opcode == Instruction::Add && *BinOp == Instruction::Mul &&
6275 AccumType->isIntegerTy(BitWidth: 32) &&
6276 (InputTypeA->isIntegerTy(BitWidth: 8) || InputTypeA->isIntegerTy(BitWidth: 16))) {
6277 if (OpAExtend != OpBExtend)
6278 Opc = ISD::PARTIAL_REDUCE_SUMLA;
6279 else if (OpAExtend == TTI::PR_SignExtend)
6280 Opc = ISD::PARTIAL_REDUCE_SMLA;
6281 else
6282 Opc = ISD::PARTIAL_REDUCE_UMLA;
6283 } else if (Opcode == Instruction::FAdd && *BinOp == Instruction::FMul &&
6284 AccumType->isFloatTy() &&
6285 (InputTypeA->isBFloatTy() || InputTypeA->isHalfTy())) {
6286 // VDPBF16PS and VDPPHPS, like the expansion, reassociate the additions and
6287 // fuse the multiplications.
6288 if (!FMF || !FMF->allowReassoc() || !FMF->allowContract())
6289 return InstructionCost::getInvalid();
6290 Opc = ISD::PARTIAL_REDUCE_FMLA;
6291 } else {
6292 return ExpandCost();
6293 }
6294
6295 unsigned Ratio =
6296 AccumType->getScalarSizeInBits() / InputTypeA->getScalarSizeInBits();
6297 if (!VF.isKnownMultipleOf(RHS: Ratio))
6298 return ExpandCost();
6299
6300 // One dot product per legal accumulator vector. Accumulators narrower than
6301 // a legal vector are widened by expanding the partial reduction instead.
6302 auto *AccVecTy = VectorType::get(ElementType: AccumType, EC: VF.divideCoefficientBy(RHS: Ratio));
6303 auto *InputVecTy = VectorType::get(ElementType: InputTypeA, EC: VF);
6304 std::pair<InstructionCost, MVT> AccLT = getTypeLegalizationCost(Ty: AccVecTy);
6305 std::pair<InstructionCost, MVT> InputLT = getTypeLegalizationCost(Ty: InputVecTy);
6306 if (AccLT.second.getFixedSizeInBits() >
6307 AccVecTy->getPrimitiveSizeInBits().getFixedValue() ||
6308 !TLI->isPartialReduceMLALegalOrCustom(Opc, AccVT: AccLT.second, InputVT: InputLT.second))
6309 return ExpandCost();
6310
6311 return AccLT.first;
6312}
6313
6314InstructionCost
6315X86TTIImpl::getArithmeticReductionCost(unsigned Opcode, VectorType *ValTy,
6316 std::optional<FastMathFlags> FMF,
6317 TTI::TargetCostKind CostKind) const {
6318 if (TTI::requiresOrderedReduction(FMF))
6319 return BaseT::getArithmeticReductionCost(Opcode, Ty: ValTy, FMF, CostKind);
6320
6321 // We use llvm-mca across all supported CPUs to measure the logic cost stats.
6322 // We use the Intel Architecture Code Analyzer(IACA) to measure the throughput
6323 // and make it as the cost. TODO: Update old IACA numbers to llvm-mca.
6324
6325 static const CostKindTblEntry SLMCostTbl[] = {
6326 { .ISD: ISD::FADD, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6327 { .ISD: ISD::ADD, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6328 };
6329
6330 static const CostKindTblEntry SSE2CostTbl[] = {
6331 { .ISD: ISD::FADD, .Type: MVT::v2f64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} },
6332 { .ISD: ISD::FADD, .Type: MVT::v2f32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} },
6333 { .ISD: ISD::FADD, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} },
6334 { .ISD: ISD::ADD, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // The data reported by the IACA tool is "1.6".
6335 { .ISD: ISD::ADD, .Type: MVT::v2i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // FIXME: chosen to be less than v4i32
6336 { .ISD: ISD::ADD, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} }, // The data reported by the IACA tool is "3.3".
6337 { .ISD: ISD::ADD, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // The data reported by the IACA tool is "4.3".
6338 { .ISD: ISD::ADD, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} }, // The data reported by the IACA tool is "4.3".
6339 { .ISD: ISD::ADD, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} }, // The data reported by the IACA tool is "4.3".
6340 { .ISD: ISD::ADD, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} },
6341 { .ISD: ISD::ADD, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} },
6342 { .ISD: ISD::ADD, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} },
6343 { .ISD: ISD::ADD, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6344
6345 { .ISD: ISD::AND, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6346 { .ISD: ISD::AND, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6347 { .ISD: ISD::AND, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 8, .SizeAndLatencyCost: 8} },
6348 { .ISD: ISD::AND, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 6,.LatencyCost: 10,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6349 { .ISD: ISD::OR, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6350 { .ISD: ISD::OR, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6351 { .ISD: ISD::OR, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 8, .SizeAndLatencyCost: 8} },
6352 { .ISD: ISD::OR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 6,.LatencyCost: 10,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6353 { .ISD: ISD::XOR, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6354 { .ISD: ISD::XOR, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6355 { .ISD: ISD::XOR, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 8, .SizeAndLatencyCost: 8} },
6356 { .ISD: ISD::XOR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 6,.LatencyCost: 10,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6357 };
6358
6359 static const CostKindTblEntry AVX1CostTbl[] = {
6360 { .ISD: ISD::FADD, .Type: MVT::v4f64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6361 { .ISD: ISD::FADD, .Type: MVT::v4f32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6362 { .ISD: ISD::FADD, .Type: MVT::v8f32, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} },
6363 { .ISD: ISD::ADD, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 1, .CodeSizeCost: 1, .SizeAndLatencyCost: 1} }, // The data reported by the IACA tool is "1.5".
6364 { .ISD: ISD::ADD, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6365 { .ISD: ISD::ADD, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6366 { .ISD: ISD::ADD, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6367 { .ISD: ISD::ADD, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} },
6368
6369 { .ISD: ISD::AND, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6370 { .ISD: ISD::AND, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6371 { .ISD: ISD::AND, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 11, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6372 { .ISD: ISD::AND, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6373 { .ISD: ISD::AND, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 6,.LatencyCost: 13,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6374 { .ISD: ISD::AND, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 10, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6375 { .ISD: ISD::OR, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6376 { .ISD: ISD::OR, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6377 { .ISD: ISD::OR, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 11, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6378 { .ISD: ISD::OR, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6379 { .ISD: ISD::OR, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 6,.LatencyCost: 13,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6380 { .ISD: ISD::OR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 10, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6381 { .ISD: ISD::XOR, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6382 { .ISD: ISD::XOR, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6383 { .ISD: ISD::XOR, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 11, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6384 { .ISD: ISD::XOR, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6385 { .ISD: ISD::XOR, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 6,.LatencyCost: 13,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6386 { .ISD: ISD::XOR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 10, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6387 };
6388
6389 static const CostKindTblEntry AVX2CostTbl[] = {
6390 { .ISD: ISD::AND, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6391 { .ISD: ISD::AND, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6392 { .ISD: ISD::AND, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6393 { .ISD: ISD::AND, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6394 { .ISD: ISD::AND, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6395 { .ISD: ISD::AND, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6396 { .ISD: ISD::AND, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 13,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6397 { .ISD: ISD::AND, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6398 { .ISD: ISD::OR, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6399 { .ISD: ISD::OR, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6400 { .ISD: ISD::OR, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6401 { .ISD: ISD::OR, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6402 { .ISD: ISD::OR, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6403 { .ISD: ISD::OR, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6404 { .ISD: ISD::OR, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 13,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6405 { .ISD: ISD::OR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6406 { .ISD: ISD::XOR, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 7, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6407 { .ISD: ISD::XOR, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6408 { .ISD: ISD::XOR, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6409 { .ISD: ISD::XOR, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6410 { .ISD: ISD::XOR, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6411 { .ISD: ISD::XOR, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6412 { .ISD: ISD::XOR, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 13,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6413 { .ISD: ISD::XOR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6414 };
6415
6416 static const CostKindTblEntry AVX512FCostTbl[] = {
6417 { .ISD: ISD::FADD, .Type: MVT::v8f64, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} },
6418 { .ISD: ISD::FADD, .Type: MVT::v16f32, .Cost: {.RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6419 { .ISD: ISD::ADD, .Type: MVT::v8i64, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} },
6420 { .ISD: ISD::ADD, .Type: MVT::v16i32, .Cost: {.RecipThroughputCost: 6, .LatencyCost: 6, .CodeSizeCost: 6, .SizeAndLatencyCost: 6} },
6421
6422 { .ISD: ISD::AND, .Type: MVT::v8i64, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6423 { .ISD: ISD::AND, .Type: MVT::v16i32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6424 { .ISD: ISD::AND, .Type: MVT::v32i16, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 14,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6425 { .ISD: ISD::AND, .Type: MVT::v64i8, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 16,.CodeSizeCost: 13,.SizeAndLatencyCost: 13} },
6426 { .ISD: ISD::AND, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6427 { .ISD: ISD::OR, .Type: MVT::v8i64, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6428 { .ISD: ISD::OR, .Type: MVT::v16i32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6429 { .ISD: ISD::OR, .Type: MVT::v32i16, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 14,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6430 { .ISD: ISD::OR, .Type: MVT::v64i8, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 16,.CodeSizeCost: 13,.SizeAndLatencyCost: 13} },
6431 { .ISD: ISD::OR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6432 { .ISD: ISD::XOR, .Type: MVT::v8i64, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6433 { .ISD: ISD::XOR, .Type: MVT::v16i32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6434 { .ISD: ISD::XOR, .Type: MVT::v32i16, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 14,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6435 { .ISD: ISD::XOR, .Type: MVT::v64i8, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 16,.CodeSizeCost: 13,.SizeAndLatencyCost: 13} },
6436 { .ISD: ISD::XOR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6437 };
6438
6439 static const CostKindTblEntry AVX512BWCostTbl[] = {
6440 { .ISD: ISD::ADD, .Type: MVT::v32i16, .Cost: {.RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6441 { .ISD: ISD::ADD, .Type: MVT::v64i8, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} },
6442 };
6443
6444 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6445 assert(ISD && "Invalid opcode");
6446
6447 // Before legalizing the type, give a chance to look up illegal narrow types
6448 // in the table.
6449 // FIXME: Is there a better way to do this?
6450 EVT VT = TLI->getValueType(DL, Ty: ValTy);
6451 if (VT.isSimple()) {
6452 MVT MTy = VT.getSimpleVT();
6453 if (ST->useSLMArithCosts())
6454 if (const auto *Entry = CostTableLookup(Table: SLMCostTbl, ISD, Ty: MTy))
6455 if (auto KindCost = Entry->Cost[CostKind])
6456 return *KindCost;
6457
6458 if (ST->hasBWI())
6459 if (const auto *Entry = CostTableLookup(Table: AVX512BWCostTbl, ISD, Ty: MTy))
6460 if (auto KindCost = Entry->Cost[CostKind])
6461 return *KindCost;
6462
6463 if (ST->hasAVX512())
6464 if (const auto *Entry = CostTableLookup(Table: AVX512FCostTbl, ISD, Ty: MTy))
6465 if (auto KindCost = Entry->Cost[CostKind])
6466 return *KindCost;
6467
6468 if (ST->hasAVX2())
6469 if (const auto *Entry = CostTableLookup(Table: AVX2CostTbl, ISD, Ty: MTy))
6470 if (auto KindCost = Entry->Cost[CostKind])
6471 return *KindCost;
6472
6473 if (ST->hasAVX())
6474 if (const auto *Entry = CostTableLookup(Table: AVX1CostTbl, ISD, Ty: MTy))
6475 if (auto KindCost = Entry->Cost[CostKind])
6476 return *KindCost;
6477
6478 if (ST->hasSSE2())
6479 if (const auto *Entry = CostTableLookup(Table: SSE2CostTbl, ISD, Ty: MTy))
6480 if (auto KindCost = Entry->Cost[CostKind])
6481 return *KindCost;
6482 }
6483
6484 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: ValTy);
6485
6486 MVT MTy = LT.second;
6487
6488 auto *ValVTy = cast<FixedVectorType>(Val: ValTy);
6489
6490 InstructionCost ArithmeticCost = 0;
6491 if (LT.first != 1 && MTy.isVector() &&
6492 MTy.getVectorNumElements() < ValVTy->getNumElements()) {
6493 // Type needs to be split. We need LT.first - 1 arithmetic ops.
6494 auto *SingleOpTy = FixedVectorType::get(ElementType: ValVTy->getElementType(),
6495 NumElts: MTy.getVectorNumElements());
6496 ArithmeticCost = getArithmeticInstrCost(Opcode, Ty: SingleOpTy, CostKind);
6497 ArithmeticCost *= LT.first - 1;
6498 }
6499
6500 // FIXME: These assume a naive kshift+binop lowering, which is probably
6501 // conservative in most cases.
6502 static const CostKindTblEntry AVX512BoolReduction[] = {
6503 { .ISD: ISD::AND, .Type: MVT::v2i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6504 { .ISD: ISD::AND, .Type: MVT::v4i1, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6505 { .ISD: ISD::AND, .Type: MVT::v8i1, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6506 { .ISD: ISD::AND, .Type: MVT::v16i1, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 9, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6507 { .ISD: ISD::AND, .Type: MVT::v32i1, .Cost: {.RecipThroughputCost: 11,.LatencyCost: 11,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6508 { .ISD: ISD::AND, .Type: MVT::v64i1, .Cost: {.RecipThroughputCost: 13,.LatencyCost: 13,.CodeSizeCost: 13,.SizeAndLatencyCost: 13} },
6509 { .ISD: ISD::OR, .Type: MVT::v2i1, .Cost: { .RecipThroughputCost: 3, .LatencyCost: 3, .CodeSizeCost: 3, .SizeAndLatencyCost: 3} },
6510 { .ISD: ISD::OR, .Type: MVT::v4i1, .Cost: { .RecipThroughputCost: 5, .LatencyCost: 5, .CodeSizeCost: 5, .SizeAndLatencyCost: 5} },
6511 { .ISD: ISD::OR, .Type: MVT::v8i1, .Cost: { .RecipThroughputCost: 7, .LatencyCost: 7, .CodeSizeCost: 7, .SizeAndLatencyCost: 7} },
6512 { .ISD: ISD::OR, .Type: MVT::v16i1, .Cost: { .RecipThroughputCost: 9, .LatencyCost: 9, .CodeSizeCost: 9, .SizeAndLatencyCost: 9} },
6513 { .ISD: ISD::OR, .Type: MVT::v32i1, .Cost: {.RecipThroughputCost: 11,.LatencyCost: 11,.CodeSizeCost: 11,.SizeAndLatencyCost: 11} },
6514 { .ISD: ISD::OR, .Type: MVT::v64i1, .Cost: {.RecipThroughputCost: 13,.LatencyCost: 13,.CodeSizeCost: 13,.SizeAndLatencyCost: 13} },
6515 };
6516
6517 static const CostKindTblEntry AVX2BoolReduction[] = {
6518 { .ISD: ISD::AND, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // vpmovmskb + cmp
6519 { .ISD: ISD::AND, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // vpmovmskb + cmp
6520 { .ISD: ISD::OR, .Type: MVT::v16i16, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // vpmovmskb + cmp
6521 { .ISD: ISD::OR, .Type: MVT::v32i8, .Cost: { .RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // vpmovmskb + cmp
6522 };
6523
6524 static const CostKindTblEntry AVX1BoolReduction[] = {
6525 { .ISD: ISD::AND, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // vmovmskpd + cmp
6526 { .ISD: ISD::AND, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // vmovmskps + cmp
6527 { .ISD: ISD::AND, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} }, // vextractf128 + vpand + vpmovmskb + cmp
6528 { .ISD: ISD::AND, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} }, // vextractf128 + vpand + vpmovmskb + cmp
6529 { .ISD: ISD::OR, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // vmovmskpd + cmp
6530 { .ISD: ISD::OR, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // vmovmskps + cmp
6531 { .ISD: ISD::OR, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} }, // vextractf128 + vpor + vpmovmskb + cmp
6532 { .ISD: ISD::OR, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 4} }, // vextractf128 + vpor + vpmovmskb + cmp
6533 };
6534
6535 static const CostKindTblEntry SSE2BoolReduction[] = {
6536 { .ISD: ISD::AND, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // movmskpd + cmp
6537 { .ISD: ISD::AND, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // movmskps + cmp
6538 { .ISD: ISD::AND, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // pmovmskb + cmp
6539 { .ISD: ISD::AND, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // pmovmskb + cmp
6540 { .ISD: ISD::OR, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // movmskpd + cmp
6541 { .ISD: ISD::OR, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // movmskps + cmp
6542 { .ISD: ISD::OR, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // pmovmskb + cmp
6543 { .ISD: ISD::OR, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 2, .SizeAndLatencyCost: 2} }, // pmovmskb + cmp
6544 };
6545
6546 // Handle bool allof/anyof vXi1 patterns before we check legal types.
6547 if (ValVTy->getElementType()->isIntegerTy(BitWidth: 1)) {
6548 if (ISD == ISD::ADD) {
6549 // vXi1 addition reduction will bitcast to scalar and perform a popcount.
6550 auto *IntTy = IntegerType::getIntNTy(C&: ValVTy->getContext(),
6551 N: ValVTy->getNumElements());
6552 IntrinsicCostAttributes ICA(Intrinsic::ctpop, IntTy, {IntTy});
6553 return getCastInstrCost(Opcode: Instruction::BitCast, Dst: IntTy, Src: ValVTy,
6554 CCH: TargetTransformInfo::CastContextHint::None,
6555 CostKind) +
6556 getIntrinsicInstrCost(ICA, CostKind);
6557 }
6558
6559 if (ST->hasAVX512())
6560 if (const auto *Entry = CostTableLookup(Table: AVX512BoolReduction, ISD, Ty: MTy))
6561 if (auto KindCost = Entry->Cost[CostKind])
6562 return ArithmeticCost + *KindCost;
6563 if (ST->hasAVX2())
6564 if (const auto *Entry = CostTableLookup(Table: AVX2BoolReduction, ISD, Ty: MTy))
6565 if (auto KindCost = Entry->Cost[CostKind])
6566 return ArithmeticCost + *KindCost;
6567 if (ST->hasAVX())
6568 if (const auto *Entry = CostTableLookup(Table: AVX1BoolReduction, ISD, Ty: MTy))
6569 if (auto KindCost = Entry->Cost[CostKind])
6570 return ArithmeticCost + *KindCost;
6571 if (ST->hasSSE2())
6572 if (const auto *Entry = CostTableLookup(Table: SSE2BoolReduction, ISD, Ty: MTy))
6573 if (auto KindCost = Entry->Cost[CostKind])
6574 return ArithmeticCost + *KindCost;
6575
6576 return BaseT::getArithmeticReductionCost(Opcode, Ty: ValVTy, FMF, CostKind);
6577 }
6578
6579 // Special case: vXi8 mul reductions are performed as vXi16.
6580 if (ISD == ISD::MUL && MTy.getScalarType() == MVT::i8) {
6581 auto *WideSclTy = IntegerType::get(C&: ValVTy->getContext(), NumBits: 16);
6582 auto *WideVecTy = FixedVectorType::get(ElementType: WideSclTy, NumElts: ValVTy->getNumElements());
6583 return getCastInstrCost(Opcode: Instruction::ZExt, Dst: WideVecTy, Src: ValTy,
6584 CCH: TargetTransformInfo::CastContextHint::None,
6585 CostKind) +
6586 getArithmeticReductionCost(Opcode, ValTy: WideVecTy, FMF, CostKind);
6587 }
6588
6589 if (ST->useSLMArithCosts())
6590 if (const auto *Entry = CostTableLookup(Table: SLMCostTbl, ISD, Ty: MTy))
6591 if (auto KindCost = Entry->Cost[CostKind])
6592 return ArithmeticCost + *KindCost;
6593
6594 if (ST->hasBWI())
6595 if (const auto *Entry = CostTableLookup(Table: AVX512BWCostTbl, ISD, Ty: MTy))
6596 if (auto KindCost = Entry->Cost[CostKind])
6597 return ArithmeticCost + *KindCost;
6598
6599 if (ST->hasAVX512())
6600 if (const auto *Entry = CostTableLookup(Table: AVX512FCostTbl, ISD, Ty: MTy))
6601 if (auto KindCost = Entry->Cost[CostKind])
6602 return ArithmeticCost + *KindCost;
6603
6604 if (ST->hasAVX2())
6605 if (const auto *Entry = CostTableLookup(Table: AVX2CostTbl, ISD, Ty: MTy))
6606 if (auto KindCost = Entry->Cost[CostKind])
6607 return ArithmeticCost + *KindCost;
6608
6609 if (ST->hasAVX())
6610 if (const auto *Entry = CostTableLookup(Table: AVX1CostTbl, ISD, Ty: MTy))
6611 if (auto KindCost = Entry->Cost[CostKind])
6612 return ArithmeticCost + *KindCost;
6613
6614 if (ST->hasSSE2())
6615 if (const auto *Entry = CostTableLookup(Table: SSE2CostTbl, ISD, Ty: MTy))
6616 if (auto KindCost = Entry->Cost[CostKind])
6617 return ArithmeticCost + *KindCost;
6618
6619 unsigned NumVecElts = ValVTy->getNumElements();
6620 unsigned ScalarSize = ValVTy->getScalarSizeInBits();
6621
6622 // Special case power of 2 reductions where the scalar type isn't changed
6623 // by type legalization.
6624 if (!isPowerOf2_32(Value: NumVecElts) || ScalarSize != MTy.getScalarSizeInBits())
6625 return BaseT::getArithmeticReductionCost(Opcode, Ty: ValVTy, FMF, CostKind);
6626
6627 InstructionCost ReductionCost = 0;
6628
6629 auto *Ty = ValVTy;
6630 if (LT.first != 1 && MTy.isVector() &&
6631 MTy.getVectorNumElements() < ValVTy->getNumElements()) {
6632 // Type needs to be split. We need LT.first - 1 arithmetic ops.
6633 Ty = FixedVectorType::get(ElementType: ValVTy->getElementType(),
6634 NumElts: MTy.getVectorNumElements());
6635 ReductionCost = getArithmeticInstrCost(Opcode, Ty, CostKind);
6636 ReductionCost *= LT.first - 1;
6637 NumVecElts = MTy.getVectorNumElements();
6638 }
6639
6640 // Now handle reduction with the legal type, taking into account size changes
6641 // at each level.
6642 while (NumVecElts > 1) {
6643 // Determine the size of the remaining vector we need to reduce.
6644 unsigned Size = NumVecElts * ScalarSize;
6645 NumVecElts /= 2;
6646 // If we're reducing from 256/512 bits, use an extract_subvector.
6647 if (Size > 128) {
6648 auto *SubTy = FixedVectorType::get(ElementType: ValVTy->getElementType(), NumElts: NumVecElts);
6649 ReductionCost += getShuffleCost(Kind: TTI::SK_ExtractSubvector, DstTy: Ty, SrcTy: Ty,
6650 CostKind, Mask: {}, Index: NumVecElts, SubTp: SubTy);
6651 Ty = SubTy;
6652 } else if (Size == 128) {
6653 // Reducing from 128 bits is a permute of v2f64/v2i64.
6654 FixedVectorType *ShufTy;
6655 if (ValVTy->isFloatingPointTy())
6656 ShufTy =
6657 FixedVectorType::get(ElementType: Type::getDoubleTy(C&: ValVTy->getContext()), NumElts: 2);
6658 else
6659 ShufTy =
6660 FixedVectorType::get(ElementType: Type::getInt64Ty(C&: ValVTy->getContext()), NumElts: 2);
6661 ReductionCost += getShuffleCost(Kind: TTI::SK_PermuteSingleSrc, DstTy: ShufTy, SrcTy: ShufTy,
6662 CostKind, Mask: {}, Index: 0, SubTp: nullptr);
6663 } else if (Size == 64) {
6664 // Reducing from 64 bits is a shuffle of v4f32/v4i32.
6665 FixedVectorType *ShufTy;
6666 if (ValVTy->isFloatingPointTy())
6667 ShufTy =
6668 FixedVectorType::get(ElementType: Type::getFloatTy(C&: ValVTy->getContext()), NumElts: 4);
6669 else
6670 ShufTy =
6671 FixedVectorType::get(ElementType: Type::getInt32Ty(C&: ValVTy->getContext()), NumElts: 4);
6672 ReductionCost += getShuffleCost(Kind: TTI::SK_PermuteSingleSrc, DstTy: ShufTy, SrcTy: ShufTy,
6673 CostKind, Mask: {}, Index: 0, SubTp: nullptr);
6674 } else {
6675 // Reducing from smaller size is a shift by immediate.
6676 auto *ShiftTy = FixedVectorType::get(
6677 ElementType: Type::getIntNTy(C&: ValVTy->getContext(), N: Size), NumElts: 128 / Size);
6678 ReductionCost += getArithmeticInstrCost(
6679 Opcode: Instruction::LShr, Ty: ShiftTy, CostKind,
6680 Op1Info: {.Kind: TargetTransformInfo::OK_AnyValue, .Properties: TargetTransformInfo::OP_None},
6681 Op2Info: {.Kind: TargetTransformInfo::OK_UniformConstantValue, .Properties: TargetTransformInfo::OP_None});
6682 }
6683
6684 // Add the arithmetic op for this level.
6685 ReductionCost += getArithmeticInstrCost(Opcode, Ty, CostKind);
6686 }
6687
6688 // Add the final extract element to the cost.
6689 return ReductionCost + getVectorInstrCost(Opcode: Instruction::ExtractElement, Val: Ty,
6690 CostKind, Index: 0, Op0: nullptr, Op1: nullptr,
6691 VIC: TTI::VectorInstrContext::None);
6692}
6693
6694InstructionCost X86TTIImpl::getMinMaxCost(Intrinsic::ID IID, Type *Ty,
6695 TTI::TargetCostKind CostKind,
6696 FastMathFlags FMF) const {
6697 IntrinsicCostAttributes ICA(IID, Ty, {Ty, Ty}, FMF);
6698 return getIntrinsicInstrCost(ICA, CostKind);
6699}
6700
6701InstructionCost
6702X86TTIImpl::getMinMaxReductionCost(Intrinsic::ID IID, VectorType *ValTy,
6703 FastMathFlags FMF,
6704 TTI::TargetCostKind CostKind) const {
6705 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty: ValTy);
6706
6707 MVT MTy = LT.second;
6708
6709 ISD::NodeType ISD;
6710 if (ValTy->isIntOrIntVectorTy()) {
6711 ISD = (IID == Intrinsic::umin || IID == Intrinsic::umax) ? ISD::UMIN
6712 : ISD::SMIN;
6713 } else {
6714 assert(ValTy->isFPOrFPVectorTy() &&
6715 "Expected float point or integer vector type.");
6716 ISD = (IID == Intrinsic::minnum || IID == Intrinsic::maxnum)
6717 ? ISD::FMINNUM
6718 : ISD::FMINIMUM;
6719 }
6720
6721 // We use llvm-mca across all supported CPUs to measure the cost stats.
6722 static const CostKindTblEntry SSE2CostTbl[] = {
6723 {.ISD: ISD::SMIN, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 6}},
6724 {.ISD: ISD::UMIN, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 6}},
6725 {.ISD: ISD::SMIN, .Type: MVT::v2i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6}},
6726 {.ISD: ISD::UMIN, .Type: MVT::v2i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 5, .SizeAndLatencyCost: 6}},
6727 {.ISD: ISD::SMIN, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 7,.CodeSizeCost: 11,.SizeAndLatencyCost: 12}},
6728 {.ISD: ISD::UMIN, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 7,.CodeSizeCost: 14,.SizeAndLatencyCost: 15}},
6729 {.ISD: ISD::SMIN, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 4}},
6730 {.ISD: ISD::UMIN, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 6}},
6731 {.ISD: ISD::SMIN, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 6, .SizeAndLatencyCost: 6}},
6732 {.ISD: ISD::UMIN, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 8, .SizeAndLatencyCost: 10}},
6733 {.ISD: ISD::SMIN, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 8, .SizeAndLatencyCost: 8}},
6734 {.ISD: ISD::UMIN, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 8,.CodeSizeCost: 12,.SizeAndLatencyCost: 14}},
6735 {.ISD: ISD::SMIN, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 5, .SizeAndLatencyCost: 6}},
6736 {.ISD: ISD::UMIN, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 4}},
6737 {.ISD: ISD::SMIN, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 6,.CodeSizeCost: 12,.SizeAndLatencyCost: 13}},
6738 {.ISD: ISD::UMIN, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6739 {.ISD: ISD::SMIN, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 5, .LatencyCost: 9,.CodeSizeCost: 18,.SizeAndLatencyCost: 19}},
6740 {.ISD: ISD::UMIN, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 9, .SizeAndLatencyCost: 9}},
6741 {.ISD: ISD::SMIN, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 7,.LatencyCost: 13,.CodeSizeCost: 24,.SizeAndLatencyCost: 25}},
6742 {.ISD: ISD::UMIN, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 10,.CodeSizeCost: 11,.SizeAndLatencyCost: 11}},
6743 };
6744
6745 static const CostKindTblEntry SSE41CostTbl[] = {
6746 {.ISD: ISD::SMIN, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 6}},
6747 {.ISD: ISD::UMIN, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 4, .SizeAndLatencyCost: 6}},
6748 {.ISD: ISD::SMIN, .Type: MVT::v2i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6749 {.ISD: ISD::UMIN, .Type: MVT::v2i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6750 {.ISD: ISD::SMIN, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6751 {.ISD: ISD::UMIN, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6752 {.ISD: ISD::UMIN, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 4}},
6753 {.ISD: ISD::SMIN, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 6, .SizeAndLatencyCost: 6}},
6754 {.ISD: ISD::UMIN, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 5, .CodeSizeCost: 6, .SizeAndLatencyCost: 6}},
6755 {.ISD: ISD::SMIN, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 8, .CodeSizeCost: 4, .SizeAndLatencyCost: 5}},
6756 {.ISD: ISD::UMIN, .Type: MVT::v8i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 5, .CodeSizeCost: 2, .SizeAndLatencyCost: 2}},
6757 {.ISD: ISD::SMIN, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 3, .CodeSizeCost: 4, .SizeAndLatencyCost: 4}},
6758 {.ISD: ISD::SMIN, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6759 {.ISD: ISD::SMIN, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 8, .CodeSizeCost: 9, .SizeAndLatencyCost: 9}},
6760 {.ISD: ISD::SMIN, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 7, .SizeAndLatencyCost: 8}},
6761 {.ISD: ISD::UMIN, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 8, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6762 };
6763
6764 static const CostKindTblEntry AVX1CostTbl[] = {
6765 {.ISD: ISD::SMIN, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 11, .CodeSizeCost: 7,.SizeAndLatencyCost: 10}},
6766 {.ISD: ISD::UMIN, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 6,.LatencyCost: 12,.CodeSizeCost: 10,.SizeAndLatencyCost: 13}},
6767 {.ISD: ISD::SMIN, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6768 {.ISD: ISD::UMIN, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 4, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6769 {.ISD: ISD::SMIN, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 15, .CodeSizeCost: 6, .SizeAndLatencyCost: 7}},
6770 {.ISD: ISD::UMIN, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 9, .CodeSizeCost: 4, .SizeAndLatencyCost: 4}},
6771 {.ISD: ISD::SMIN, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 17, .CodeSizeCost: 8, .SizeAndLatencyCost: 9}},
6772 {.ISD: ISD::UMIN, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 11, .CodeSizeCost: 6, .SizeAndLatencyCost: 6}},
6773 };
6774
6775 static const CostKindTblEntry AVX2CostTbl[] = {
6776 {.ISD: ISD::SMIN, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 11, .CodeSizeCost: 7,.SizeAndLatencyCost: 10}},
6777 {.ISD: ISD::UMIN, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 12,.CodeSizeCost: 10,.SizeAndLatencyCost: 13}},
6778 {.ISD: ISD::SMIN, .Type: MVT::v2i32, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6779 {.ISD: ISD::UMIN, .Type: MVT::v2i32, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6780 {.ISD: ISD::UMIN, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6781 {.ISD: ISD::SMIN, .Type: MVT::v4i32, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6782 {.ISD: ISD::SMIN, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6783 {.ISD: ISD::UMIN, .Type: MVT::v8i32, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 9, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6784 {.ISD: ISD::SMIN, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6785 {.ISD: ISD::UMIN, .Type: MVT::v4i16, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6786 {.ISD: ISD::SMIN, .Type: MVT::v16i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 15, .CodeSizeCost: 6, .SizeAndLatencyCost: 7}},
6787 {.ISD: ISD::SMIN, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6788 {.ISD: ISD::UMIN, .Type: MVT::v8i8, .Cost: {.RecipThroughputCost: 3, .LatencyCost: 6, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6789 {.ISD: ISD::SMIN, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 17, .CodeSizeCost: 8, .SizeAndLatencyCost: 9}},
6790 };
6791
6792 static const CostKindTblEntry AVX512FCostTbl[] = {
6793 {.ISD: ISD::SMIN, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6794 {.ISD: ISD::UMIN, .Type: MVT::v2i64, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6795 {.ISD: ISD::SMIN, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6796 {.ISD: ISD::UMIN, .Type: MVT::v4i64, .Cost: {.RecipThroughputCost: 3,.LatencyCost: 10, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6797 {.ISD: ISD::SMIN, .Type: MVT::v8i64, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 16, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6798 {.ISD: ISD::UMIN, .Type: MVT::v8i64, .Cost: {.RecipThroughputCost: 5,.LatencyCost: 16, .CodeSizeCost: 7, .SizeAndLatencyCost: 7}},
6799 {.ISD: ISD::SMIN, .Type: MVT::v16i32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 9}},
6800 {.ISD: ISD::UMIN, .Type: MVT::v16i32, .Cost: {.RecipThroughputCost: 4,.LatencyCost: 12, .CodeSizeCost: 9, .SizeAndLatencyCost: 9}},
6801 };
6802
6803 static const CostKindTblEntry AVX512BWCostTbl[] = {
6804 {.ISD: ISD::SMIN, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6805 {.ISD: ISD::UMIN, .Type: MVT::v2i16, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6806 {.ISD: ISD::SMIN, .Type: MVT::v32i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 19, .CodeSizeCost: 8, .SizeAndLatencyCost: 9}},
6807 {.ISD: ISD::UMIN, .Type: MVT::v32i16, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 12, .CodeSizeCost: 6, .SizeAndLatencyCost: 6}},
6808 {.ISD: ISD::SMIN, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6809 {.ISD: ISD::UMIN, .Type: MVT::v2i8, .Cost: {.RecipThroughputCost: 1, .LatencyCost: 2, .CodeSizeCost: 3, .SizeAndLatencyCost: 3}},
6810 {.ISD: ISD::SMIN, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6811 {.ISD: ISD::UMIN, .Type: MVT::v4i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 4, .CodeSizeCost: 5, .SizeAndLatencyCost: 5}},
6812 {.ISD: ISD::SMIN, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 10, .CodeSizeCost: 6, .SizeAndLatencyCost: 7}},
6813 {.ISD: ISD::UMIN, .Type: MVT::v16i8, .Cost: {.RecipThroughputCost: 2, .LatencyCost: 6, .CodeSizeCost: 4, .SizeAndLatencyCost: 4}},
6814 {.ISD: ISD::SMIN, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 17, .CodeSizeCost: 8, .SizeAndLatencyCost: 9}},
6815 {.ISD: ISD::UMIN, .Type: MVT::v32i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 10, .CodeSizeCost: 6, .SizeAndLatencyCost: 6}},
6816 {.ISD: ISD::SMIN, .Type: MVT::v64i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 21,.CodeSizeCost: 10,.SizeAndLatencyCost: 11}},
6817 {.ISD: ISD::UMIN, .Type: MVT::v64i8, .Cost: {.RecipThroughputCost: 2,.LatencyCost: 14, .CodeSizeCost: 8, .SizeAndLatencyCost: 8}},
6818 };
6819
6820 // Before legalizing the type, give a chance to look up illegal narrow types
6821 // in the table.
6822 // FIXME: Is there a better way to do this?
6823 EVT VT = TLI->getValueType(DL, Ty: ValTy);
6824 if (VT.isSimple()) {
6825 MVT MTy = VT.getSimpleVT();
6826 if (ST->hasBWI())
6827 if (const auto *Entry = CostTableLookup(Table: AVX512BWCostTbl, ISD, Ty: MTy))
6828 if (auto KindCost = Entry->Cost[CostKind])
6829 return *KindCost;
6830
6831 if (ST->hasAVX512())
6832 if (const auto *Entry = CostTableLookup(Table: AVX512FCostTbl, ISD, Ty: MTy))
6833 if (auto KindCost = Entry->Cost[CostKind])
6834 return *KindCost;
6835
6836 if (ST->hasAVX2())
6837 if (const auto *Entry = CostTableLookup(Table: AVX2CostTbl, ISD, Ty: MTy))
6838 if (auto KindCost = Entry->Cost[CostKind])
6839 return *KindCost;
6840
6841 if (ST->hasAVX())
6842 if (const auto *Entry = CostTableLookup(Table: AVX1CostTbl, ISD, Ty: MTy))
6843 if (auto KindCost = Entry->Cost[CostKind])
6844 return *KindCost;
6845
6846 if (ST->hasSSE41())
6847 if (const auto *Entry = CostTableLookup(Table: SSE41CostTbl, ISD, Ty: MTy))
6848 if (auto KindCost = Entry->Cost[CostKind])
6849 return *KindCost;
6850
6851 if (ST->hasSSE2())
6852 if (const auto *Entry = CostTableLookup(Table: SSE2CostTbl, ISD, Ty: MTy))
6853 if (auto KindCost = Entry->Cost[CostKind])
6854 return *KindCost;
6855 }
6856
6857 auto *ValVTy = cast<FixedVectorType>(Val: ValTy);
6858 unsigned NumVecElts = ValVTy->getNumElements();
6859
6860 auto *Ty = ValVTy;
6861 InstructionCost MinMaxCost = 0;
6862 if (LT.first != 1 && MTy.isVector() &&
6863 MTy.getVectorNumElements() < ValVTy->getNumElements()) {
6864 // Type needs to be split. We need LT.first - 1 operations ops.
6865 Ty = FixedVectorType::get(ElementType: ValVTy->getElementType(),
6866 NumElts: MTy.getVectorNumElements());
6867 MinMaxCost = getMinMaxCost(IID, Ty, CostKind, FMF);
6868 MinMaxCost *= LT.first - 1;
6869 NumVecElts = MTy.getVectorNumElements();
6870 }
6871
6872 if (ST->hasBWI())
6873 if (const auto *Entry = CostTableLookup(Table: AVX512BWCostTbl, ISD, Ty: MTy))
6874 if (auto KindCost = Entry->Cost[CostKind])
6875 return MinMaxCost + *KindCost;
6876
6877 if (ST->hasAVX512())
6878 if (const auto *Entry = CostTableLookup(Table: AVX512FCostTbl, ISD, Ty: MTy))
6879 if (auto KindCost = Entry->Cost[CostKind])
6880 return MinMaxCost + *KindCost;
6881
6882 if (ST->hasAVX2())
6883 if (const auto *Entry = CostTableLookup(Table: AVX2CostTbl, ISD, Ty: MTy))
6884 if (auto KindCost = Entry->Cost[CostKind])
6885 return MinMaxCost + *KindCost;
6886
6887 if (ST->hasAVX())
6888 if (const auto *Entry = CostTableLookup(Table: AVX1CostTbl, ISD, Ty: MTy))
6889 if (auto KindCost = Entry->Cost[CostKind])
6890 return MinMaxCost + *KindCost;
6891
6892 if (ST->hasSSE41())
6893 if (const auto *Entry = CostTableLookup(Table: SSE41CostTbl, ISD, Ty: MTy))
6894 if (auto KindCost = Entry->Cost[CostKind])
6895 return MinMaxCost + *KindCost;
6896
6897 if (ST->hasSSE2())
6898 if (const auto *Entry = CostTableLookup(Table: SSE2CostTbl, ISD, Ty: MTy))
6899 if (auto KindCost = Entry->Cost[CostKind])
6900 return MinMaxCost + *KindCost;
6901
6902 unsigned ScalarSize = ValTy->getScalarSizeInBits();
6903
6904 // Special case power of 2 reductions where the scalar type isn't changed
6905 // by type legalization.
6906 if (!isPowerOf2_32(Value: ValVTy->getNumElements()) ||
6907 ScalarSize != MTy.getScalarSizeInBits())
6908 return BaseT::getMinMaxReductionCost(IID, Ty: ValTy, FMF, CostKind);
6909
6910 // Now handle reduction with the legal type, taking into account size changes
6911 // at each level.
6912 while (NumVecElts > 1) {
6913 // Determine the size of the remaining vector we need to reduce.
6914 unsigned Size = NumVecElts * ScalarSize;
6915 NumVecElts /= 2;
6916 // If we're reducing from 256/512 bits, use an extract_subvector.
6917 if (Size > 128) {
6918 auto *SubTy = FixedVectorType::get(ElementType: ValVTy->getElementType(), NumElts: NumVecElts);
6919 MinMaxCost += getShuffleCost(Kind: TTI::SK_ExtractSubvector, DstTy: Ty, SrcTy: Ty, CostKind,
6920 Mask: {}, Index: NumVecElts, SubTp: SubTy);
6921 Ty = SubTy;
6922 } else if (Size == 128) {
6923 // Reducing from 128 bits is a permute of v2f64/v2i64.
6924 VectorType *ShufTy;
6925 if (ValTy->isFloatingPointTy())
6926 ShufTy =
6927 FixedVectorType::get(ElementType: Type::getDoubleTy(C&: ValTy->getContext()), NumElts: 2);
6928 else
6929 ShufTy = FixedVectorType::get(ElementType: Type::getInt64Ty(C&: ValTy->getContext()), NumElts: 2);
6930 MinMaxCost += getShuffleCost(Kind: TTI::SK_PermuteSingleSrc, DstTy: ShufTy, SrcTy: ShufTy,
6931 CostKind, Mask: {}, Index: 0, SubTp: nullptr);
6932 } else if (Size == 64) {
6933 // Reducing from 64 bits is a shuffle of v4f32/v4i32.
6934 FixedVectorType *ShufTy;
6935 if (ValTy->isFloatingPointTy())
6936 ShufTy = FixedVectorType::get(ElementType: Type::getFloatTy(C&: ValTy->getContext()), NumElts: 4);
6937 else
6938 ShufTy = FixedVectorType::get(ElementType: Type::getInt32Ty(C&: ValTy->getContext()), NumElts: 4);
6939 MinMaxCost += getShuffleCost(Kind: TTI::SK_PermuteSingleSrc, DstTy: ShufTy, SrcTy: ShufTy,
6940 CostKind, Mask: {}, Index: 0, SubTp: nullptr);
6941 } else {
6942 // Reducing from smaller size is a shift by immediate.
6943 auto *ShiftTy = FixedVectorType::get(
6944 ElementType: Type::getIntNTy(C&: ValTy->getContext(), N: Size), NumElts: 128 / Size);
6945 MinMaxCost += getArithmeticInstrCost(
6946 Opcode: Instruction::LShr, Ty: ShiftTy, CostKind: TTI::TCK_RecipThroughput,
6947 Op1Info: {.Kind: TargetTransformInfo::OK_AnyValue, .Properties: TargetTransformInfo::OP_None},
6948 Op2Info: {.Kind: TargetTransformInfo::OK_UniformConstantValue, .Properties: TargetTransformInfo::OP_None});
6949 }
6950
6951 // Add the arithmetic op for this level.
6952 MinMaxCost += getMinMaxCost(IID, Ty, CostKind, FMF);
6953 }
6954
6955 // Add the final extract element to the cost.
6956 return MinMaxCost + getVectorInstrCost(Opcode: Instruction::ExtractElement, Val: Ty,
6957 CostKind, Index: 0, Op0: nullptr, Op1: nullptr,
6958 VIC: TTI::VectorInstrContext::None);
6959}
6960
6961/// Calculate the cost of materializing a 64-bit value. This helper
6962/// method might only calculate a fraction of a larger immediate. Therefore it
6963/// is valid to return a cost of ZERO.
6964InstructionCost X86TTIImpl::getIntImmCost(int64_t Val) const {
6965 if (Val == 0)
6966 return TTI::TCC_Free;
6967
6968 if (isInt<32>(x: Val))
6969 return TTI::TCC_Basic;
6970
6971 return 2 * TTI::TCC_Basic;
6972}
6973
6974InstructionCost X86TTIImpl::getIntImmCost(const APInt &Imm, Type *Ty,
6975 TTI::TargetCostKind CostKind) const {
6976 assert(Ty->isIntegerTy());
6977
6978 unsigned BitSize = Ty->getPrimitiveSizeInBits();
6979 if (BitSize == 0)
6980 return ~0U;
6981
6982 // Never hoist constants larger than 128bit, because this might lead to
6983 // incorrect code generation or assertions in codegen.
6984 // Fixme: Create a cost model for types larger than i128 once the codegen
6985 // issues have been fixed.
6986 if (BitSize > 128)
6987 return TTI::TCC_Free;
6988
6989 if (Imm == 0)
6990 return TTI::TCC_Free;
6991
6992 // Sign-extend all constants to a multiple of 64-bit.
6993 APInt ImmVal = Imm;
6994 if (BitSize % 64 != 0)
6995 ImmVal = Imm.sext(width: alignTo(Value: BitSize, Align: 64));
6996
6997 // Split the constant into 64-bit chunks and calculate the cost for each
6998 // chunk.
6999 InstructionCost Cost = 0;
7000 for (unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
7001 APInt Tmp = ImmVal.ashr(ShiftAmt: ShiftVal).sextOrTrunc(width: 64);
7002 int64_t Val = Tmp.getSExtValue();
7003 Cost += getIntImmCost(Val);
7004 }
7005 // We need at least one instruction to materialize the constant.
7006 return std::max<InstructionCost>(a: 1, b: Cost);
7007}
7008
7009InstructionCost X86TTIImpl::getIntImmCostInst(unsigned Opcode, unsigned Idx,
7010 const APInt &Imm, Type *Ty,
7011 TTI::TargetCostKind CostKind,
7012 Instruction *Inst) const {
7013 assert(Ty->isIntegerTy());
7014
7015 unsigned BitSize = Ty->getPrimitiveSizeInBits();
7016 unsigned ImmBitWidth = Imm.getBitWidth();
7017
7018 // There is no cost model for constants with a bit size of 0. Return TCC_Free
7019 // here, so that constant hoisting will ignore this constant.
7020 if (BitSize == 0)
7021 return TTI::TCC_Free;
7022
7023 unsigned ImmIdx = ~0U;
7024 switch (Opcode) {
7025 default:
7026 return TTI::TCC_Free;
7027 case Instruction::GetElementPtr:
7028 // Always hoist the base address of a GetElementPtr. This prevents the
7029 // creation of new constants for every base constant that gets constant
7030 // folded with the offset.
7031 if (Idx == 0)
7032 return 2 * TTI::TCC_Basic;
7033 return TTI::TCC_Free;
7034 case Instruction::Store:
7035 ImmIdx = 0;
7036 break;
7037 case Instruction::ICmp:
7038 // This is an imperfect hack to prevent constant hoisting of
7039 // compares that might be trying to check if a 64-bit value fits in
7040 // 32-bits. The backend can optimize these cases using a right shift by 32.
7041 // There are other predicates and immediates the backend can use shifts for.
7042 if (Idx == 1 && ImmBitWidth == 64) {
7043 uint64_t ImmVal = Imm.getZExtValue();
7044 if (ImmVal == 0x100000000ULL || ImmVal == 0xffffffff)
7045 return TTI::TCC_Free;
7046
7047 if (auto *Cmp = dyn_cast_or_null<CmpInst>(Val: Inst)) {
7048 if (Cmp->isEquality()) {
7049 KnownBits Known = computeKnownBits(V: Cmp->getOperand(i_nocapture: 0), DL);
7050 if (Known.countMinTrailingZeros() >= 32)
7051 return TTI::TCC_Free;
7052 }
7053 }
7054 }
7055 ImmIdx = 1;
7056 break;
7057 case Instruction::And:
7058 // We support 64-bit ANDs with immediates with 32-bits of leading zeroes
7059 // by using a 32-bit operation with implicit zero extension. Detect such
7060 // immediates here as the normal path expects bit 31 to be sign extended.
7061 if (Idx == 1 && ImmBitWidth == 64 && Imm.isIntN(N: 32))
7062 return TTI::TCC_Free;
7063 // If we have BMI then we can use BEXTR/BZHI to mask out upper i64 bits.
7064 if (Idx == 1 && ImmBitWidth == 64 && ST->is64Bit() && ST->hasBMI() &&
7065 Imm.isMask())
7066 return X86TTIImpl::getIntImmCost(Val: ST->hasBMI2() ? 255 : 65535);
7067 ImmIdx = 1;
7068 break;
7069 case Instruction::Add:
7070 case Instruction::Sub:
7071 // For add/sub, we can use the opposite instruction for INT32_MIN.
7072 if (Idx == 1 && ImmBitWidth == 64 && Imm.getZExtValue() == 0x80000000)
7073 return TTI::TCC_Free;
7074 ImmIdx = 1;
7075 break;
7076 case Instruction::UDiv:
7077 case Instruction::SDiv:
7078 case Instruction::URem:
7079 case Instruction::SRem:
7080 // Division by constant is typically expanded later into a different
7081 // instruction sequence. This completely changes the constants.
7082 // Report them as "free" to stop ConstantHoist from marking them as opaque.
7083 return TTI::TCC_Free;
7084 case Instruction::Mul:
7085 case Instruction::Or:
7086 case Instruction::Xor:
7087 ImmIdx = 1;
7088 break;
7089 // Always return TCC_Free for the shift value of a shift instruction.
7090 case Instruction::Shl:
7091 case Instruction::LShr:
7092 case Instruction::AShr:
7093 if (Idx == 1)
7094 return TTI::TCC_Free;
7095 break;
7096 case Instruction::Trunc:
7097 case Instruction::ZExt:
7098 case Instruction::SExt:
7099 case Instruction::IntToPtr:
7100 case Instruction::PtrToInt:
7101 case Instruction::BitCast:
7102 case Instruction::PHI:
7103 case Instruction::Call:
7104 case Instruction::Select:
7105 case Instruction::Ret:
7106 case Instruction::Load:
7107 break;
7108 }
7109
7110 if (Idx == ImmIdx) {
7111 uint64_t NumConstants = divideCeil(Numerator: BitSize, Denominator: 64);
7112 InstructionCost Cost = X86TTIImpl::getIntImmCost(Imm, Ty, CostKind);
7113 return (Cost <= NumConstants * TTI::TCC_Basic)
7114 ? static_cast<int>(TTI::TCC_Free)
7115 : Cost;
7116 }
7117
7118 return X86TTIImpl::getIntImmCost(Imm, Ty, CostKind);
7119}
7120
7121InstructionCost
7122X86TTIImpl::getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx,
7123 const APInt &Imm, Type *Ty,
7124 TTI::TargetCostKind CostKind) const {
7125 assert(Ty->isIntegerTy());
7126
7127 unsigned BitSize = Ty->getPrimitiveSizeInBits();
7128 // There is no cost model for constants with a bit size of 0. Return TCC_Free
7129 // here, so that constant hoisting will ignore this constant.
7130 if (BitSize == 0)
7131 return TTI::TCC_Free;
7132
7133 switch (IID) {
7134 default:
7135 return TTI::TCC_Free;
7136 case Intrinsic::sadd_with_overflow:
7137 case Intrinsic::uadd_with_overflow:
7138 case Intrinsic::ssub_with_overflow:
7139 case Intrinsic::usub_with_overflow:
7140 case Intrinsic::smul_with_overflow:
7141 case Intrinsic::umul_with_overflow:
7142 if ((Idx == 1) && Imm.getBitWidth() <= 64 && Imm.isSignedIntN(N: 32))
7143 return TTI::TCC_Free;
7144 break;
7145 case Intrinsic::experimental_stackmap:
7146 if ((Idx < 2) || (Imm.getBitWidth() <= 64 && Imm.isSignedIntN(N: 64)))
7147 return TTI::TCC_Free;
7148 break;
7149 case Intrinsic::experimental_patchpoint_void:
7150 case Intrinsic::experimental_patchpoint:
7151 if ((Idx < 4) || (Imm.getBitWidth() <= 64 && Imm.isSignedIntN(N: 64)))
7152 return TTI::TCC_Free;
7153 break;
7154 }
7155 return X86TTIImpl::getIntImmCost(Imm, Ty, CostKind);
7156}
7157
7158InstructionCost X86TTIImpl::getCFInstrCost(unsigned Opcode,
7159 TTI::TargetCostKind CostKind,
7160 const Instruction *I) const {
7161 if (CostKind != TTI::TCK_RecipThroughput)
7162 return Opcode == Instruction::PHI ? TTI::TCC_Free : TTI::TCC_Basic;
7163 // Branches are assumed to be predicted.
7164 return TTI::TCC_Free;
7165}
7166
7167int X86TTIImpl::getGatherOverhead() const {
7168 // Some CPUs have more overhead for gather. The specified overhead is relative
7169 // to the Load operation. "2" is the number provided by Intel architects. This
7170 // parameter is used for cost estimation of Gather Op and comparison with
7171 // other alternatives.
7172 // TODO: Remove the explicit hasAVX512()?, That would mean we would only
7173 // enable gather with a -march.
7174 if (ST->hasAVX512() || (ST->hasAVX2() && ST->hasFastGather()))
7175 return 2;
7176
7177 return 1024;
7178}
7179
7180int X86TTIImpl::getScatterOverhead() const {
7181 if (ST->hasAVX512())
7182 return 2;
7183
7184 return 1024;
7185}
7186
7187// Return an average cost of Gather / Scatter instruction, maybe improved later.
7188InstructionCost X86TTIImpl::getGSVectorCost(unsigned Opcode,
7189 TTI::TargetCostKind CostKind,
7190 Type *SrcVTy, const Value *Ptr,
7191 Align Alignment,
7192 unsigned AddressSpace) const {
7193
7194 assert(isa<VectorType>(SrcVTy) && "Unexpected type in getGSVectorCost");
7195 unsigned VF = cast<FixedVectorType>(Val: SrcVTy)->getNumElements();
7196
7197 // Try to reduce index size from 64 bit (default for GEP)
7198 // to 32. It is essential for VF 16. If the index can't be reduced to 32, the
7199 // operation will use 16 x 64 indices which do not fit in a zmm and needs
7200 // to split. Also check that the base pointer is the same for all lanes,
7201 // and that there's at most one variable index.
7202 auto getIndexSizeInBits = [](const Value *Ptr, const DataLayout &DL) {
7203 unsigned IndexSize = DL.getPointerSizeInBits();
7204 const GetElementPtrInst *GEP = dyn_cast_or_null<GetElementPtrInst>(Val: Ptr);
7205 if (IndexSize < 64 || !GEP)
7206 return IndexSize;
7207
7208 unsigned NumOfVarIndices = 0;
7209 const Value *Ptrs = GEP->getPointerOperand();
7210 if (Ptrs->getType()->isVectorTy() && !getSplatValue(V: Ptrs))
7211 return IndexSize;
7212 for (unsigned I = 1, E = GEP->getNumOperands(); I != E; ++I) {
7213 if (isa<Constant>(Val: GEP->getOperand(i_nocapture: I)))
7214 continue;
7215 Type *IndxTy = GEP->getOperand(i_nocapture: I)->getType();
7216 if (auto *IndexVTy = dyn_cast<VectorType>(Val: IndxTy))
7217 IndxTy = IndexVTy->getElementType();
7218 if ((IndxTy->getPrimitiveSizeInBits() == 64 &&
7219 !isa<SExtInst>(Val: GEP->getOperand(i_nocapture: I))) ||
7220 ++NumOfVarIndices > 1)
7221 return IndexSize; // 64
7222 }
7223 return (unsigned)32;
7224 };
7225
7226 // Trying to reduce IndexSize to 32 bits for vector 16.
7227 // By default the IndexSize is equal to pointer size.
7228 unsigned IndexSize = (ST->hasAVX512() && VF >= 16)
7229 ? getIndexSizeInBits(Ptr, DL)
7230 : DL.getPointerSizeInBits();
7231
7232 auto *IndexVTy = FixedVectorType::get(
7233 ElementType: IntegerType::get(C&: SrcVTy->getContext(), NumBits: IndexSize), NumElts: VF);
7234 std::pair<InstructionCost, MVT> IdxsLT = getTypeLegalizationCost(Ty: IndexVTy);
7235 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Ty: SrcVTy);
7236 InstructionCost::CostType SplitFactor =
7237 std::max(a: IdxsLT.first, b: SrcLT.first).getValue();
7238 if (SplitFactor > 1) {
7239 // Handle splitting of vector of pointers
7240 auto *SplitSrcTy =
7241 FixedVectorType::get(ElementType: SrcVTy->getScalarType(), NumElts: VF / SplitFactor);
7242 return SplitFactor * getGSVectorCost(Opcode, CostKind, SrcVTy: SplitSrcTy, Ptr,
7243 Alignment, AddressSpace);
7244 }
7245
7246 // If we didn't split, this will be a single gather/scatter instruction.
7247 if (CostKind == TTI::TCK_CodeSize)
7248 return 1;
7249
7250 // The gather / scatter cost is given by Intel architects. It is a rough
7251 // number since we are looking at one instruction in a time.
7252 const int GSOverhead = (Opcode == Instruction::Load) ? getGatherOverhead()
7253 : getScatterOverhead();
7254 return GSOverhead + VF * getMemoryOpCost(Opcode, Src: SrcVTy->getScalarType(),
7255 Alignment, AddressSpace, CostKind);
7256}
7257
7258/// Calculate the cost of Gather / Scatter operation
7259InstructionCost
7260X86TTIImpl::getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA,
7261 TTI::TargetCostKind CostKind) const {
7262 bool IsLoad = MICA.getID() == Intrinsic::masked_gather ||
7263 MICA.getID() == Intrinsic::vp_gather;
7264 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
7265 Type *SrcVTy = MICA.getDataType();
7266 const Value *Ptr = MICA.getPointer();
7267 Align Alignment = MICA.getAlignment();
7268 if ((Opcode == Instruction::Load &&
7269 (!isLegalMaskedGather(DataType: SrcVTy, Alignment: Align(Alignment)) ||
7270 forceScalarizeMaskedGather(VTy: cast<VectorType>(Val: SrcVTy),
7271 Alignment: Align(Alignment)))) ||
7272 (Opcode == Instruction::Store &&
7273 (!isLegalMaskedScatter(DataType: SrcVTy, Alignment: Align(Alignment)) ||
7274 forceScalarizeMaskedScatter(VTy: cast<VectorType>(Val: SrcVTy),
7275 Alignment: Align(Alignment)))))
7276 return BaseT::getMemIntrinsicInstrCost(MICA, CostKind);
7277
7278 assert(SrcVTy->isVectorTy() && "Unexpected data type for Gather/Scatter");
7279 unsigned AddressSpace = MICA.getAddressSpace();
7280 InstructionCost Cost =
7281 getGSVectorCost(Opcode, CostKind, SrcVTy, Ptr, Alignment, AddressSpace);
7282
7283 // Predicate fanout (see getPredicateFanoutCost); variable mask only.
7284 if (MICA.getVariableMask()) {
7285 auto *MaskVecTy =
7286 FixedVectorType::get(ElementType: Type::getInt1Ty(C&: SrcVTy->getContext()),
7287 NumElts: cast<FixedVectorType>(Val: SrcVTy)->getNumElements());
7288 Cost += getPredicateFanoutCost(TTI: *this, DataVTy: SrcVTy, MaskVTy: MaskVecTy);
7289 }
7290 return Cost;
7291}
7292
7293bool X86TTIImpl::isLSRCostLess(const TargetTransformInfo::LSRCost &C1,
7294 const TargetTransformInfo::LSRCost &C2) const {
7295 // X86 specific here are "instruction number 1st priority".
7296 return std::tie(args: C1.Insns, args: C1.NumRegs, args: C1.AddRecCost, args: C1.NumIVMuls,
7297 args: C1.NumBaseAdds, args: C1.ScaleCost, args: C1.ImmCost, args: C1.SetupCost) <
7298 std::tie(args: C2.Insns, args: C2.NumRegs, args: C2.AddRecCost, args: C2.NumIVMuls,
7299 args: C2.NumBaseAdds, args: C2.ScaleCost, args: C2.ImmCost, args: C2.SetupCost);
7300}
7301
7302bool X86TTIImpl::canMacroFuseCmp() const {
7303 return ST->hasMacroFusion() || ST->hasBranchFusion();
7304}
7305
7306static bool isLegalMaskedLoadStore(Type *ScalarTy, const X86Subtarget *ST) {
7307 if (!ST->hasAVX())
7308 return false;
7309
7310 if (ScalarTy->isPointerTy())
7311 return true;
7312
7313 if (ScalarTy->isFloatTy() || ScalarTy->isDoubleTy())
7314 return true;
7315
7316 if (ScalarTy->isHalfTy() && ST->hasBWI())
7317 return true;
7318
7319 if (ScalarTy->isBFloatTy() && ST->hasBF16())
7320 return true;
7321
7322 if (!ScalarTy->isIntegerTy())
7323 return false;
7324
7325 unsigned IntWidth = ScalarTy->getIntegerBitWidth();
7326 return IntWidth == 32 || IntWidth == 64 ||
7327 ((IntWidth == 8 || IntWidth == 16) && ST->hasBWI());
7328}
7329
7330bool X86TTIImpl::isLegalMaskedLoad(Type *DataTy, Align Alignment,
7331 unsigned AddressSpace,
7332 TTI::MaskKind MaskKind) const {
7333 Type *ScalarTy = DataTy->getScalarType();
7334
7335 // The backend can't handle a single element vector w/o CFCMOV.
7336 if (isa<VectorType>(Val: DataTy) &&
7337 cast<FixedVectorType>(Val: DataTy)->getNumElements() == 1)
7338 return ST->hasCF() &&
7339 hasConditionalLoadStoreForType(Ty: ScalarTy, /*IsStore=*/false);
7340
7341 return isLegalMaskedLoadStore(ScalarTy, ST);
7342}
7343
7344bool X86TTIImpl::isLegalMaskedStore(Type *DataTy, Align Alignment,
7345 unsigned AddressSpace,
7346 TTI::MaskKind MaskKind) const {
7347 Type *ScalarTy = DataTy->getScalarType();
7348
7349 // The backend can't handle a single element vector w/o CFCMOV.
7350 if (isa<VectorType>(Val: DataTy) &&
7351 cast<FixedVectorType>(Val: DataTy)->getNumElements() == 1)
7352 return ST->hasCF() &&
7353 hasConditionalLoadStoreForType(Ty: ScalarTy, /*IsStore=*/true);
7354
7355 return isLegalMaskedLoadStore(ScalarTy, ST);
7356}
7357
7358bool X86TTIImpl::isLegalNTLoad(Type *DataType, Align Alignment) const {
7359 unsigned DataSize = DL.getTypeStoreSize(Ty: DataType);
7360 // The only supported nontemporal loads are for aligned vectors of 16 or 32
7361 // bytes. Note that 32-byte nontemporal vector loads are supported by AVX2
7362 // (the equivalent stores only require AVX).
7363 if (Alignment >= DataSize && (DataSize == 16 || DataSize == 32))
7364 return DataSize == 16 ? ST->hasSSE1() : ST->hasAVX2();
7365
7366 return false;
7367}
7368
7369bool X86TTIImpl::isLegalNTStore(Type *DataType, Align Alignment) const {
7370 unsigned DataSize = DL.getTypeStoreSize(Ty: DataType);
7371
7372 // SSE4A supports nontemporal stores of float and double at arbitrary
7373 // alignment.
7374 if (ST->hasSSE4A() && (DataType->isFloatTy() || DataType->isDoubleTy()))
7375 return true;
7376
7377 // Besides the SSE4A subtarget exception above, only aligned stores are
7378 // available nontemporaly on any other subtarget. And only stores with a size
7379 // of 4..32 bytes (powers of 2, only) are permitted.
7380 if (Alignment < DataSize || DataSize < 4 || DataSize > 32 ||
7381 !isPowerOf2_32(Value: DataSize))
7382 return false;
7383
7384 // 32-byte vector nontemporal stores are supported by AVX (the equivalent
7385 // loads require AVX2).
7386 if (DataSize == 32)
7387 return ST->hasAVX();
7388 if (DataSize == 16)
7389 return ST->hasSSE1();
7390 return true;
7391}
7392
7393bool X86TTIImpl::isLegalBroadcastLoad(Type *ElementTy,
7394 ElementCount NumElements) const {
7395 // movddup
7396 return ST->hasSSE3() && !NumElements.isScalable() &&
7397 NumElements.getFixedValue() == 2 &&
7398 ElementTy == Type::getDoubleTy(C&: ElementTy->getContext());
7399}
7400
7401bool X86TTIImpl::isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const {
7402 if (!isa<VectorType>(Val: DataTy))
7403 return false;
7404
7405 if (!ST->hasAVX512())
7406 return false;
7407
7408 // The backend can't handle a single element vector.
7409 if (cast<FixedVectorType>(Val: DataTy)->getNumElements() == 1)
7410 return false;
7411
7412 Type *ScalarTy = cast<VectorType>(Val: DataTy)->getElementType();
7413
7414 if (ScalarTy->isFloatTy() || ScalarTy->isDoubleTy())
7415 return true;
7416
7417 if (!ScalarTy->isIntegerTy())
7418 return false;
7419
7420 unsigned IntWidth = ScalarTy->getIntegerBitWidth();
7421 return IntWidth == 32 || IntWidth == 64 ||
7422 ((IntWidth == 8 || IntWidth == 16) && ST->hasVBMI2());
7423}
7424
7425bool X86TTIImpl::isLegalMaskedCompressStore(Type *DataTy,
7426 Align Alignment) const {
7427 return isLegalMaskedExpandLoad(DataTy, Alignment);
7428}
7429
7430bool X86TTIImpl::supportsGather() const {
7431 // Some CPUs have better gather performance than others.
7432 // TODO: Remove the explicit ST->hasAVX512()?, That would mean we would only
7433 // enable gather with a -march.
7434 return ST->hasAVX512() || (ST->hasFastGather() && ST->hasAVX2());
7435}
7436
7437bool X86TTIImpl::forceScalarizeMaskedGather(VectorType *VTy,
7438 Align Alignment) const {
7439 // Gather / Scatter for vector 2 is not profitable on KNL / SKX
7440 // Vector-4 of gather/scatter instruction does not exist on KNL. We can extend
7441 // it to 8 elements, but zeroing upper bits of the mask vector will add more
7442 // instructions. Right now we give the scalar cost of vector-4 for KNL. TODO:
7443 // Check, maybe the gather/scatter instruction is better in the VariableMask
7444 // case.
7445 unsigned NumElts = cast<FixedVectorType>(Val: VTy)->getNumElements();
7446 return NumElts == 1 ||
7447 (ST->hasAVX512() && (NumElts == 2 || (NumElts == 4 && !ST->hasVLX())));
7448}
7449
7450bool X86TTIImpl::isLegalMaskedGatherScatter(Type *DataTy,
7451 Align Alignment) const {
7452 Type *ScalarTy = DataTy->getScalarType();
7453 if (ScalarTy->isPointerTy())
7454 return true;
7455
7456 if (ScalarTy->isFloatTy() || ScalarTy->isDoubleTy())
7457 return true;
7458
7459 if (!ScalarTy->isIntegerTy())
7460 return false;
7461
7462 unsigned IntWidth = ScalarTy->getIntegerBitWidth();
7463 return IntWidth == 32 || IntWidth == 64;
7464}
7465
7466bool X86TTIImpl::isLegalMaskedGather(Type *DataTy, Align Alignment) const {
7467 if (!supportsGather() || !ST->preferGather())
7468 return false;
7469 return isLegalMaskedGatherScatter(DataTy, Alignment);
7470}
7471
7472bool X86TTIImpl::isLegalAltInstr(VectorType *VecTy, unsigned Opcode0,
7473 unsigned Opcode1,
7474 const SmallBitVector &OpcodeMask,
7475 ArrayRef<const Value *> Scalars) const {
7476 // ADDSUBPS 4xf32 SSE3
7477 // VADDSUBPS 4xf32 AVX
7478 // VADDSUBPS 8xf32 AVX2
7479 // ADDSUBPD 2xf64 SSE3
7480 // VADDSUBPD 2xf64 AVX
7481 // VADDSUBPD 4xf64 AVX2
7482
7483 unsigned NumElements = cast<FixedVectorType>(Val: VecTy)->getNumElements();
7484 assert(OpcodeMask.size() == NumElements && "Mask and VecTy are incompatible");
7485 if (!isPowerOf2_32(Value: NumElements))
7486 return false;
7487 // Check the opcode pattern. We apply the mask on the opcode arguments and
7488 // then check if it is what we expect.
7489 auto IsLaneOrder = [&](unsigned EvenOpc, unsigned OddOpc) {
7490 return all_of(Range: seq<unsigned>(Begin: 0, End: NumElements), P: [&](unsigned Lane) {
7491 return (OpcodeMask.test(Idx: Lane) ? Opcode1 : Opcode0) ==
7492 (Lane % 2 == 0 ? EvenOpc : OddOpc);
7493 });
7494 };
7495 // We expect FSub for even lanes and FAdd for odd lanes.
7496 const bool IsAddSub = IsLaneOrder(Instruction::FSub, Instruction::FAdd);
7497 // Now check that the pattern is supported by the target ISA.
7498 Type *ElemTy = cast<VectorType>(Val: VecTy)->getElementType();
7499 if (IsAddSub && ST->hasSSE3() &&
7500 ((ElemTy->isFloatTy() && NumElements % 4 == 0) ||
7501 (ElemTy->isDoubleTy() && NumElements % 2 == 0)))
7502 return true;
7503 // The multiplication is fused into every lane: fmaddsub (the even lanes
7504 // subtract and the odd lanes add the product) or fmsubadd (the other way
7505 // round). The subtracted product is the first operand of the subtraction.
7506 using namespace PatternMatch;
7507 auto IsFusedWithFMul = [](const Value *V) {
7508 const auto *I = dyn_cast<Instruction>(Val: V);
7509 if (!I || !I->hasAllowContract())
7510 return false;
7511 auto IsFMul = [&](const Value *Op) {
7512 return match(V: Op,
7513 P: m_OneUse(SubPattern: m_AllowContract(SubPattern: m_FMul(L: m_Value(), R: m_Value())))) &&
7514 cast<Instruction>(Val: Op)->getParent() == I->getParent();
7515 };
7516 return IsFMul(I->getOperand(i: 0)) ||
7517 (I->getOpcode() == Instruction::FAdd && IsFMul(I->getOperand(i: 1)));
7518 };
7519 return ST->hasFMA() && !Scalars.empty() &&
7520 (ElemTy->isFloatTy() || ElemTy->isDoubleTy()) &&
7521 (IsAddSub || IsLaneOrder(Instruction::FAdd, Instruction::FSub)) &&
7522 all_of(Range&: Scalars, P: IsFusedWithFMul);
7523}
7524
7525bool X86TTIImpl::isLegalMaskedScatter(Type *DataType, Align Alignment) const {
7526 // AVX2 doesn't support scatter
7527 if (!ST->hasAVX512() || !ST->preferScatter())
7528 return false;
7529 return isLegalMaskedGatherScatter(DataTy: DataType, Alignment);
7530}
7531
7532bool X86TTIImpl::hasDivRemOp(Type *DataType, bool IsSigned) const {
7533 EVT VT = TLI->getValueType(DL, Ty: DataType);
7534 return TLI->isOperationLegal(Op: IsSigned ? ISD::SDIVREM : ISD::UDIVREM, VT);
7535}
7536
7537bool X86TTIImpl::isExpensiveToSpeculativelyExecute(const Instruction *I) const {
7538 // FDIV is always expensive, even if it has a very low uop count.
7539 // TODO: Still necessary for recent CPUs with low latency/throughput fdiv?
7540 if (I->getOpcode() == Instruction::FDiv)
7541 return true;
7542
7543 return BaseT::isExpensiveToSpeculativelyExecute(I);
7544}
7545
7546bool X86TTIImpl::isFCmpOrdCheaperThanFCmpZero(Type *Ty) const { return false; }
7547
7548bool X86TTIImpl::areInlineCompatible(const Function *Caller,
7549 const Function *Callee) const {
7550 const TargetMachine &TM = getTLI()->getTargetMachine();
7551
7552 // Work this as a subsetting of subtarget features.
7553 const X86Subtarget &CallerSubtarget = TM.getSubtarget<X86Subtarget>(F: *Caller);
7554 const X86Subtarget &CalleeSubtarget = TM.getSubtarget<X86Subtarget>(F: *Callee);
7555 const FeatureBitset &CallerBits = CallerSubtarget.getFeatureBits();
7556 const FeatureBitset &CalleeBits = CalleeSubtarget.getFeatureBits();
7557
7558 // Check whether callee features are a subset of caller features
7559 // (apart from the ignore list).
7560 const FeatureBitset &InlineIgnoreFeatures =
7561 CallerSubtarget.getInlineIgnoreFeatures();
7562 FeatureBitset RealCallerBits = CallerBits & ~InlineIgnoreFeatures;
7563 FeatureBitset RealCalleeBits = CalleeBits & ~InlineIgnoreFeatures;
7564 if ((RealCallerBits & RealCalleeBits) != RealCalleeBits)
7565 return false;
7566
7567 // If the features are not exactly the same (or there is a difference in
7568 // AVX512 register usage), we need to additionally check for calls
7569 // that may become ABI-incompatible as a result of inlining.
7570 if (RealCallerBits == RealCalleeBits &&
7571 CallerSubtarget.useAVX512Regs() == CalleeSubtarget.useAVX512Regs())
7572 return true;
7573
7574 for (const Instruction &I : instructions(F: Callee)) {
7575 if (const auto *CB = dyn_cast<CallBase>(Val: &I)) {
7576 // Having more target features is fine for inline ASM and intrinsics.
7577 if (CB->isInlineAsm() || CB->getIntrinsicID() != Intrinsic::not_intrinsic)
7578 continue;
7579
7580 SmallVector<Type *, 8> Types;
7581 for (Value *Arg : CB->args())
7582 Types.push_back(Elt: Arg->getType());
7583 if (!CB->getType()->isVoidTy())
7584 Types.push_back(Elt: CB->getType());
7585
7586 // Simple types are always ABI compatible.
7587 auto IsSimpleTy = [](Type *Ty) {
7588 return !Ty->isVectorTy() && !Ty->isAggregateType();
7589 };
7590 if (all_of(Range&: Types, P: IsSimpleTy))
7591 continue;
7592
7593 // Do a precise compatibility check.
7594 if (!areTypesABICompatible(Caller, Callee, Type: Types))
7595 return false;
7596 }
7597 }
7598 return true;
7599}
7600
7601bool X86TTIImpl::areTypesABICompatible(const Function *Caller,
7602 const Function *Callee,
7603 ArrayRef<Type *> Types) const {
7604 const TargetMachine &TM = getTLI()->getTargetMachine();
7605 const TargetLowering *CallerTLI =
7606 TM.getSubtargetImpl(*Caller)->getTargetLowering();
7607 const TargetLowering *CalleeTLI =
7608 TM.getSubtargetImpl(*Callee)->getTargetLowering();
7609
7610 LLVMContext &Ctx = Caller->getContext();
7611 const DataLayout &DL = Caller->getDataLayout();
7612 CallingConv::ID CC = Callee->getCallingConv();
7613 return all_of(Range&: Types, P: [&](Type *Ty) {
7614 SmallVector<EVT> VTs;
7615 ComputeValueVTs(TLI: *CallerTLI, DL, Ty, ValueVTs&: VTs);
7616 return all_of(Range&: VTs, P: [&](EVT VT) {
7617 return CallerTLI->getRegisterTypeForCallingConv(Context&: Ctx, CC, VT) ==
7618 CalleeTLI->getRegisterTypeForCallingConv(Context&: Ctx, CC, VT);
7619 });
7620 });
7621}
7622
7623X86TTIImpl::TTI::MemCmpExpansionOptions
7624X86TTIImpl::enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
7625 TTI::MemCmpExpansionOptions Options;
7626 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
7627 Options.NumLoadsPerBlock = IsZeroCmp ? 2 : 1;
7628 // All GPR and vector loads can be unaligned.
7629 Options.AllowOverlappingLoads = true;
7630 if (IsZeroCmp) {
7631 // Only enable vector loads for equality comparison. Right now the vector
7632 // version is not as fast for three way compare (see #33329).
7633 const unsigned PreferredWidth = ST->getPreferVectorWidth();
7634 if (PreferredWidth >= 512 && ST->hasAVX512())
7635 Options.LoadSizes.push_back(Elt: 64);
7636 if (PreferredWidth >= 256 && ST->hasAVX()) Options.LoadSizes.push_back(Elt: 32);
7637 if (PreferredWidth >= 128 && ST->hasSSE2()) Options.LoadSizes.push_back(Elt: 16);
7638 }
7639 if (ST->is64Bit()) {
7640 Options.LoadSizes.push_back(Elt: 8);
7641 }
7642 Options.LoadSizes.push_back(Elt: 4);
7643 Options.LoadSizes.push_back(Elt: 2);
7644 Options.LoadSizes.push_back(Elt: 1);
7645 return Options;
7646}
7647
7648bool X86TTIImpl::prefersVectorizedAddressing() const {
7649 return supportsGather();
7650}
7651
7652bool X86TTIImpl::supportsEfficientVectorElementLoadStore() const {
7653 return false;
7654}
7655
7656bool X86TTIImpl::enableInterleavedAccessVectorization() const {
7657 // TODO: We expect this to be beneficial regardless of arch,
7658 // but there are currently some unexplained performance artifacts on Atom.
7659 // As a temporary solution, disable on Atom.
7660 return !(ST->isAtom());
7661}
7662
7663bool X86TTIImpl::shouldExpandReduction(const IntrinsicInst *II) const {
7664 switch (II->getIntrinsicID()) {
7665 default:
7666 return true;
7667 case Intrinsic::vector_reduce_and:
7668 case Intrinsic::vector_reduce_or:
7669 case Intrinsic::vector_reduce_xor:
7670 case Intrinsic::vector_reduce_mul:
7671 case Intrinsic::vector_reduce_smax:
7672 case Intrinsic::vector_reduce_smin:
7673 case Intrinsic::vector_reduce_umax:
7674 case Intrinsic::vector_reduce_umin:
7675 return false;
7676 }
7677}
7678
7679// Get estimation for interleaved load/store operations and strided load.
7680// \p Indices contains indices for strided load.
7681// \p Factor - the factor of interleaving.
7682// AVX-512 provides 3-src shuffles that significantly reduces the cost.
7683InstructionCost X86TTIImpl::getInterleavedMemoryOpCostAVX512(
7684 unsigned Opcode, FixedVectorType *VecTy, unsigned Factor,
7685 ArrayRef<unsigned> Indices, Align Alignment, unsigned AddressSpace,
7686 TTI::TargetCostKind CostKind, bool UseMaskForCond,
7687 bool UseMaskForGaps) const {
7688 // VecTy for interleave memop is <VF*Factor x Elt>.
7689 // So, for VF=4, Interleave Factor = 3, Element type = i32 we have
7690 // VecTy = <12 x i32>.
7691
7692 // Calculate the number of memory operations (NumOfMemOps), required
7693 // for load/store the VecTy.
7694 MVT LegalVT = getTypeLegalizationCost(Ty: VecTy).second;
7695 unsigned VecTySize = DL.getTypeStoreSize(Ty: VecTy);
7696 unsigned LegalVTSize = LegalVT.getStoreSize();
7697 unsigned NumOfMemOps = (VecTySize + LegalVTSize - 1) / LegalVTSize;
7698
7699 // Get the cost of one memory operation.
7700 auto *SingleMemOpTy = FixedVectorType::get(ElementType: VecTy->getElementType(),
7701 NumElts: LegalVT.getVectorNumElements());
7702 InstructionCost MemOpCost;
7703 bool UseMaskedMemOp = UseMaskForCond || UseMaskForGaps;
7704 if (UseMaskedMemOp) {
7705 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
7706 : Intrinsic::masked_store;
7707 MemOpCost = getMaskedMemoryOpCost(
7708 MICA: {IID, SingleMemOpTy, Alignment, AddressSpace}, CostKind);
7709 } else
7710 MemOpCost = getMemoryOpCost(Opcode, Src: SingleMemOpTy, Alignment, AddressSpace,
7711 CostKind);
7712
7713 unsigned VF = VecTy->getNumElements() / Factor;
7714 MVT VT =
7715 MVT::getVectorVT(VT: TLI->getSimpleValueType(DL, Ty: VecTy->getScalarType()), NumElements: VF);
7716
7717 InstructionCost MaskCost;
7718 if (UseMaskedMemOp) {
7719 APInt DemandedLoadStoreElts = APInt::getZero(numBits: VecTy->getNumElements());
7720 for (unsigned Index : Indices) {
7721 assert(Index < Factor && "Invalid index for interleaved memory op");
7722 for (unsigned Elm = 0; Elm < VF; Elm++)
7723 DemandedLoadStoreElts.setBit(Index + Elm * Factor);
7724 }
7725
7726 Type *I1Type = Type::getInt1Ty(C&: VecTy->getContext());
7727
7728 MaskCost = getReplicationShuffleCost(
7729 EltTy: I1Type, ReplicationFactor: Factor, VF,
7730 DemandedDstElts: UseMaskForGaps ? DemandedLoadStoreElts
7731 : APInt::getAllOnes(numBits: VecTy->getNumElements()),
7732 CostKind);
7733
7734 // The Gaps mask is invariant and created outside the loop, therefore the
7735 // cost of creating it is not accounted for here. However if we have both
7736 // a MaskForGaps and some other mask that guards the execution of the
7737 // memory access, we need to account for the cost of And-ing the two masks
7738 // inside the loop.
7739 if (UseMaskForGaps) {
7740 auto *MaskVT = FixedVectorType::get(ElementType: I1Type, NumElts: VecTy->getNumElements());
7741 MaskCost += getArithmeticInstrCost(Opcode: BinaryOperator::And, Ty: MaskVT, CostKind);
7742 }
7743 }
7744
7745 if (Opcode == Instruction::Load) {
7746 // The tables (AVX512InterleavedLoadTbl and AVX512InterleavedStoreTbl)
7747 // contain the cost of the optimized shuffle sequence that the
7748 // X86InterleavedAccess pass will generate.
7749 // The cost of loads and stores are computed separately from the table.
7750
7751 // X86InterleavedAccess support only the following interleaved-access group.
7752 static const CostTblEntry AVX512InterleavedLoadTbl[] = {
7753 {.ISD: 3, .Type: MVT::v16i8, .Cost: 12}, //(load 48i8 and) deinterleave into 3 x 16i8
7754 {.ISD: 3, .Type: MVT::v32i8, .Cost: 14}, //(load 96i8 and) deinterleave into 3 x 32i8
7755 {.ISD: 3, .Type: MVT::v64i8, .Cost: 22}, //(load 96i8 and) deinterleave into 3 x 32i8
7756 };
7757
7758 if (const auto *Entry =
7759 CostTableLookup(Table: AVX512InterleavedLoadTbl, ISD: Factor, Ty: VT))
7760 return MaskCost + NumOfMemOps * MemOpCost + Entry->Cost;
7761 //If an entry does not exist, fallback to the default implementation.
7762
7763 // Kind of shuffle depends on number of loaded values.
7764 // If we load the entire data in one register, we can use a 1-src shuffle.
7765 // Otherwise, we'll merge 2 sources in each operation.
7766 TTI::ShuffleKind ShuffleKind =
7767 (NumOfMemOps > 1) ? TTI::SK_PermuteTwoSrc : TTI::SK_PermuteSingleSrc;
7768
7769 InstructionCost ShuffleCost = getShuffleCost(
7770 Kind: ShuffleKind, DstTy: SingleMemOpTy, SrcTy: SingleMemOpTy, CostKind, Mask: {}, Index: 0, SubTp: nullptr);
7771
7772 unsigned NumOfLoadsInInterleaveGrp =
7773 Indices.size() ? Indices.size() : Factor;
7774 auto *ResultTy = FixedVectorType::get(ElementType: VecTy->getElementType(),
7775 NumElts: VecTy->getNumElements() / Factor);
7776 InstructionCost NumOfResults =
7777 getTypeLegalizationCost(Ty: ResultTy).first * NumOfLoadsInInterleaveGrp;
7778
7779 // About a half of the loads may be folded in shuffles when we have only
7780 // one result. If we have more than one result, or the loads are masked,
7781 // we do not fold loads at all.
7782 unsigned NumOfUnfoldedLoads =
7783 UseMaskedMemOp || NumOfResults > 1 ? NumOfMemOps : NumOfMemOps / 2;
7784
7785 // Get a number of shuffle operations per result.
7786 unsigned NumOfShufflesPerResult =
7787 std::max(a: (unsigned)1, b: (unsigned)(NumOfMemOps - 1));
7788
7789 // The SK_MergeTwoSrc shuffle clobbers one of src operands.
7790 // When we have more than one destination, we need additional instructions
7791 // to keep sources.
7792 InstructionCost NumOfMoves = 0;
7793 if (NumOfResults > 1 && ShuffleKind == TTI::SK_PermuteTwoSrc)
7794 NumOfMoves = NumOfResults * NumOfShufflesPerResult / 2;
7795
7796 InstructionCost Cost = NumOfResults * NumOfShufflesPerResult * ShuffleCost +
7797 MaskCost + NumOfUnfoldedLoads * MemOpCost +
7798 NumOfMoves;
7799
7800 return Cost;
7801 }
7802
7803 // Store.
7804 assert(Opcode == Instruction::Store &&
7805 "Expected Store Instruction at this point");
7806 // X86InterleavedAccess support only the following interleaved-access group.
7807 static const CostTblEntry AVX512InterleavedStoreTbl[] = {
7808 {.ISD: 3, .Type: MVT::v16i8, .Cost: 12}, // interleave 3 x 16i8 into 48i8 (and store)
7809 {.ISD: 3, .Type: MVT::v32i8, .Cost: 14}, // interleave 3 x 32i8 into 96i8 (and store)
7810 {.ISD: 3, .Type: MVT::v64i8, .Cost: 26}, // interleave 3 x 64i8 into 96i8 (and store)
7811
7812 {.ISD: 4, .Type: MVT::v8i8, .Cost: 10}, // interleave 4 x 8i8 into 32i8 (and store)
7813 {.ISD: 4, .Type: MVT::v16i8, .Cost: 11}, // interleave 4 x 16i8 into 64i8 (and store)
7814 {.ISD: 4, .Type: MVT::v32i8, .Cost: 14}, // interleave 4 x 32i8 into 128i8 (and store)
7815 {.ISD: 4, .Type: MVT::v64i8, .Cost: 24} // interleave 4 x 32i8 into 256i8 (and store)
7816 };
7817
7818 if (const auto *Entry =
7819 CostTableLookup(Table: AVX512InterleavedStoreTbl, ISD: Factor, Ty: VT))
7820 return MaskCost + NumOfMemOps * MemOpCost + Entry->Cost;
7821 //If an entry does not exist, fallback to the default implementation.
7822
7823 // There is no strided stores meanwhile. And store can't be folded in
7824 // shuffle.
7825 unsigned NumOfSources = Factor; // The number of values to be merged.
7826 InstructionCost ShuffleCost =
7827 getShuffleCost(Kind: TTI::SK_PermuteTwoSrc, DstTy: SingleMemOpTy, SrcTy: SingleMemOpTy,
7828 CostKind, Mask: {}, Index: 0, SubTp: nullptr);
7829 unsigned NumOfShufflesPerStore = NumOfSources - 1;
7830
7831 // The SK_MergeTwoSrc shuffle clobbers one of src operands.
7832 // We need additional instructions to keep sources.
7833 unsigned NumOfMoves = NumOfMemOps * NumOfShufflesPerStore / 2;
7834 InstructionCost Cost =
7835 MaskCost +
7836 NumOfMemOps * (MemOpCost + NumOfShufflesPerStore * ShuffleCost) +
7837 NumOfMoves;
7838 return Cost;
7839}
7840
7841InstructionCost X86TTIImpl::getInterleavedMemoryOpCost(
7842 unsigned Opcode, Type *BaseTy, unsigned Factor, ArrayRef<unsigned> Indices,
7843 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
7844 bool UseMaskForCond, bool UseMaskForGaps) const {
7845 auto *VecTy = cast<FixedVectorType>(Val: BaseTy);
7846
7847 auto isSupportedOnAVX512 = [&](Type *VecTy) {
7848 Type *EltTy = cast<VectorType>(Val: VecTy)->getElementType();
7849 if (EltTy->isFloatTy() || EltTy->isDoubleTy() || EltTy->isIntegerTy(BitWidth: 64) ||
7850 EltTy->isIntegerTy(BitWidth: 32) || EltTy->isPointerTy())
7851 return true;
7852 if (EltTy->isIntegerTy(BitWidth: 16) || EltTy->isIntegerTy(BitWidth: 8) || EltTy->isHalfTy())
7853 return ST->hasBWI();
7854 if (EltTy->isBFloatTy())
7855 return ST->hasBF16();
7856 return false;
7857 };
7858 if (ST->hasAVX512() && isSupportedOnAVX512(VecTy))
7859 return getInterleavedMemoryOpCostAVX512(
7860 Opcode, VecTy, Factor, Indices, Alignment,
7861 AddressSpace, CostKind, UseMaskForCond, UseMaskForGaps);
7862
7863 if (UseMaskForCond || UseMaskForGaps)
7864 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
7865 Alignment, AddressSpace, CostKind,
7866 UseMaskForCond, UseMaskForGaps);
7867
7868 // Get estimation for interleaved load/store operations for SSE-AVX2.
7869 // As opposed to AVX-512, SSE-AVX2 do not have generic shuffles that allow
7870 // computing the cost using a generic formula as a function of generic
7871 // shuffles. We therefore use a lookup table instead, filled according to
7872 // the instruction sequences that codegen currently generates.
7873
7874 // VecTy for interleave memop is <VF*Factor x Elt>.
7875 // So, for VF=4, Interleave Factor = 3, Element type = i32 we have
7876 // VecTy = <12 x i32>.
7877 MVT LegalVT = getTypeLegalizationCost(Ty: VecTy).second;
7878
7879 // This function can be called with VecTy=<6xi128>, Factor=3, in which case
7880 // the VF=2, while v2i128 is an unsupported MVT vector type
7881 // (see MachineValueType.h::getVectorVT()).
7882 if (!LegalVT.isVector())
7883 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
7884 Alignment, AddressSpace, CostKind);
7885
7886 unsigned VF = VecTy->getNumElements() / Factor;
7887 Type *ScalarTy = VecTy->getElementType();
7888 // Deduplicate entries, model floats/pointers as appropriately-sized integers.
7889 if (!ScalarTy->isIntegerTy())
7890 ScalarTy =
7891 Type::getIntNTy(C&: ScalarTy->getContext(), N: DL.getTypeSizeInBits(Ty: ScalarTy));
7892
7893 // Get the cost of all the memory operations.
7894 // FIXME: discount dead loads.
7895 InstructionCost MemOpCosts =
7896 getMemoryOpCost(Opcode, Src: VecTy, Alignment, AddressSpace, CostKind);
7897
7898 auto *VT = FixedVectorType::get(ElementType: ScalarTy, NumElts: VF);
7899 EVT ETy = TLI->getValueType(DL, Ty: VT);
7900 if (!ETy.isSimple())
7901 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
7902 Alignment, AddressSpace, CostKind);
7903
7904 // TODO: Complete for other data-types and strides.
7905 // Each combination of Stride, element bit width and VF results in a different
7906 // sequence; The cost tables are therefore accessed with:
7907 // Factor (stride) and VectorType=VFxiN.
7908 // The Cost accounts only for the shuffle sequence;
7909 // The cost of the loads/stores is accounted for separately.
7910 //
7911 static const CostTblEntry AVX2InterleavedLoadTbl[] = {
7912 {.ISD: 2, .Type: MVT::v2i8, .Cost: 2}, // (load 4i8 and) deinterleave into 2 x 2i8
7913 {.ISD: 2, .Type: MVT::v4i8, .Cost: 2}, // (load 8i8 and) deinterleave into 2 x 4i8
7914 {.ISD: 2, .Type: MVT::v8i8, .Cost: 2}, // (load 16i8 and) deinterleave into 2 x 8i8
7915 {.ISD: 2, .Type: MVT::v16i8, .Cost: 4}, // (load 32i8 and) deinterleave into 2 x 16i8
7916 {.ISD: 2, .Type: MVT::v32i8, .Cost: 6}, // (load 64i8 and) deinterleave into 2 x 32i8
7917
7918 {.ISD: 2, .Type: MVT::v8i16, .Cost: 6}, // (load 16i16 and) deinterleave into 2 x 8i16
7919 {.ISD: 2, .Type: MVT::v16i16, .Cost: 9}, // (load 32i16 and) deinterleave into 2 x 16i16
7920 {.ISD: 2, .Type: MVT::v32i16, .Cost: 18}, // (load 64i16 and) deinterleave into 2 x 32i16
7921
7922 {.ISD: 2, .Type: MVT::v8i32, .Cost: 4}, // (load 16i32 and) deinterleave into 2 x 8i32
7923 {.ISD: 2, .Type: MVT::v16i32, .Cost: 8}, // (load 32i32 and) deinterleave into 2 x 16i32
7924 {.ISD: 2, .Type: MVT::v32i32, .Cost: 16}, // (load 64i32 and) deinterleave into 2 x 32i32
7925
7926 {.ISD: 2, .Type: MVT::v4i64, .Cost: 4}, // (load 8i64 and) deinterleave into 2 x 4i64
7927 {.ISD: 2, .Type: MVT::v8i64, .Cost: 8}, // (load 16i64 and) deinterleave into 2 x 8i64
7928 {.ISD: 2, .Type: MVT::v16i64, .Cost: 16}, // (load 32i64 and) deinterleave into 2 x 16i64
7929 {.ISD: 2, .Type: MVT::v32i64, .Cost: 32}, // (load 64i64 and) deinterleave into 2 x 32i64
7930
7931 {.ISD: 3, .Type: MVT::v2i8, .Cost: 3}, // (load 6i8 and) deinterleave into 3 x 2i8
7932 {.ISD: 3, .Type: MVT::v4i8, .Cost: 3}, // (load 12i8 and) deinterleave into 3 x 4i8
7933 {.ISD: 3, .Type: MVT::v8i8, .Cost: 6}, // (load 24i8 and) deinterleave into 3 x 8i8
7934 {.ISD: 3, .Type: MVT::v16i8, .Cost: 11}, // (load 48i8 and) deinterleave into 3 x 16i8
7935 {.ISD: 3, .Type: MVT::v32i8, .Cost: 14}, // (load 96i8 and) deinterleave into 3 x 32i8
7936
7937 {.ISD: 3, .Type: MVT::v2i16, .Cost: 5}, // (load 6i16 and) deinterleave into 3 x 2i16
7938 {.ISD: 3, .Type: MVT::v4i16, .Cost: 7}, // (load 12i16 and) deinterleave into 3 x 4i16
7939 {.ISD: 3, .Type: MVT::v8i16, .Cost: 9}, // (load 24i16 and) deinterleave into 3 x 8i16
7940 {.ISD: 3, .Type: MVT::v16i16, .Cost: 28}, // (load 48i16 and) deinterleave into 3 x 16i16
7941 {.ISD: 3, .Type: MVT::v32i16, .Cost: 56}, // (load 96i16 and) deinterleave into 3 x 32i16
7942
7943 {.ISD: 3, .Type: MVT::v2i32, .Cost: 3}, // (load 6i32 and) deinterleave into 3 x 2i32
7944 {.ISD: 3, .Type: MVT::v4i32, .Cost: 3}, // (load 12i32 and) deinterleave into 3 x 4i32
7945 {.ISD: 3, .Type: MVT::v8i32, .Cost: 7}, // (load 24i32 and) deinterleave into 3 x 8i32
7946 {.ISD: 3, .Type: MVT::v16i32, .Cost: 14}, // (load 48i32 and) deinterleave into 3 x 16i32
7947 {.ISD: 3, .Type: MVT::v32i32, .Cost: 32}, // (load 96i32 and) deinterleave into 3 x 32i32
7948
7949 {.ISD: 3, .Type: MVT::v2i64, .Cost: 1}, // (load 6i64 and) deinterleave into 3 x 2i64
7950 {.ISD: 3, .Type: MVT::v4i64, .Cost: 5}, // (load 12i64 and) deinterleave into 3 x 4i64
7951 {.ISD: 3, .Type: MVT::v8i64, .Cost: 10}, // (load 24i64 and) deinterleave into 3 x 8i64
7952 {.ISD: 3, .Type: MVT::v16i64, .Cost: 20}, // (load 48i64 and) deinterleave into 3 x 16i64
7953
7954 {.ISD: 4, .Type: MVT::v2i8, .Cost: 4}, // (load 8i8 and) deinterleave into 4 x 2i8
7955 {.ISD: 4, .Type: MVT::v4i8, .Cost: 4}, // (load 16i8 and) deinterleave into 4 x 4i8
7956 {.ISD: 4, .Type: MVT::v8i8, .Cost: 12}, // (load 32i8 and) deinterleave into 4 x 8i8
7957 {.ISD: 4, .Type: MVT::v16i8, .Cost: 24}, // (load 64i8 and) deinterleave into 4 x 16i8
7958 {.ISD: 4, .Type: MVT::v32i8, .Cost: 56}, // (load 128i8 and) deinterleave into 4 x 32i8
7959
7960 {.ISD: 4, .Type: MVT::v2i16, .Cost: 6}, // (load 8i16 and) deinterleave into 4 x 2i16
7961 {.ISD: 4, .Type: MVT::v4i16, .Cost: 17}, // (load 16i16 and) deinterleave into 4 x 4i16
7962 {.ISD: 4, .Type: MVT::v8i16, .Cost: 33}, // (load 32i16 and) deinterleave into 4 x 8i16
7963 {.ISD: 4, .Type: MVT::v16i16, .Cost: 75}, // (load 64i16 and) deinterleave into 4 x 16i16
7964 {.ISD: 4, .Type: MVT::v32i16, .Cost: 150}, // (load 128i16 and) deinterleave into 4 x 32i16
7965
7966 {.ISD: 4, .Type: MVT::v2i32, .Cost: 4}, // (load 8i32 and) deinterleave into 4 x 2i32
7967 {.ISD: 4, .Type: MVT::v4i32, .Cost: 8}, // (load 16i32 and) deinterleave into 4 x 4i32
7968 {.ISD: 4, .Type: MVT::v8i32, .Cost: 16}, // (load 32i32 and) deinterleave into 4 x 8i32
7969 {.ISD: 4, .Type: MVT::v16i32, .Cost: 32}, // (load 64i32 and) deinterleave into 4 x 16i32
7970 {.ISD: 4, .Type: MVT::v32i32, .Cost: 68}, // (load 128i32 and) deinterleave into 4 x 32i32
7971
7972 {.ISD: 4, .Type: MVT::v2i64, .Cost: 6}, // (load 8i64 and) deinterleave into 4 x 2i64
7973 {.ISD: 4, .Type: MVT::v4i64, .Cost: 8}, // (load 16i64 and) deinterleave into 4 x 4i64
7974 {.ISD: 4, .Type: MVT::v8i64, .Cost: 20}, // (load 32i64 and) deinterleave into 4 x 8i64
7975 {.ISD: 4, .Type: MVT::v16i64, .Cost: 40}, // (load 64i64 and) deinterleave into 4 x 16i64
7976
7977 {.ISD: 6, .Type: MVT::v2i8, .Cost: 6}, // (load 12i8 and) deinterleave into 6 x 2i8
7978 {.ISD: 6, .Type: MVT::v4i8, .Cost: 14}, // (load 24i8 and) deinterleave into 6 x 4i8
7979 {.ISD: 6, .Type: MVT::v8i8, .Cost: 18}, // (load 48i8 and) deinterleave into 6 x 8i8
7980 {.ISD: 6, .Type: MVT::v16i8, .Cost: 43}, // (load 96i8 and) deinterleave into 6 x 16i8
7981 {.ISD: 6, .Type: MVT::v32i8, .Cost: 82}, // (load 192i8 and) deinterleave into 6 x 32i8
7982
7983 {.ISD: 6, .Type: MVT::v2i16, .Cost: 13}, // (load 12i16 and) deinterleave into 6 x 2i16
7984 {.ISD: 6, .Type: MVT::v4i16, .Cost: 9}, // (load 24i16 and) deinterleave into 6 x 4i16
7985 {.ISD: 6, .Type: MVT::v8i16, .Cost: 39}, // (load 48i16 and) deinterleave into 6 x 8i16
7986 {.ISD: 6, .Type: MVT::v16i16, .Cost: 106}, // (load 96i16 and) deinterleave into 6 x 16i16
7987 {.ISD: 6, .Type: MVT::v32i16, .Cost: 212}, // (load 192i16 and) deinterleave into 6 x 32i16
7988
7989 {.ISD: 6, .Type: MVT::v2i32, .Cost: 6}, // (load 12i32 and) deinterleave into 6 x 2i32
7990 {.ISD: 6, .Type: MVT::v4i32, .Cost: 15}, // (load 24i32 and) deinterleave into 6 x 4i32
7991 {.ISD: 6, .Type: MVT::v8i32, .Cost: 31}, // (load 48i32 and) deinterleave into 6 x 8i32
7992 {.ISD: 6, .Type: MVT::v16i32, .Cost: 64}, // (load 96i32 and) deinterleave into 6 x 16i32
7993
7994 {.ISD: 6, .Type: MVT::v2i64, .Cost: 6}, // (load 12i64 and) deinterleave into 6 x 2i64
7995 {.ISD: 6, .Type: MVT::v4i64, .Cost: 18}, // (load 24i64 and) deinterleave into 6 x 4i64
7996 {.ISD: 6, .Type: MVT::v8i64, .Cost: 36}, // (load 48i64 and) deinterleave into 6 x 8i64
7997
7998 {.ISD: 8, .Type: MVT::v8i32, .Cost: 40} // (load 64i32 and) deinterleave into 8 x 8i32
7999 };
8000
8001 static const CostTblEntry SSSE3InterleavedLoadTbl[] = {
8002 {.ISD: 2, .Type: MVT::v4i16, .Cost: 2}, // (load 8i16 and) deinterleave into 2 x 4i16
8003 };
8004
8005 static const CostTblEntry SSE2InterleavedLoadTbl[] = {
8006 {.ISD: 2, .Type: MVT::v2i16, .Cost: 2}, // (load 4i16 and) deinterleave into 2 x 2i16
8007 {.ISD: 2, .Type: MVT::v4i16, .Cost: 7}, // (load 8i16 and) deinterleave into 2 x 4i16
8008
8009 {.ISD: 2, .Type: MVT::v2i32, .Cost: 2}, // (load 4i32 and) deinterleave into 2 x 2i32
8010 {.ISD: 2, .Type: MVT::v4i32, .Cost: 2}, // (load 8i32 and) deinterleave into 2 x 4i32
8011
8012 {.ISD: 2, .Type: MVT::v2i64, .Cost: 2}, // (load 4i64 and) deinterleave into 2 x 2i64
8013 };
8014
8015 static const CostTblEntry AVX2InterleavedStoreTbl[] = {
8016 {.ISD: 2, .Type: MVT::v16i8, .Cost: 3}, // interleave 2 x 16i8 into 32i8 (and store)
8017 {.ISD: 2, .Type: MVT::v32i8, .Cost: 4}, // interleave 2 x 32i8 into 64i8 (and store)
8018
8019 {.ISD: 2, .Type: MVT::v8i16, .Cost: 3}, // interleave 2 x 8i16 into 16i16 (and store)
8020 {.ISD: 2, .Type: MVT::v16i16, .Cost: 4}, // interleave 2 x 16i16 into 32i16 (and store)
8021 {.ISD: 2, .Type: MVT::v32i16, .Cost: 8}, // interleave 2 x 32i16 into 64i16 (and store)
8022
8023 {.ISD: 2, .Type: MVT::v4i32, .Cost: 2}, // interleave 2 x 4i32 into 8i32 (and store)
8024 {.ISD: 2, .Type: MVT::v8i32, .Cost: 4}, // interleave 2 x 8i32 into 16i32 (and store)
8025 {.ISD: 2, .Type: MVT::v16i32, .Cost: 8}, // interleave 2 x 16i32 into 32i32 (and store)
8026 {.ISD: 2, .Type: MVT::v32i32, .Cost: 16}, // interleave 2 x 32i32 into 64i32 (and store)
8027
8028 {.ISD: 2, .Type: MVT::v2i64, .Cost: 2}, // interleave 2 x 2i64 into 4i64 (and store)
8029 {.ISD: 2, .Type: MVT::v4i64, .Cost: 4}, // interleave 2 x 4i64 into 8i64 (and store)
8030 {.ISD: 2, .Type: MVT::v8i64, .Cost: 8}, // interleave 2 x 8i64 into 16i64 (and store)
8031 {.ISD: 2, .Type: MVT::v16i64, .Cost: 16}, // interleave 2 x 16i64 into 32i64 (and store)
8032 {.ISD: 2, .Type: MVT::v32i64, .Cost: 32}, // interleave 2 x 32i64 into 64i64 (and store)
8033
8034 {.ISD: 3, .Type: MVT::v2i8, .Cost: 4}, // interleave 3 x 2i8 into 6i8 (and store)
8035 {.ISD: 3, .Type: MVT::v4i8, .Cost: 4}, // interleave 3 x 4i8 into 12i8 (and store)
8036 {.ISD: 3, .Type: MVT::v8i8, .Cost: 6}, // interleave 3 x 8i8 into 24i8 (and store)
8037 {.ISD: 3, .Type: MVT::v16i8, .Cost: 11}, // interleave 3 x 16i8 into 48i8 (and store)
8038 {.ISD: 3, .Type: MVT::v32i8, .Cost: 13}, // interleave 3 x 32i8 into 96i8 (and store)
8039
8040 {.ISD: 3, .Type: MVT::v2i16, .Cost: 4}, // interleave 3 x 2i16 into 6i16 (and store)
8041 {.ISD: 3, .Type: MVT::v4i16, .Cost: 6}, // interleave 3 x 4i16 into 12i16 (and store)
8042 {.ISD: 3, .Type: MVT::v8i16, .Cost: 12}, // interleave 3 x 8i16 into 24i16 (and store)
8043 {.ISD: 3, .Type: MVT::v16i16, .Cost: 27}, // interleave 3 x 16i16 into 48i16 (and store)
8044 {.ISD: 3, .Type: MVT::v32i16, .Cost: 54}, // interleave 3 x 32i16 into 96i16 (and store)
8045
8046 {.ISD: 3, .Type: MVT::v2i32, .Cost: 4}, // interleave 3 x 2i32 into 6i32 (and store)
8047 {.ISD: 3, .Type: MVT::v4i32, .Cost: 5}, // interleave 3 x 4i32 into 12i32 (and store)
8048 {.ISD: 3, .Type: MVT::v8i32, .Cost: 11}, // interleave 3 x 8i32 into 24i32 (and store)
8049 {.ISD: 3, .Type: MVT::v16i32, .Cost: 22}, // interleave 3 x 16i32 into 48i32 (and store)
8050 {.ISD: 3, .Type: MVT::v32i32, .Cost: 48}, // interleave 3 x 32i32 into 96i32 (and store)
8051
8052 {.ISD: 3, .Type: MVT::v2i64, .Cost: 4}, // interleave 3 x 2i64 into 6i64 (and store)
8053 {.ISD: 3, .Type: MVT::v4i64, .Cost: 6}, // interleave 3 x 4i64 into 12i64 (and store)
8054 {.ISD: 3, .Type: MVT::v8i64, .Cost: 12}, // interleave 3 x 8i64 into 24i64 (and store)
8055 {.ISD: 3, .Type: MVT::v16i64, .Cost: 24}, // interleave 3 x 16i64 into 48i64 (and store)
8056
8057 {.ISD: 4, .Type: MVT::v2i8, .Cost: 4}, // interleave 4 x 2i8 into 8i8 (and store)
8058 {.ISD: 4, .Type: MVT::v4i8, .Cost: 4}, // interleave 4 x 4i8 into 16i8 (and store)
8059 {.ISD: 4, .Type: MVT::v8i8, .Cost: 4}, // interleave 4 x 8i8 into 32i8 (and store)
8060 {.ISD: 4, .Type: MVT::v16i8, .Cost: 8}, // interleave 4 x 16i8 into 64i8 (and store)
8061 {.ISD: 4, .Type: MVT::v32i8, .Cost: 12}, // interleave 4 x 32i8 into 128i8 (and store)
8062
8063 {.ISD: 4, .Type: MVT::v2i16, .Cost: 2}, // interleave 4 x 2i16 into 8i16 (and store)
8064 {.ISD: 4, .Type: MVT::v4i16, .Cost: 6}, // interleave 4 x 4i16 into 16i16 (and store)
8065 {.ISD: 4, .Type: MVT::v8i16, .Cost: 10}, // interleave 4 x 8i16 into 32i16 (and store)
8066 {.ISD: 4, .Type: MVT::v16i16, .Cost: 32}, // interleave 4 x 16i16 into 64i16 (and store)
8067 {.ISD: 4, .Type: MVT::v32i16, .Cost: 64}, // interleave 4 x 32i16 into 128i16 (and store)
8068
8069 {.ISD: 4, .Type: MVT::v2i32, .Cost: 5}, // interleave 4 x 2i32 into 8i32 (and store)
8070 {.ISD: 4, .Type: MVT::v4i32, .Cost: 6}, // interleave 4 x 4i32 into 16i32 (and store)
8071 {.ISD: 4, .Type: MVT::v8i32, .Cost: 16}, // interleave 4 x 8i32 into 32i32 (and store)
8072 {.ISD: 4, .Type: MVT::v16i32, .Cost: 32}, // interleave 4 x 16i32 into 64i32 (and store)
8073 {.ISD: 4, .Type: MVT::v32i32, .Cost: 64}, // interleave 4 x 32i32 into 128i32 (and store)
8074
8075 {.ISD: 4, .Type: MVT::v2i64, .Cost: 6}, // interleave 4 x 2i64 into 8i64 (and store)
8076 {.ISD: 4, .Type: MVT::v4i64, .Cost: 8}, // interleave 4 x 4i64 into 16i64 (and store)
8077 {.ISD: 4, .Type: MVT::v8i64, .Cost: 20}, // interleave 4 x 8i64 into 32i64 (and store)
8078 {.ISD: 4, .Type: MVT::v16i64, .Cost: 40}, // interleave 4 x 16i64 into 64i64 (and store)
8079
8080 {.ISD: 6, .Type: MVT::v2i8, .Cost: 7}, // interleave 6 x 2i8 into 12i8 (and store)
8081 {.ISD: 6, .Type: MVT::v4i8, .Cost: 9}, // interleave 6 x 4i8 into 24i8 (and store)
8082 {.ISD: 6, .Type: MVT::v8i8, .Cost: 16}, // interleave 6 x 8i8 into 48i8 (and store)
8083 {.ISD: 6, .Type: MVT::v16i8, .Cost: 27}, // interleave 6 x 16i8 into 96i8 (and store)
8084 {.ISD: 6, .Type: MVT::v32i8, .Cost: 90}, // interleave 6 x 32i8 into 192i8 (and store)
8085
8086 {.ISD: 6, .Type: MVT::v2i16, .Cost: 10}, // interleave 6 x 2i16 into 12i16 (and store)
8087 {.ISD: 6, .Type: MVT::v4i16, .Cost: 15}, // interleave 6 x 4i16 into 24i16 (and store)
8088 {.ISD: 6, .Type: MVT::v8i16, .Cost: 21}, // interleave 6 x 8i16 into 48i16 (and store)
8089 {.ISD: 6, .Type: MVT::v16i16, .Cost: 58}, // interleave 6 x 16i16 into 96i16 (and store)
8090 {.ISD: 6, .Type: MVT::v32i16, .Cost: 90}, // interleave 6 x 32i16 into 192i16 (and store)
8091
8092 {.ISD: 6, .Type: MVT::v2i32, .Cost: 9}, // interleave 6 x 2i32 into 12i32 (and store)
8093 {.ISD: 6, .Type: MVT::v4i32, .Cost: 12}, // interleave 6 x 4i32 into 24i32 (and store)
8094 {.ISD: 6, .Type: MVT::v8i32, .Cost: 33}, // interleave 6 x 8i32 into 48i32 (and store)
8095 {.ISD: 6, .Type: MVT::v16i32, .Cost: 66}, // interleave 6 x 16i32 into 96i32 (and store)
8096
8097 {.ISD: 6, .Type: MVT::v2i64, .Cost: 8}, // interleave 6 x 2i64 into 12i64 (and store)
8098 {.ISD: 6, .Type: MVT::v4i64, .Cost: 15}, // interleave 6 x 4i64 into 24i64 (and store)
8099 {.ISD: 6, .Type: MVT::v8i64, .Cost: 30}, // interleave 6 x 8i64 into 48i64 (and store)
8100 };
8101
8102 static const CostTblEntry SSE2InterleavedStoreTbl[] = {
8103 {.ISD: 2, .Type: MVT::v2i8, .Cost: 1}, // interleave 2 x 2i8 into 4i8 (and store)
8104 {.ISD: 2, .Type: MVT::v4i8, .Cost: 1}, // interleave 2 x 4i8 into 8i8 (and store)
8105 {.ISD: 2, .Type: MVT::v8i8, .Cost: 1}, // interleave 2 x 8i8 into 16i8 (and store)
8106
8107 {.ISD: 2, .Type: MVT::v2i16, .Cost: 1}, // interleave 2 x 2i16 into 4i16 (and store)
8108 {.ISD: 2, .Type: MVT::v4i16, .Cost: 1}, // interleave 2 x 4i16 into 8i16 (and store)
8109
8110 {.ISD: 2, .Type: MVT::v2i32, .Cost: 1}, // interleave 2 x 2i32 into 4i32 (and store)
8111 };
8112
8113 if (Opcode == Instruction::Load) {
8114 auto GetDiscountedCost = [Factor, NumMembers = Indices.size(),
8115 MemOpCosts](const CostTblEntry *Entry) {
8116 // NOTE: this is just an approximation!
8117 // It can over/under -estimate the cost!
8118 return MemOpCosts + divideCeil(Numerator: NumMembers * Entry->Cost, Denominator: Factor);
8119 };
8120
8121 if (ST->hasAVX2())
8122 if (const auto *Entry = CostTableLookup(Table: AVX2InterleavedLoadTbl, ISD: Factor,
8123 Ty: ETy.getSimpleVT()))
8124 return GetDiscountedCost(Entry);
8125
8126 if (ST->hasSSSE3())
8127 if (const auto *Entry = CostTableLookup(Table: SSSE3InterleavedLoadTbl, ISD: Factor,
8128 Ty: ETy.getSimpleVT()))
8129 return GetDiscountedCost(Entry);
8130
8131 if (ST->hasSSE2())
8132 if (const auto *Entry = CostTableLookup(Table: SSE2InterleavedLoadTbl, ISD: Factor,
8133 Ty: ETy.getSimpleVT()))
8134 return GetDiscountedCost(Entry);
8135 } else {
8136 assert(Opcode == Instruction::Store &&
8137 "Expected Store Instruction at this point");
8138 assert((!Indices.size() || Indices.size() == Factor) &&
8139 "Interleaved store only supports fully-interleaved groups.");
8140 if (ST->hasAVX2())
8141 if (const auto *Entry = CostTableLookup(Table: AVX2InterleavedStoreTbl, ISD: Factor,
8142 Ty: ETy.getSimpleVT()))
8143 return MemOpCosts + Entry->Cost;
8144
8145 if (ST->hasSSE2())
8146 if (const auto *Entry = CostTableLookup(Table: SSE2InterleavedStoreTbl, ISD: Factor,
8147 Ty: ETy.getSimpleVT()))
8148 return MemOpCosts + Entry->Cost;
8149 }
8150
8151 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
8152 Alignment, AddressSpace, CostKind,
8153 UseMaskForCond, UseMaskForGaps);
8154}
8155
8156InstructionCost X86TTIImpl::getScalingFactorCost(Type *Ty, GlobalValue *BaseGV,
8157 StackOffset BaseOffset,
8158 bool HasBaseReg, int64_t Scale,
8159 unsigned AddrSpace) const {
8160 // Scaling factors are not free at all.
8161 // An indexed folded instruction, i.e., inst (reg1, reg2, scale),
8162 // will take 2 allocations in the out of order engine instead of 1
8163 // for plain addressing mode, i.e. inst (reg1).
8164 // E.g.,
8165 // vaddps (%rsi,%rdx), %ymm0, %ymm1
8166 // Requires two allocations (one for the load, one for the computation)
8167 // whereas:
8168 // vaddps (%rsi), %ymm0, %ymm1
8169 // Requires just 1 allocation, i.e., freeing allocations for other operations
8170 // and having less micro operations to execute.
8171 //
8172 // For some X86 architectures, this is even worse because for instance for
8173 // stores, the complex addressing mode forces the instruction to use the
8174 // "load" ports instead of the dedicated "store" port.
8175 // E.g., on Haswell:
8176 // vmovaps %ymm1, (%r8, %rdi) can use port 2 or 3.
8177 // vmovaps %ymm1, (%r8) can use port 2, 3, or 7.
8178 TargetLoweringBase::AddrMode AM;
8179 AM.BaseGV = BaseGV;
8180 AM.BaseOffs = BaseOffset.getFixed();
8181 AM.HasBaseReg = HasBaseReg;
8182 AM.Scale = Scale;
8183 AM.ScalableOffset = BaseOffset.getScalable();
8184 if (getTLI()->isLegalAddressingMode(DL, AM, Ty, AS: AddrSpace))
8185 // Scale represents reg2 * scale, thus account for 1
8186 // as soon as we use a second register.
8187 return AM.Scale != 0;
8188 return InstructionCost::getInvalid();
8189}
8190
8191InstructionCost X86TTIImpl::getBranchMispredictPenalty() const {
8192 // TODO: Hook MispredictPenalty of SchedMachineModel into this.
8193 return 14;
8194}
8195
8196bool X86TTIImpl::isVectorShiftByScalarCheap(Type *Ty) const {
8197 unsigned Bits = Ty->getScalarSizeInBits();
8198
8199 // XOP has v16i8/v8i16/v4i32/v2i64 variable vector shifts.
8200 // Splitting for v32i8/v16i16 on XOP+AVX2 targets is still preferred.
8201 if (ST->hasXOP() && (Bits == 8 || Bits == 16 || Bits == 32 || Bits == 64))
8202 return false;
8203
8204 // AVX2 has vpsllv[dq] instructions (and other shifts) that make variable
8205 // shifts just as cheap as scalar ones.
8206 if (ST->hasAVX2() && (Bits == 32 || Bits == 64))
8207 return false;
8208
8209 // AVX512BW has shifts such as vpsllvw.
8210 if (ST->hasBWI() && Bits == 16)
8211 return false;
8212
8213 // Otherwise, it's significantly cheaper to shift by a scalar amount than by a
8214 // fully general vector.
8215 return true;
8216}
8217
8218unsigned X86TTIImpl::getStoreMinimumVF(unsigned VF, Type *ScalarMemTy,
8219 Type *ScalarValTy, Align Alignment,
8220 unsigned AddrSpace) const {
8221 if (ST->hasF16C() && ScalarMemTy->isHalfTy()) {
8222 return 4;
8223 }
8224 return BaseT::getStoreMinimumVF(VF, ScalarMemTy, ScalarValTy, Alignment,
8225 AddrSpace);
8226}
8227
8228bool X86TTIImpl::isProfitableToSinkOperands(Instruction *I,
8229 SmallVectorImpl<Use *> &Ops) const {
8230 using namespace llvm::PatternMatch;
8231
8232 if (I->getOpcode() == Instruction::And &&
8233 (ST->hasBMI() || (I->getType()->isVectorTy() && ST->hasSSE2()))) {
8234 for (auto &Op : I->operands()) {
8235 // (and X, (not Y)) -> (andn X, Y)
8236 if (match(V: Op.get(), P: m_Not(V: m_Value())) && !I->getType()->isIntegerTy(BitWidth: 8)) {
8237 Ops.push_back(Elt: &Op);
8238 return true;
8239 }
8240 // (and X, (splat (not Y))) -> (andn X, (splat Y))
8241 if (match(V: Op.get(),
8242 P: m_Shuffle(v1: m_InsertElt(Val: m_Value(), Elt: m_Not(V: m_Value()), Idx: m_ZeroInt()),
8243 v2: m_Value(), mask: m_ZeroMask()))) {
8244 Use &InsertElt = cast<Instruction>(Val&: Op)->getOperandUse(i: 0);
8245 Use &Not = cast<Instruction>(Val&: InsertElt)->getOperandUse(i: 1);
8246 Ops.push_back(Elt: &Not);
8247 Ops.push_back(Elt: &InsertElt);
8248 Ops.push_back(Elt: &Op);
8249 return true;
8250 }
8251 }
8252 }
8253
8254 FixedVectorType *VTy = dyn_cast<FixedVectorType>(Val: I->getType());
8255 if (!VTy)
8256 return false;
8257
8258 if (I->getOpcode() == Instruction::Mul &&
8259 VTy->getElementType()->isIntegerTy(BitWidth: 64)) {
8260 for (auto &Op : I->operands()) {
8261 // Make sure we are not already sinking this operand
8262 if (any_of(Range&: Ops, P: [&](Use *U) { return U->get() == Op; }))
8263 continue;
8264
8265 // Look for PMULDQ pattern where the input is a sext_inreg from vXi32 or
8266 // the PMULUDQ pattern where the input is a zext_inreg from vXi32.
8267 if (ST->hasSSE41() &&
8268 match(V: Op.get(), P: m_AShr(L: m_Shl(L: m_Value(), R: m_SpecificInt(V: 32)),
8269 R: m_SpecificInt(V: 32)))) {
8270 Ops.push_back(Elt: &cast<Instruction>(Val&: Op)->getOperandUse(i: 0));
8271 Ops.push_back(Elt: &Op);
8272 } else if (ST->hasSSE2() &&
8273 match(V: Op.get(),
8274 P: m_And(L: m_Value(), R: m_SpecificInt(UINT64_C(0xffffffff))))) {
8275 Ops.push_back(Elt: &Op);
8276 }
8277 }
8278
8279 return !Ops.empty();
8280 }
8281
8282 // A uniform shift amount in a vector shift or funnel shift may be much
8283 // cheaper than a generic variable vector shift, so make that pattern visible
8284 // to SDAG by sinking the shuffle instruction next to the shift.
8285 int ShiftAmountOpNum = -1;
8286 if (I->isShift())
8287 ShiftAmountOpNum = 1;
8288 else if (auto *II = dyn_cast<IntrinsicInst>(Val: I)) {
8289 if (II->getIntrinsicID() == Intrinsic::fshl ||
8290 II->getIntrinsicID() == Intrinsic::fshr)
8291 ShiftAmountOpNum = 2;
8292 }
8293
8294 if (ShiftAmountOpNum == -1)
8295 return false;
8296
8297 auto *Shuf = dyn_cast<ShuffleVectorInst>(Val: I->getOperand(i: ShiftAmountOpNum));
8298 if (Shuf && getSplatIndex(Mask: Shuf->getShuffleMask()) >= 0 &&
8299 isVectorShiftByScalarCheap(Ty: I->getType())) {
8300 Ops.push_back(Elt: &I->getOperandUse(i: ShiftAmountOpNum));
8301 return true;
8302 }
8303
8304 return false;
8305}
8306
8307bool X86TTIImpl::useFastCCForInternalCall(Function &F) const {
8308 bool HasEGPR = ST->hasEGPR();
8309 const TargetMachine &TM = getTLI()->getTargetMachine();
8310
8311 for (User *U : F.users()) {
8312 CallBase *CB = dyn_cast<CallBase>(Val: U);
8313 if (!CB || CB->getCalledOperand() != &F)
8314 continue;
8315 Function *CallerFunc = CB->getFunction();
8316 if (TM.getSubtarget<X86Subtarget>(F: *CallerFunc).hasEGPR() != HasEGPR)
8317 return false;
8318 }
8319
8320 return true;
8321}
8322