1//===--- InterpBuiltin.cpp - Interpreter for the constexpr VM ---*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8#include "../ExprConstShared.h"
9#include "Boolean.h"
10#include "Char.h"
11#include "EvalEmitter.h"
12#include "Interp.h"
13#include "InterpBuiltinBitCast.h"
14#include "InterpHelpers.h"
15#include "PrimType.h"
16#include "Program.h"
17#include "clang/AST/ASTContext.h"
18#include "clang/AST/ExprCXX.h"
19#include "clang/AST/InferAlloc.h"
20#include "clang/AST/OSLog.h"
21#include "clang/AST/RecordLayout.h"
22#include "clang/Basic/Builtins.h"
23#include "clang/Basic/TargetBuiltins.h"
24#include "clang/Basic/TargetInfo.h"
25#include "llvm/ADT/StringExtras.h"
26#include "llvm/Support/AllocToken.h"
27#include "llvm/Support/CRC.h"
28#include "llvm/Support/ErrorHandling.h"
29#include "llvm/Support/MathExtras.h"
30#include "llvm/Support/SipHash.h"
31
32namespace clang {
33namespace interp {
34
35[[maybe_unused]] static bool isNoopBuiltin(unsigned ID) {
36 switch (ID) {
37 case Builtin::BIas_const:
38 case Builtin::BIforward:
39 case Builtin::BIforward_like:
40 case Builtin::BImove:
41 case Builtin::BImove_if_noexcept:
42 case Builtin::BIaddressof:
43 case Builtin::BI__addressof:
44 case Builtin::BI__builtin_addressof:
45 case Builtin::BI__builtin_launder:
46 return true;
47 default:
48 return false;
49 }
50 return false;
51}
52
53static void discard(InterpStack &Stk, PrimType T) {
54 TYPE_SWITCH(T, { Stk.discard<T>(); });
55}
56
57static bool popToUInt64(const InterpState &S, const Expr *E, uint64_t &Out) {
58 INT_TYPE_SWITCH(*S.getContext().classify(E->getType()), {
59 const auto &Val = S.Stk.pop<T>();
60 if (!Val.isNumber())
61 return false;
62 Out = static_cast<uint64_t>(Val);
63 return true;
64 });
65}
66
67static bool popToAPSInt(InterpStack &Stk, PrimType T, APSInt &Out) {
68 INT_TYPE_SWITCH(T, {
69 const auto &Val = Stk.pop<T>();
70 if (!Val.isNumber())
71 return false;
72 Out = Val.toAPSInt();
73 return true;
74 });
75}
76
77static bool popToAPSInt(InterpState &S, const Expr *E, APSInt &Out) {
78 return popToAPSInt(Stk&: S.Stk, T: *S.getContext().classify(T: E->getType()), Out);
79}
80static bool popToAPSInt(InterpState &S, QualType T, APSInt &Out) {
81 return popToAPSInt(Stk&: S.Stk, T: *S.getContext().classify(T), Out);
82}
83
84/// Check for common reasons a pointer can't be read from, which
85/// are usually not diagnosed in a builtin function.
86static bool isReadable(const Pointer &P) {
87 if (P.isDummy())
88 return false;
89 if (!P.isReadablePointerType())
90 return false;
91 if (!P.isLive())
92 return false;
93 if (P.isOnePastEnd())
94 return false;
95 return true;
96}
97
98/// Pushes \p Val on the stack as the type given by \p QT.
99static void pushInteger(InterpState &S, const APSInt &Val, QualType QT) {
100 assert(QT->isSignedIntegerOrEnumerationType() ||
101 QT->isUnsignedIntegerOrEnumerationType());
102 OptPrimType T = *S.getContext().classify(T: QT);
103 assert(T);
104
105 if (T == PT_IntAPS) {
106 unsigned BitWidth = S.getASTContext().getIntWidth(T: QT);
107 auto Result = S.allocAP<IntegralAP<true>>(BitWidth);
108 Result.copy(V: Val.extOrTrunc(width: BitWidth));
109 S.Stk.push<IntegralAP<true>>(Args&: Result);
110 return;
111 }
112
113 if (T == PT_IntAP) {
114 unsigned BitWidth = S.getASTContext().getIntWidth(T: QT);
115 auto Result = S.allocAP<IntegralAP<false>>(BitWidth);
116 Result.copy(V: Val.extOrTrunc(width: BitWidth));
117 S.Stk.push<IntegralAP<false>>(Args&: Result);
118 return;
119 }
120
121 if (isSignedType(T: *T)) {
122 int64_t V = Val.getSExtValue();
123 INT_TYPE_SWITCH(*T, { S.Stk.push<T>(T::from(V)); });
124 } else {
125 assert(QT->isUnsignedIntegerOrEnumerationType());
126 uint64_t V = Val.getZExtValue();
127 INT_TYPE_SWITCH(*T, { S.Stk.push<T>(T::from(V)); });
128 }
129}
130
131template <typename T>
132static void pushInteger(InterpState &S, T Val, QualType QT) {
133 if constexpr (std::is_same_v<T, APInt>)
134 pushInteger(S, Val: APSInt(Val, !std::is_signed_v<T>), QT);
135 else if constexpr (std::is_same_v<T, APSInt>)
136 pushInteger(S, Val, QT);
137 else
138 pushInteger(S,
139 Val: APSInt(APInt(sizeof(T) * 8, static_cast<uint64_t>(Val),
140 std::is_signed_v<T>),
141 !std::is_signed_v<T>),
142 QT);
143}
144
145static void assignIntegral(InterpState &S, const Pointer &Dest, PrimType ValueT,
146 const APSInt &Value) {
147
148 if (ValueT == PT_IntAPS) {
149 Dest.deref<IntegralAP<true>>() =
150 S.allocAP<IntegralAP<true>>(BitWidth: Value.getBitWidth());
151 Dest.deref<IntegralAP<true>>().copy(V: Value);
152 } else if (ValueT == PT_IntAP) {
153 Dest.deref<IntegralAP<false>>() =
154 S.allocAP<IntegralAP<false>>(BitWidth: Value.getBitWidth());
155 Dest.deref<IntegralAP<false>>().copy(V: Value);
156 } else if (ValueT == PT_Bool) {
157 Dest.deref<Boolean>() = Boolean::from(Value: !Value.isZero());
158 } else {
159 INT_TYPE_SWITCH_NO_BOOL(
160 ValueT, { Dest.deref<T>() = T::from(static_cast<T>(Value)); });
161 }
162}
163
164static QualType getElemType(const Pointer &P) {
165 if (P.isStringPointer()) {
166 return P.asStringPointer()
167 .getLiteral()
168 ->getType()
169 ->getAsArrayTypeUnsafe()
170 ->getElementType();
171 }
172
173 if (P.isOpaquePointer() || P.isIntegralPointer())
174 return P.getType();
175
176 const Descriptor *Desc = P.getFieldDesc();
177 QualType T = Desc->getType();
178 if (Desc->isPrimitive())
179 return T;
180 if (T->isPointerType())
181 return T->castAs<PointerType>()->getPointeeType();
182 if (Desc->isArray())
183 return Desc->getElemQualType();
184 if (const auto *AT = T->getAsArrayTypeUnsafe())
185 return AT->getElementType();
186 return T;
187}
188
189static void diagnoseNonConstexprBuiltin(InterpState &S, CodePtr OpPC,
190 unsigned ID) {
191 if (!S.diagnosing())
192 return;
193
194 auto Loc = S.Current->getSource(PC: OpPC);
195 if (S.getLangOpts().CPlusPlus11)
196 S.CCEDiag(SI: Loc, DiagId: diag::note_constexpr_invalid_function)
197 << /*isConstexpr=*/0 << /*isConstructor=*/0
198 << S.getASTContext().BuiltinInfo.getQuotedName(ID);
199 else
200 S.CCEDiag(SI: Loc, DiagId: diag::note_invalid_subexpr_in_const_expr);
201}
202
203static llvm::APSInt convertBoolVectorToInt(const Pointer &Val) {
204 assert(Val.getFieldDesc()->isPrimitiveArray() &&
205 Val.getFieldDesc()->getElemQualType()->isBooleanType() &&
206 "Not a boolean vector");
207 unsigned NumElems = Val.getNumElems();
208
209 // Each element is one bit, so create an integer with NumElts bits.
210 llvm::APSInt Result(NumElems, 0);
211 for (unsigned I = 0; I != NumElems; ++I) {
212 if (Val.elem<bool>(I))
213 Result.setBit(I);
214 }
215
216 return Result;
217}
218
219// Strict double -> float conversion used for X86 PD2PS/cvtsd2ss intrinsics.
220// Reject NaN/Inf/Subnormal inputs and any lossy/inexact conversions.
221static bool convertDoubleToFloatStrict(const APFloat &Src, Floating &Dst,
222 InterpState &S, const Expr *DiagExpr) {
223 if (Src.isInfinity()) {
224 if (S.diagnosing())
225 S.CCEDiag(E: DiagExpr, DiagId: diag::note_constexpr_float_arithmetic) << 0;
226 return false;
227 }
228 if (Src.isNaN()) {
229 if (S.diagnosing())
230 S.CCEDiag(E: DiagExpr, DiagId: diag::note_constexpr_float_arithmetic) << 1;
231 return false;
232 }
233 APFloat Val = Src;
234 bool LosesInfo = false;
235 APFloat::opStatus Status = Val.convert(
236 ToSemantics: APFloat::IEEEsingle(), RM: APFloat::rmNearestTiesToEven, losesInfo: &LosesInfo);
237 if (LosesInfo || Val.isDenormal()) {
238 if (S.diagnosing())
239 S.CCEDiag(E: DiagExpr, DiagId: diag::note_constexpr_float_arithmetic_strict);
240 return false;
241 }
242 if (Status != APFloat::opOK) {
243 if (S.diagnosing())
244 S.CCEDiag(E: DiagExpr, DiagId: diag::note_invalid_subexpr_in_const_expr);
245 return false;
246 }
247 Dst.copy(F: Val);
248 return true;
249}
250
251static bool interp__builtin_is_constant_evaluated(InterpState &S, CodePtr OpPC,
252 const InterpFrame *Frame,
253 const CallExpr *Call) {
254 unsigned Depth = S.Current->getDepth();
255 auto isStdCall = [](const FunctionDecl *F) -> bool {
256 return F && F->isInStdNamespace() && F->getIdentifier() &&
257 F->getIdentifier()->isStr(Str: "is_constant_evaluated");
258 };
259 const InterpFrame *Caller = Frame->Caller;
260 // The current frame is the one for __builtin_is_constant_evaluated.
261 // The one above that, potentially the one for std::is_constant_evaluated().
262 if (S.inConstantContext() && !S.checkingPotentialConstantExpression() &&
263 S.getEvalStatus().Diag &&
264 (Depth == 0 || (Depth == 1 && isStdCall(Frame->getCallee())))) {
265 if (Caller && isStdCall(Frame->getCallee())) {
266 const Expr *E = Caller->getExpr(PC: Caller->getRetPC());
267 S.report(Loc: E->getExprLoc(),
268 DiagId: diag::warn_is_constant_evaluated_always_true_constexpr)
269 << "std::is_constant_evaluated" << E->getSourceRange();
270 } else {
271 S.report(Loc: Call->getExprLoc(),
272 DiagId: diag::warn_is_constant_evaluated_always_true_constexpr)
273 << "__builtin_is_constant_evaluated" << Call->getSourceRange();
274 }
275 }
276
277 S.Stk.push<Boolean>(Args: Boolean::from(Value: S.inConstantContext()));
278 return true;
279}
280
281// __builtin_assume
282// __assume (MS extension)
283static bool interp__builtin_assume(InterpState &S, CodePtr OpPC,
284 const InterpFrame *Frame,
285 const CallExpr *Call) {
286 // Nothing to be done here since the argument is NOT evaluated.
287 assert(Call->getNumArgs() == 1);
288 return true;
289}
290
291static bool interp__builtin_strcmp(InterpState &S, CodePtr OpPC,
292 const InterpFrame *Frame,
293 const CallExpr *Call, unsigned ID) {
294 uint64_t Limit = ~static_cast<uint64_t>(0);
295 if (ID == Builtin::BIstrncmp || ID == Builtin::BI__builtin_strncmp ||
296 ID == Builtin::BIwcsncmp || ID == Builtin::BI__builtin_wcsncmp) {
297 if (!popToUInt64(S, E: Call->getArg(Arg: 2), Out&: Limit))
298 return false;
299 }
300
301 const Pointer &B = S.Stk.pop<Pointer>();
302 const Pointer &A = S.Stk.pop<Pointer>();
303 if (ID == Builtin::BIstrcmp || ID == Builtin::BIstrncmp ||
304 ID == Builtin::BIwcscmp || ID == Builtin::BIwcsncmp)
305 diagnoseNonConstexprBuiltin(S, OpPC, ID);
306
307 if (Limit == 0) {
308 pushInteger(S, Val: 0, QT: Call->getType());
309 return true;
310 }
311
312 if (!CheckLive(S, OpPC, Ptr: A, AK: AK_Read) || !CheckLive(S, OpPC, Ptr: B, AK: AK_Read))
313 return false;
314
315 if (!A.isReadablePointerType() || !B.isReadablePointerType())
316 return false;
317
318 if (A.isDummy() || B.isDummy() || A.isUnknownSizeArray() ||
319 B.isUnknownSizeArray())
320 return false;
321
322 bool IsWide = ID == Builtin::BIwcscmp || ID == Builtin::BIwcsncmp ||
323 ID == Builtin::BI__builtin_wcscmp ||
324 ID == Builtin::BI__builtin_wcsncmp;
325
326 QualType ElemTy = getElemType(P: A);
327 // Different element types shouldn't happen, but with casts they can.
328 if (!S.getASTContext().hasSameUnqualifiedType(T1: ElemTy, T2: getElemType(P: B)) ||
329 !S.getContext().canClassify(T: ElemTy))
330 return false;
331
332 PrimType ElemT = *S.getContext().classify(T: ElemTy);
333
334 auto returnResult = [&](int V) -> bool {
335 pushInteger(S, Val: V, QT: Call->getType());
336 return true;
337 };
338
339 unsigned IndexA = A.getIndex();
340 unsigned IndexB = B.getIndex();
341 unsigned NumElemsA = A.getNumElems();
342 unsigned NumElemsB = B.getNumElems();
343 uint64_t Steps = 0;
344 for (;; ++IndexA, ++IndexB, ++Steps) {
345
346 if (Steps >= Limit)
347 break;
348
349 // Diagnose this as a read of one-past-the-end.
350 if (IndexA >= NumElemsA || IndexB >= NumElemsB) {
351 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_access_past_end)
352 << AK_Read << S.Current->getRange(PC: OpPC);
353 return false;
354 }
355
356 if (IsWide) {
357 INT_TYPE_SWITCH(ElemT, {
358 T CA = A.loadElem<T>(IndexA);
359 T CB = B.loadElem<T>(IndexB);
360 if (CA > CB)
361 return returnResult(1);
362 if (CA < CB)
363 return returnResult(-1);
364 if (CA.isZero() || CB.isZero())
365 return returnResult(0);
366 });
367 continue;
368 }
369
370 uint8_t CA = A.loadElem<uint8_t>(I: IndexA);
371 uint8_t CB = B.loadElem<uint8_t>(I: IndexB);
372
373 if (CA > CB)
374 return returnResult(1);
375 if (CA < CB)
376 return returnResult(-1);
377 if (CA == 0 || CB == 0)
378 return returnResult(0);
379 }
380
381 return returnResult(0);
382}
383
384static bool interp__builtin_strlen(InterpState &S, CodePtr OpPC,
385 const InterpFrame *Frame,
386 const CallExpr *Call, unsigned ID) {
387 const Pointer &StrPtr = S.Stk.pop<Pointer>().expand();
388
389 if (ID == Builtin::BIstrlen || ID == Builtin::BIwcslen)
390 diagnoseNonConstexprBuiltin(S, OpPC, ID);
391
392 if (StrPtr.isConstexprUnknown())
393 return false;
394
395 if (!CheckArray(S, OpPC, Ptr: StrPtr))
396 return false;
397
398 if (!CheckLive(S, OpPC, Ptr: StrPtr, AK: AK_Read))
399 return false;
400
401 // For string literal pointers, this is pretty simple.
402 if (StrPtr.isStringPointer()) {
403 if (StrPtr.isOnePastEnd())
404 return CheckRange(S, OpPC, Ptr: StrPtr, AK: AK_Read);
405
406 const auto *Lit = StrPtr.asStringPointer().getLiteral();
407 int64_t Off = StrPtr.getByteOffset();
408 if (Off < 0)
409 return false;
410
411 UnsignedOrNone ZeroIndex = Lit->findZeroCodeUnit(StartIndex: Off);
412 if (!ZeroIndex)
413 return false;
414 pushInteger(S, Val: *ZeroIndex, QT: Call->getType());
415 return true;
416 }
417
418 if (!StrPtr.isBlockPointer())
419 return false;
420
421 if (!CheckDummy(S, OpPC, Ptr: StrPtr, AK: AK_Read))
422 return false;
423
424 if (!StrPtr.getFieldDesc()->isPrimitiveArray())
425 return false;
426
427 assert(StrPtr.getFieldDesc()->isPrimitiveArray());
428 PrimType ElemT = StrPtr.getFieldDesc()->getPrimType();
429 unsigned ElemSize = StrPtr.getFieldDesc()->getElemDataSize();
430 if (ElemSize != 1 && ElemSize != 2 && ElemSize != 4)
431 return Invalid(S, OpPC);
432
433 if (ID == Builtin::BI__builtin_wcslen || ID == Builtin::BIwcslen) {
434 const ASTContext &AC = S.getASTContext();
435 unsigned WCharSize = AC.getTypeSizeInChars(T: AC.getWCharType()).getQuantity();
436 if (StrPtr.getFieldDesc()->getElemDataSize() != WCharSize)
437 return false;
438 }
439
440 size_t Len = 0;
441 for (size_t I = StrPtr.getIndex();; ++I, ++Len) {
442 PtrView ElemPtr = StrPtr.view().atIndex(Idx: I);
443
444 if (!CheckRange(S, OpPC, Ptr: ElemPtr, AK: AK_Read))
445 return false;
446
447 uint32_t Val;
448 FIXED_SIZE_INT_TYPE_SWITCH(
449 ElemT, { Val = static_cast<uint32_t>(ElemPtr.deref<T>()); });
450 if (Val == 0)
451 break;
452 }
453
454 pushInteger(S, Val: Len, QT: Call->getType());
455
456 return true;
457}
458
459static bool interp__builtin_nan(InterpState &S, CodePtr OpPC,
460 const InterpFrame *Frame, const CallExpr *Call,
461 bool Signaling) {
462 const Pointer &Arg = S.Stk.pop<Pointer>();
463
464 if (!CheckLoad(S, OpPC, Ptr: Arg))
465 return false;
466
467 // Convert the given string to an integer using StringRef's API.
468 llvm::APInt Fill;
469 if (Arg.isBlockPointer()) {
470 if (!Arg.getFieldDesc()->isPrimitiveArray())
471 return Invalid(S, OpPC);
472
473 std::string Str;
474 unsigned ArgLength = Arg.getNumElems();
475 bool FoundZero = false;
476 for (unsigned I = 0; I != ArgLength; ++I) {
477 if (!Arg.isElementInitialized(Index: I))
478 return false;
479
480 if (Arg.loadElem<int8_t>(I) == 0) {
481 FoundZero = true;
482 break;
483 }
484 Str += Arg.elem<char>(I);
485 }
486
487 // If we didn't find a NUL byte, diagnose as a one-past-the-end read.
488 if (!FoundZero)
489 return CheckRange(S, OpPC, Ptr: Arg.atIndex(Idx: ArgLength), AK: AK_Read);
490
491 // Treat empty strings as if they were zero.
492 if (Str.empty())
493 Fill = llvm::APInt(32, 0);
494 else if (StringRef(Str).getAsInteger(Radix: 0, Result&: Fill))
495 return false;
496 } else if (Arg.isStringPointer()) {
497 if (!Arg.asStringPointer().getLiteral()->isOrdinary())
498 return false;
499 StringRef Str = Arg.asStringPointer().getLiteral()->getString();
500 // Treat empty strings as if they were zero.
501 if (Str.empty())
502 Fill = llvm::APInt(32, 0);
503 else if (StringRef(Str).getAsInteger(Radix: 0, Result&: Fill))
504 return false;
505 } else {
506 return false;
507 }
508
509 const llvm::fltSemantics &TargetSemantics =
510 S.getASTContext().getFloatTypeSemantics(
511 T: Call->getDirectCallee()->getReturnType());
512
513 Floating Result = S.allocFloat(Sem: TargetSemantics);
514 if (S.getASTContext().getTargetInfo().isNan2008()) {
515 if (Signaling)
516 Result.copy(
517 F: llvm::APFloat::getSNaN(Sem: TargetSemantics, /*Negative=*/false, payload: &Fill));
518 else
519 Result.copy(
520 F: llvm::APFloat::getQNaN(Sem: TargetSemantics, /*Negative=*/false, payload: &Fill));
521 } else {
522 // Prior to IEEE 754-2008, architectures were allowed to choose whether
523 // the first bit of their significand was set for qNaN or sNaN. MIPS chose
524 // a different encoding to what became a standard in 2008, and for pre-
525 // 2008 revisions, MIPS interpreted sNaN-2008 as qNan and qNaN-2008 as
526 // sNaN. This is now known as "legacy NaN" encoding.
527 if (Signaling)
528 Result.copy(
529 F: llvm::APFloat::getQNaN(Sem: TargetSemantics, /*Negative=*/false, payload: &Fill));
530 else
531 Result.copy(
532 F: llvm::APFloat::getSNaN(Sem: TargetSemantics, /*Negative=*/false, payload: &Fill));
533 }
534
535 S.Stk.push<Floating>(Args&: Result);
536 return true;
537}
538
539static bool interp__builtin_inf(InterpState &S, CodePtr OpPC,
540 const InterpFrame *Frame,
541 const CallExpr *Call) {
542 const llvm::fltSemantics &TargetSemantics =
543 S.getASTContext().getFloatTypeSemantics(
544 T: Call->getDirectCallee()->getReturnType());
545
546 Floating Result = S.allocFloat(Sem: TargetSemantics);
547 Result.copy(F: APFloat::getInf(Sem: TargetSemantics));
548 S.Stk.push<Floating>(Args&: Result);
549 return true;
550}
551
552static bool interp__builtin_copysign(InterpState &S, CodePtr OpPC,
553 const InterpFrame *Frame) {
554 const Floating &Arg2 = S.Stk.pop<Floating>();
555 const Floating &Arg1 = S.Stk.pop<Floating>();
556 Floating Result = S.allocFloat(Sem: Arg1.getSemantics());
557
558 APFloat Copy = Arg1.getAPFloat();
559 Copy.copySign(RHS: Arg2.getAPFloat());
560 Result.copy(F: Copy);
561 S.Stk.push<Floating>(Args&: Result);
562
563 return true;
564}
565
566static bool interp__builtin_fmin(InterpState &S, CodePtr OpPC,
567 const InterpFrame *Frame, bool IsNumBuiltin) {
568 const Floating &RHS = S.Stk.pop<Floating>();
569 const Floating &LHS = S.Stk.pop<Floating>();
570 Floating Result = S.allocFloat(Sem: LHS.getSemantics());
571
572 if (IsNumBuiltin)
573 Result.copy(F: llvm::minimumnum(A: LHS.getAPFloat(), B: RHS.getAPFloat()));
574 else
575 Result.copy(F: minnum(A: LHS.getAPFloat(), B: RHS.getAPFloat()));
576 S.Stk.push<Floating>(Args&: Result);
577 return true;
578}
579
580static bool interp__builtin_fmax(InterpState &S, CodePtr OpPC,
581 const InterpFrame *Frame, bool IsNumBuiltin) {
582 const Floating &RHS = S.Stk.pop<Floating>();
583 const Floating &LHS = S.Stk.pop<Floating>();
584 Floating Result = S.allocFloat(Sem: LHS.getSemantics());
585
586 if (IsNumBuiltin)
587 Result.copy(F: llvm::maximumnum(A: LHS.getAPFloat(), B: RHS.getAPFloat()));
588 else
589 Result.copy(F: maxnum(A: LHS.getAPFloat(), B: RHS.getAPFloat()));
590 S.Stk.push<Floating>(Args&: Result);
591 return true;
592}
593
594/// Defined as __builtin_isnan(...), to accommodate the fact that it can
595/// take a float, double, long double, etc.
596/// But for us, that's all a Floating anyway.
597static bool interp__builtin_isnan(InterpState &S, CodePtr OpPC,
598 const InterpFrame *Frame,
599 const CallExpr *Call) {
600 const Floating &Arg = S.Stk.pop<Floating>();
601
602 pushInteger(S, Val: Arg.isNan(), QT: Call->getType());
603 return true;
604}
605
606static bool interp__builtin_issignaling(InterpState &S, CodePtr OpPC,
607 const InterpFrame *Frame,
608 const CallExpr *Call) {
609 const Floating &Arg = S.Stk.pop<Floating>();
610
611 pushInteger(S, Val: Arg.isSignaling(), QT: Call->getType());
612 return true;
613}
614
615static bool interp__builtin_isinf(InterpState &S, CodePtr OpPC,
616 const InterpFrame *Frame, bool CheckSign,
617 const CallExpr *Call) {
618 const Floating &Arg = S.Stk.pop<Floating>();
619 APFloat F = Arg.getAPFloat();
620 bool IsInf = F.isInfinity();
621
622 if (CheckSign)
623 pushInteger(S, Val: IsInf ? (F.isNegative() ? -1 : 1) : 0, QT: Call->getType());
624 else
625 pushInteger(S, Val: IsInf, QT: Call->getType());
626 return true;
627}
628
629static bool interp__builtin_isfinite(InterpState &S, CodePtr OpPC,
630 const InterpFrame *Frame,
631 const CallExpr *Call) {
632 const Floating &Arg = S.Stk.pop<Floating>();
633
634 pushInteger(S, Val: Arg.isFinite(), QT: Call->getType());
635 return true;
636}
637
638static bool interp__builtin_isnormal(InterpState &S, CodePtr OpPC,
639 const InterpFrame *Frame,
640 const CallExpr *Call) {
641 const Floating &Arg = S.Stk.pop<Floating>();
642
643 pushInteger(S, Val: Arg.isNormal(), QT: Call->getType());
644 return true;
645}
646
647static bool interp__builtin_issubnormal(InterpState &S, CodePtr OpPC,
648 const InterpFrame *Frame,
649 const CallExpr *Call) {
650 const Floating &Arg = S.Stk.pop<Floating>();
651
652 pushInteger(S, Val: Arg.isDenormal(), QT: Call->getType());
653 return true;
654}
655
656static bool interp__builtin_iszero(InterpState &S, CodePtr OpPC,
657 const InterpFrame *Frame,
658 const CallExpr *Call) {
659 const Floating &Arg = S.Stk.pop<Floating>();
660
661 pushInteger(S, Val: Arg.isZero(), QT: Call->getType());
662 return true;
663}
664
665static bool interp__builtin_signbit(InterpState &S, CodePtr OpPC,
666 const InterpFrame *Frame,
667 const CallExpr *Call) {
668 const Floating &Arg = S.Stk.pop<Floating>();
669
670 pushInteger(S, Val: Arg.isNegative(), QT: Call->getType());
671 return true;
672}
673
674static bool interp_floating_comparison(InterpState &S, CodePtr OpPC,
675 const CallExpr *Call, unsigned ID) {
676 const Floating &RHS = S.Stk.pop<Floating>();
677 const Floating &LHS = S.Stk.pop<Floating>();
678
679 pushInteger(
680 S,
681 Val: [&] {
682 switch (ID) {
683 case Builtin::BI__builtin_isgreater:
684 return LHS > RHS;
685 case Builtin::BI__builtin_isgreaterequal:
686 return LHS >= RHS;
687 case Builtin::BI__builtin_isless:
688 return LHS < RHS;
689 case Builtin::BI__builtin_islessequal:
690 return LHS <= RHS;
691 case Builtin::BI__builtin_islessgreater: {
692 ComparisonCategoryResult Cmp = LHS.compare(RHS);
693 return Cmp == ComparisonCategoryResult::Less ||
694 Cmp == ComparisonCategoryResult::Greater;
695 }
696 case Builtin::BI__builtin_isunordered:
697 return LHS.compare(RHS) == ComparisonCategoryResult::Unordered;
698 default:
699 llvm_unreachable("Unexpected builtin ID: Should be a floating point "
700 "comparison function");
701 }
702 }(),
703 QT: Call->getType());
704 return true;
705}
706
707/// First parameter to __builtin_isfpclass is the floating value, the
708/// second one is an integral value.
709static bool interp__builtin_isfpclass(InterpState &S, CodePtr OpPC,
710 const InterpFrame *Frame,
711 const CallExpr *Call) {
712 APSInt FPClassArg;
713 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: FPClassArg))
714 return false;
715 const Floating &F = S.Stk.pop<Floating>();
716
717 int32_t Result = static_cast<int32_t>(
718 (F.classify() & std::move(FPClassArg)).getZExtValue());
719 pushInteger(S, Val: Result, QT: Call->getType());
720
721 return true;
722}
723
724/// Five int values followed by one floating value.
725/// __builtin_fpclassify(int, int, int, int, int, float)
726static bool interp__builtin_fpclassify(InterpState &S, CodePtr OpPC,
727 const InterpFrame *Frame,
728 const CallExpr *Call) {
729 const Floating &Val = S.Stk.pop<Floating>();
730
731 PrimType IntT = *S.getContext().classify(E: Call->getArg(Arg: 0));
732 APSInt Values[5];
733 for (unsigned I = 0; I != 5; ++I) {
734 if (!popToAPSInt(Stk&: S.Stk, T: IntT, Out&: Values[4 - I]))
735 return false;
736 }
737
738 unsigned Index;
739 switch (Val.getCategory()) {
740 case APFloat::fcNaN:
741 Index = 0;
742 break;
743 case APFloat::fcInfinity:
744 Index = 1;
745 break;
746 case APFloat::fcNormal:
747 Index = Val.isDenormal() ? 3 : 2;
748 break;
749 case APFloat::fcZero:
750 Index = 4;
751 break;
752 }
753
754 // The last argument is first on the stack.
755 assert(Index <= 4);
756
757 pushInteger(S, Val: Values[Index], QT: Call->getType());
758 return true;
759}
760
761static inline Floating abs(InterpState &S, const Floating &In) {
762 if (!In.isNegative())
763 return In;
764
765 Floating Output = S.allocFloat(Sem: In.getSemantics());
766 APFloat New = In.getAPFloat();
767 New.changeSign();
768 Output.copy(F: New);
769 return Output;
770}
771
772// The C standard says "fabs raises no floating-point exceptions,
773// even if x is a signaling NaN. The returned value is independent of
774// the current rounding direction mode." Therefore constant folding can
775// proceed without regard to the floating point settings.
776// Reference, WG14 N2478 F.10.4.3
777static bool interp__builtin_fabs(InterpState &S, CodePtr OpPC,
778 const InterpFrame *Frame) {
779 const Floating &Val = S.Stk.pop<Floating>();
780 S.Stk.push<Floating>(Args: abs(S, In: Val));
781 return true;
782}
783
784static bool interp__builtin_abs(InterpState &S, CodePtr OpPC,
785 const InterpFrame *Frame,
786 const CallExpr *Call) {
787 APSInt Val;
788 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Val))
789 return false;
790 if (Val ==
791 APSInt(APInt::getSignedMinValue(numBits: Val.getBitWidth()), /*IsUnsigned=*/false))
792 return false;
793 if (Val.isNegative())
794 Val.negate();
795 pushInteger(S, Val, QT: Call->getType());
796 return true;
797}
798
799static bool interp__builtin_popcount(InterpState &S, CodePtr OpPC,
800 const InterpFrame *Frame,
801 const CallExpr *Call) {
802 APSInt Val;
803 if (Call->getArg(Arg: 0)->getType()->isExtVectorBoolType()) {
804 const Pointer &Arg = S.Stk.pop<Pointer>();
805 Val = convertBoolVectorToInt(Val: Arg);
806 } else {
807 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Val))
808 return false;
809 }
810 pushInteger(S, Val: Val.popcount(), QT: Call->getType());
811 return true;
812}
813
814static bool interp__builtin_ia32_crc32(InterpState &S, CodePtr OpPC,
815 const InterpFrame *Frame,
816 const CallExpr *Call,
817 unsigned DataBytes) {
818 uint64_t DataVal;
819 if (!popToUInt64(S, E: Call->getArg(Arg: 1), Out&: DataVal))
820 return false;
821 uint64_t CRCVal;
822 if (!popToUInt64(S, E: Call->getArg(Arg: 0), Out&: CRCVal))
823 return false;
824
825 // CRC32C polynomial (iSCSI polynomial, bit-reversed)
826 static const uint32_t CRC32C_POLY = 0x82F63B78;
827
828 uint32_t Result = llvm::calculateReflectedCRC32(
829 Crc: static_cast<uint32_t>(CRCVal), Data: DataVal, DataBytes, Poly: CRC32C_POLY);
830
831 pushInteger(S, Val: Result, QT: Call->getType());
832 return true;
833}
834
835static bool interp__builtin_classify_type(InterpState &S, CodePtr OpPC,
836 const InterpFrame *Frame,
837 const CallExpr *Call) {
838 // This is an unevaluated call, so there are no arguments on the stack.
839 assert(Call->getNumArgs() == 1);
840 const Expr *Arg = Call->getArg(Arg: 0);
841
842 GCCTypeClass ResultClass =
843 EvaluateBuiltinClassifyType(T: Arg->getType(), LangOpts: S.getLangOpts());
844 int32_t ReturnVal = static_cast<int32_t>(ResultClass);
845 pushInteger(S, Val: ReturnVal, QT: Call->getType());
846 return true;
847}
848
849// __builtin_expect(long, long)
850// __builtin_expect_with_probability(long, long, double)
851static bool interp__builtin_expect(InterpState &S, CodePtr OpPC,
852 const InterpFrame *Frame,
853 const CallExpr *Call) {
854 // The return value is simply the value of the first parameter.
855 // We ignore the probability.
856 unsigned NumArgs = Call->getNumArgs();
857 assert(NumArgs == 2 || NumArgs == 3);
858
859 PrimType ArgT = *S.getContext().classify(T: Call->getArg(Arg: 0)->getType());
860 if (NumArgs == 3)
861 S.Stk.discard<Floating>();
862 discard(Stk&: S.Stk, T: ArgT);
863 // Top of the stack is now the first paramter. Leave it there as the return
864 // value.
865
866 return true;
867}
868
869static bool interp__builtin_addressof(InterpState &S, CodePtr OpPC,
870 const InterpFrame *Frame,
871 const CallExpr *Call) {
872#ifndef NDEBUG
873 assert(Call->getArg(0)->isLValue());
874 PrimType PtrT = S.getContext().classify(Call->getArg(0)).value_or(PT_Ptr);
875 assert(PtrT == PT_Ptr &&
876 "Unsupported pointer type passed to __builtin_addressof()");
877#endif
878 return true;
879}
880
881static bool interp__builtin_move(InterpState &S, CodePtr OpPC,
882 const InterpFrame *Frame,
883 const CallExpr *Call) {
884 return Call->getDirectCallee()->isConstexpr();
885}
886
887static bool interp__builtin_eh_return_data_regno(InterpState &S, CodePtr OpPC,
888 const InterpFrame *Frame,
889 const CallExpr *Call) {
890 APSInt Arg;
891 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Arg))
892 return false;
893
894 int Result = S.getASTContext().getTargetInfo().getEHDataRegisterNumber(
895 RegNo: Arg.getZExtValue());
896 pushInteger(S, Val: Result, QT: Call->getType());
897 return true;
898}
899
900// Two integral values followed by a pointer (lhs, rhs, resultOut)
901static bool interp__builtin_overflowop(InterpState &S, CodePtr OpPC,
902 const CallExpr *Call,
903 unsigned BuiltinOp) {
904 const Pointer &ResultPtr = S.Stk.pop<Pointer>();
905 if (ResultPtr.isDummy() || !ResultPtr.isBlockPointer())
906 return false;
907
908 PrimType RHST = *S.getContext().classify(T: Call->getArg(Arg: 1)->getType());
909 PrimType LHST = *S.getContext().classify(T: Call->getArg(Arg: 0)->getType());
910 APSInt RHS;
911 if (!popToAPSInt(Stk&: S.Stk, T: RHST, Out&: RHS))
912 return false;
913 APSInt LHS;
914 if (!popToAPSInt(Stk&: S.Stk, T: LHST, Out&: LHS))
915 return false;
916 QualType ResultType = Call->getArg(Arg: 2)->getType()->getPointeeType();
917 PrimType ResultT = *S.getContext().classify(T: ResultType);
918 bool Overflow;
919
920 APSInt Result;
921 if (BuiltinOp == Builtin::BI__builtin_add_overflow ||
922 BuiltinOp == Builtin::BI__builtin_sub_overflow ||
923 BuiltinOp == Builtin::BI__builtin_mul_overflow) {
924 bool IsSigned = LHS.isSigned() || RHS.isSigned() ||
925 ResultType->isSignedIntegerOrEnumerationType();
926 bool AllSigned = LHS.isSigned() && RHS.isSigned() &&
927 ResultType->isSignedIntegerOrEnumerationType();
928 uint64_t LHSSize = LHS.getBitWidth();
929 uint64_t RHSSize = RHS.getBitWidth();
930 uint64_t ResultSize = S.getASTContext().getIntWidth(T: ResultType);
931 uint64_t MaxBits = std::max(a: std::max(a: LHSSize, b: RHSSize), b: ResultSize);
932
933 // Add an additional bit if the signedness isn't uniformly agreed to. We
934 // could do this ONLY if there is a signed and an unsigned that both have
935 // MaxBits, but the code to check that is pretty nasty. The issue will be
936 // caught in the shrink-to-result later anyway.
937 if (IsSigned && !AllSigned)
938 ++MaxBits;
939
940 LHS = APSInt(LHS.extOrTrunc(width: MaxBits), !IsSigned);
941 RHS = APSInt(RHS.extOrTrunc(width: MaxBits), !IsSigned);
942 Result = APSInt(MaxBits, !IsSigned);
943 }
944
945 // Find largest int.
946 switch (BuiltinOp) {
947 default:
948 llvm_unreachable("Invalid value for BuiltinOp");
949 case Builtin::BI__builtin_add_overflow:
950 case Builtin::BI__builtin_sadd_overflow:
951 case Builtin::BI__builtin_saddl_overflow:
952 case Builtin::BI__builtin_saddll_overflow:
953 case Builtin::BI__builtin_uadd_overflow:
954 case Builtin::BI__builtin_uaddl_overflow:
955 case Builtin::BI__builtin_uaddll_overflow:
956 Result = LHS.isSigned() ? LHS.sadd_ov(RHS, Overflow)
957 : LHS.uadd_ov(RHS, Overflow);
958 break;
959 case Builtin::BI__builtin_sub_overflow:
960 case Builtin::BI__builtin_ssub_overflow:
961 case Builtin::BI__builtin_ssubl_overflow:
962 case Builtin::BI__builtin_ssubll_overflow:
963 case Builtin::BI__builtin_usub_overflow:
964 case Builtin::BI__builtin_usubl_overflow:
965 case Builtin::BI__builtin_usubll_overflow:
966 Result = LHS.isSigned() ? LHS.ssub_ov(RHS, Overflow)
967 : LHS.usub_ov(RHS, Overflow);
968 break;
969 case Builtin::BI__builtin_mul_overflow:
970 case Builtin::BI__builtin_smul_overflow:
971 case Builtin::BI__builtin_smull_overflow:
972 case Builtin::BI__builtin_smulll_overflow:
973 case Builtin::BI__builtin_umul_overflow:
974 case Builtin::BI__builtin_umull_overflow:
975 case Builtin::BI__builtin_umulll_overflow:
976 Result = LHS.isSigned() ? LHS.smul_ov(RHS, Overflow)
977 : LHS.umul_ov(RHS, Overflow);
978 break;
979 }
980
981 // In the case where multiple sizes are allowed, truncate and see if
982 // the values are the same.
983 if (BuiltinOp == Builtin::BI__builtin_add_overflow ||
984 BuiltinOp == Builtin::BI__builtin_sub_overflow ||
985 BuiltinOp == Builtin::BI__builtin_mul_overflow) {
986 // APSInt doesn't have a TruncOrSelf, so we use extOrTrunc instead,
987 // since it will give us the behavior of a TruncOrSelf in the case where
988 // its parameter <= its size. We previously set Result to be at least the
989 // integer width of the result, so getIntWidth(ResultType) <=
990 // Result.BitWidth
991 APSInt Temp = Result.extOrTrunc(width: S.getASTContext().getIntWidth(T: ResultType));
992 Temp.setIsSigned(ResultType->isSignedIntegerOrEnumerationType());
993
994 if (!APSInt::isSameValue(I1: Temp, I2: Result))
995 Overflow = true;
996 Result = std::move(Temp);
997 }
998
999 // Write Result to ResultPtr and put Overflow on the stack.
1000 assignIntegral(S, Dest: ResultPtr, ValueT: ResultT, Value: Result);
1001 if (ResultPtr.canBeInitialized())
1002 ResultPtr.initialize();
1003
1004 assert(Call->getDirectCallee()->getReturnType()->isBooleanType());
1005 S.Stk.push<Boolean>(Args&: Overflow);
1006 return true;
1007}
1008
1009/// Three integral values followed by a pointer (lhs, rhs, carry, carryOut).
1010static bool interp__builtin_carryop(InterpState &S, CodePtr OpPC,
1011 const InterpFrame *Frame,
1012 const CallExpr *Call, unsigned BuiltinOp) {
1013 const Pointer &CarryOutPtr = S.Stk.pop<Pointer>();
1014 PrimType LHST = *S.getContext().classify(T: Call->getArg(Arg: 0)->getType());
1015 PrimType RHST = *S.getContext().classify(T: Call->getArg(Arg: 1)->getType());
1016 APSInt CarryIn;
1017 if (!popToAPSInt(Stk&: S.Stk, T: LHST, Out&: CarryIn))
1018 return false;
1019 APSInt RHS;
1020 if (!popToAPSInt(Stk&: S.Stk, T: RHST, Out&: RHS))
1021 return false;
1022 APSInt LHS;
1023 if (!popToAPSInt(Stk&: S.Stk, T: LHST, Out&: LHS))
1024 return false;
1025
1026 if (!isReadable(P: CarryOutPtr))
1027 return false;
1028
1029 APSInt CarryOut;
1030
1031 APSInt Result;
1032 // Copy the number of bits and sign.
1033 Result = LHS;
1034 CarryOut = LHS;
1035
1036 bool FirstOverflowed = false;
1037 bool SecondOverflowed = false;
1038 switch (BuiltinOp) {
1039 default:
1040 llvm_unreachable("Invalid value for BuiltinOp");
1041 case Builtin::BI__builtin_addcb:
1042 case Builtin::BI__builtin_addcs:
1043 case Builtin::BI__builtin_addc:
1044 case Builtin::BI__builtin_addcl:
1045 case Builtin::BI__builtin_addcll:
1046 Result =
1047 LHS.uadd_ov(RHS, Overflow&: FirstOverflowed).uadd_ov(RHS: CarryIn, Overflow&: SecondOverflowed);
1048 break;
1049 case Builtin::BI__builtin_subcb:
1050 case Builtin::BI__builtin_subcs:
1051 case Builtin::BI__builtin_subc:
1052 case Builtin::BI__builtin_subcl:
1053 case Builtin::BI__builtin_subcll:
1054 Result =
1055 LHS.usub_ov(RHS, Overflow&: FirstOverflowed).usub_ov(RHS: CarryIn, Overflow&: SecondOverflowed);
1056 break;
1057 }
1058 // It is possible for both overflows to happen but CGBuiltin uses an OR so
1059 // this is consistent.
1060 CarryOut = (uint64_t)(FirstOverflowed | SecondOverflowed);
1061
1062 QualType CarryOutType = Call->getArg(Arg: 3)->getType()->getPointeeType();
1063 PrimType CarryOutT = *S.getContext().classify(T: CarryOutType);
1064 assignIntegral(S, Dest: CarryOutPtr, ValueT: CarryOutT, Value: CarryOut);
1065 if (CarryOutPtr.canBeInitialized())
1066 CarryOutPtr.initialize();
1067
1068 assert(S.getASTContext().hasSimilarType(Call->getType(),
1069 Call->getArg(0)->getType()));
1070 pushInteger(S, Val: Result, QT: Call->getType());
1071 return true;
1072}
1073
1074static bool interp__builtin_clz(InterpState &S, CodePtr OpPC,
1075 const InterpFrame *Frame, const CallExpr *Call,
1076 unsigned BuiltinOp) {
1077
1078 std::optional<APSInt> Fallback;
1079 if (BuiltinOp == Builtin::BI__builtin_clzg && Call->getNumArgs() == 2) {
1080 APSInt FallbackVal;
1081 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: FallbackVal))
1082 return false;
1083 Fallback = FallbackVal;
1084 }
1085
1086 APSInt Val;
1087 if (Call->getArg(Arg: 0)->getType()->isExtVectorBoolType()) {
1088 const Pointer &Arg = S.Stk.pop<Pointer>();
1089 Val = convertBoolVectorToInt(Val: Arg);
1090 } else {
1091 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Val))
1092 return false;
1093 }
1094
1095 // When the argument is 0, the result of GCC builtins is undefined, whereas
1096 // for Microsoft intrinsics, the result is the bit-width of the argument.
1097 bool ZeroIsUndefined = BuiltinOp != Builtin::BI__lzcnt16 &&
1098 BuiltinOp != Builtin::BI__lzcnt &&
1099 BuiltinOp != Builtin::BI__lzcnt64;
1100
1101 if (Val == 0) {
1102 if (Fallback) {
1103 pushInteger(S, Val: *Fallback, QT: Call->getType());
1104 return true;
1105 }
1106
1107 if (ZeroIsUndefined)
1108 return false;
1109 }
1110
1111 pushInteger(S, Val: Val.countl_zero(), QT: Call->getType());
1112 return true;
1113}
1114
1115static bool interp__builtin_ctz(InterpState &S, CodePtr OpPC,
1116 const InterpFrame *Frame, const CallExpr *Call,
1117 unsigned BuiltinID) {
1118 std::optional<APSInt> Fallback;
1119 if (BuiltinID == Builtin::BI__builtin_ctzg && Call->getNumArgs() == 2) {
1120 APSInt FallbackVal;
1121 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: FallbackVal))
1122 return false;
1123 Fallback = FallbackVal;
1124 }
1125
1126 APSInt Val;
1127 if (Call->getArg(Arg: 0)->getType()->isExtVectorBoolType()) {
1128 const Pointer &Arg = S.Stk.pop<Pointer>();
1129 Val = convertBoolVectorToInt(Val: Arg);
1130 } else {
1131 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Val))
1132 return false;
1133 }
1134
1135 if (Val == 0) {
1136 if (Fallback) {
1137 pushInteger(S, Val: *Fallback, QT: Call->getType());
1138 return true;
1139 }
1140 return false;
1141 }
1142
1143 pushInteger(S, Val: Val.countr_zero(), QT: Call->getType());
1144 return true;
1145}
1146
1147static bool interp__builtin_bswap(InterpState &S, CodePtr OpPC,
1148 const InterpFrame *Frame,
1149 const CallExpr *Call) {
1150 APSInt Val;
1151 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Val))
1152 return false;
1153 if (Val.getBitWidth() == 8 || Val.getBitWidth() == 1)
1154 pushInteger(S, Val, QT: Call->getType());
1155 else
1156 pushInteger(S, Val: Val.byteSwap(), QT: Call->getType());
1157 return true;
1158}
1159
1160/// bool __atomic_always_lock_free(size_t, void const volatile*)
1161/// bool __atomic_is_lock_free(size_t, void const volatile*)
1162static bool interp__builtin_atomic_lock_free(InterpState &S, CodePtr OpPC,
1163 const InterpFrame *Frame,
1164 const CallExpr *Call,
1165 unsigned BuiltinOp) {
1166 auto returnBool = [&S](bool Value) -> bool {
1167 S.Stk.push<Boolean>(Args&: Value);
1168 return true;
1169 };
1170
1171 const Pointer &Ptr = S.Stk.pop<Pointer>();
1172 uint64_t SizeVal;
1173 if (!popToUInt64(S, E: Call->getArg(Arg: 0), Out&: SizeVal))
1174 return false;
1175
1176 // For __atomic_is_lock_free(sizeof(_Atomic(T))), if the size is a power
1177 // of two less than or equal to the maximum inline atomic width, we know it
1178 // is lock-free. If the size isn't a power of two, or greater than the
1179 // maximum alignment where we promote atomics, we know it is not lock-free
1180 // (at least not in the sense of atomic_is_lock_free). Otherwise,
1181 // the answer can only be determined at runtime; for example, 16-byte
1182 // atomics have lock-free implementations on some, but not all,
1183 // x86-64 processors.
1184
1185 // Check power-of-two.
1186 CharUnits Size = CharUnits::fromQuantity(Quantity: SizeVal);
1187 if (Size.isPowerOfTwo()) {
1188 // Check against inlining width.
1189 unsigned InlineWidthBits =
1190 S.getASTContext().getTargetInfo().getMaxAtomicInlineWidth();
1191 if (Size <= S.getASTContext().toCharUnitsFromBits(BitSize: InlineWidthBits)) {
1192
1193 // OK, we will inline appropriately-aligned operations of this size,
1194 // and _Atomic(T) is appropriately-aligned.
1195 if (Size == CharUnits::One())
1196 return returnBool(true);
1197
1198 // Same for null pointers.
1199 assert(BuiltinOp != Builtin::BI__c11_atomic_is_lock_free);
1200 if (Ptr.isZero())
1201 return returnBool(true);
1202
1203 if (Ptr.isIntegralPointer()) {
1204 uint64_t IntVal = Ptr.getIntegerRepresentation();
1205 if (APSInt(APInt(64, IntVal, false), true).isAligned(A: Size.getAsAlign()))
1206 return returnBool(true);
1207 }
1208
1209 const Expr *PtrArg = Call->getArg(Arg: 1);
1210 // Otherwise, check if the type's alignment against Size.
1211 if (const auto *ICE = dyn_cast<ImplicitCastExpr>(Val: PtrArg)) {
1212 // Drop the potential implicit-cast to 'const volatile void*', getting
1213 // the underlying type.
1214 if (ICE->getCastKind() == CK_BitCast)
1215 PtrArg = ICE->getSubExpr();
1216 }
1217
1218 if (const auto *PtrTy = PtrArg->getType()->getAs<PointerType>()) {
1219 QualType PointeeType = PtrTy->getPointeeType();
1220 if (!PointeeType->isIncompleteType() &&
1221 S.getASTContext().getTypeAlignInChars(T: PointeeType) >= Size) {
1222 // OK, we will inline operations on this object.
1223 return returnBool(true);
1224 }
1225 }
1226 }
1227 }
1228
1229 if (BuiltinOp == Builtin::BI__atomic_always_lock_free)
1230 return returnBool(false);
1231
1232 return Invalid(S, OpPC);
1233}
1234
1235/// bool __c11_atomic_is_lock_free(size_t)
1236static bool interp__builtin_c11_atomic_is_lock_free(InterpState &S,
1237 CodePtr OpPC,
1238 const InterpFrame *Frame,
1239 const CallExpr *Call) {
1240 uint64_t SizeVal;
1241 if (!popToUInt64(S, E: Call->getArg(Arg: 0), Out&: SizeVal))
1242 return false;
1243
1244 CharUnits Size = CharUnits::fromQuantity(Quantity: SizeVal);
1245 if (Size.isPowerOfTwo()) {
1246 // Check against inlining width.
1247 unsigned InlineWidthBits =
1248 S.getASTContext().getTargetInfo().getMaxAtomicInlineWidth();
1249 if (Size <= S.getASTContext().toCharUnitsFromBits(BitSize: InlineWidthBits)) {
1250 S.Stk.push<Boolean>(Args: true);
1251 return true;
1252 }
1253 }
1254
1255 return false; // returnBool(false);
1256}
1257
1258/// __builtin_complex(Float A, float B);
1259static bool interp__builtin_complex(InterpState &S, CodePtr OpPC,
1260 const InterpFrame *Frame,
1261 const CallExpr *Call) {
1262 const Floating &Arg2 = S.Stk.pop<Floating>();
1263 const Floating &Arg1 = S.Stk.pop<Floating>();
1264 Pointer &Result = S.Stk.peek<Pointer>();
1265
1266 Result.elem<Floating>(I: 0) = Arg1;
1267 Result.elem<Floating>(I: 1) = Arg2;
1268 Result.initializeAllElements();
1269
1270 return true;
1271}
1272
1273/// __builtin_is_aligned()
1274/// __builtin_align_up()
1275/// __builtin_align_down()
1276/// The first parameter is either an integer or a pointer.
1277/// The second parameter is the requested alignment as an integer.
1278static bool interp__builtin_is_aligned_up_down(InterpState &S, CodePtr OpPC,
1279 const InterpFrame *Frame,
1280 const CallExpr *Call,
1281 unsigned BuiltinOp) {
1282 APSInt Alignment;
1283 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: Alignment))
1284 return false;
1285
1286 if (Alignment < 0 || !Alignment.isPowerOf2()) {
1287 S.FFDiag(E: Call, DiagId: diag::note_constexpr_invalid_alignment) << Alignment;
1288 return false;
1289 }
1290 unsigned SrcWidth = S.getASTContext().getIntWidth(T: Call->getArg(Arg: 0)->getType());
1291 APSInt MaxValue(APInt::getOneBitSet(numBits: SrcWidth, BitNo: SrcWidth - 1));
1292 if (APSInt::compareValues(I1: Alignment, I2: MaxValue) > 0) {
1293 S.FFDiag(E: Call, DiagId: diag::note_constexpr_alignment_too_big)
1294 << MaxValue << Call->getArg(Arg: 0)->getType() << Alignment;
1295 return false;
1296 }
1297
1298 // The first parameter is either an integer or a pointer.
1299 PrimType FirstArgT = *S.Ctx.classify(E: Call->getArg(Arg: 0));
1300
1301 if (isIntegerType(T: FirstArgT)) {
1302 APSInt Src;
1303 if (!popToAPSInt(Stk&: S.Stk, T: FirstArgT, Out&: Src))
1304 return false;
1305 APInt AlignMinusOne = Alignment.extOrTrunc(width: Src.getBitWidth()) - 1;
1306 if (BuiltinOp == Builtin::BI__builtin_align_up) {
1307 APSInt AlignedVal =
1308 APSInt((Src + AlignMinusOne) & ~AlignMinusOne, Src.isUnsigned());
1309 pushInteger(S, Val: AlignedVal, QT: Call->getType());
1310 } else if (BuiltinOp == Builtin::BI__builtin_align_down) {
1311 APSInt AlignedVal = APSInt(Src & ~AlignMinusOne, Src.isUnsigned());
1312 pushInteger(S, Val: AlignedVal, QT: Call->getType());
1313 } else {
1314 assert(*S.Ctx.classify(Call->getType()) == PT_Bool);
1315 S.Stk.push<Boolean>(Args: (Src & AlignMinusOne) == 0);
1316 }
1317 return true;
1318 }
1319 assert(FirstArgT == PT_Ptr);
1320 const Pointer &Ptr = S.Stk.pop<Pointer>();
1321
1322 // Null pointers are always aligned. Preserve null pointers for
1323 // align_up/align_down and return true for is_aligned.
1324 if (Ptr.isZero()) {
1325 if (BuiltinOp == Builtin::BI__builtin_is_aligned) {
1326 S.Stk.push<Boolean>(Args: true);
1327 return true;
1328 }
1329
1330 assert(BuiltinOp == Builtin::BI__builtin_align_up ||
1331 BuiltinOp == Builtin::BI__builtin_align_down);
1332
1333 S.Stk.push<Pointer>(Args: Ptr);
1334 return true;
1335 }
1336
1337 if (!Ptr.isBlockPointer() && !Ptr.isOpaquePointer()) {
1338 S.FFDiag(E: Call->getArg(Arg: 0), DiagId: diag::note_constexpr_alignment_compute)
1339 << Alignment;
1340 return false;
1341 }
1342
1343 const VarDecl *PtrDecl = Ptr.getRootVarDecl();
1344 // We need a pointer for a declaration here.
1345 if (!PtrDecl) {
1346 if (BuiltinOp == Builtin::BI__builtin_is_aligned)
1347 S.FFDiag(E: Call->getArg(Arg: 0), DiagId: diag::note_constexpr_alignment_compute)
1348 << Alignment;
1349 else
1350 S.FFDiag(E: Call->getArg(Arg: 0), DiagId: diag::note_constexpr_alignment_adjust)
1351 << Alignment;
1352 return false;
1353 }
1354
1355 unsigned PtrOffset;
1356 if (Ptr.isBlockPointer()) {
1357 // For one-past-end pointers, we can't call getIndex() since it asserts.
1358 // Use getNumElems() instead which gives the correct index for past-end.
1359 PtrOffset = Ptr.isElementPastEnd() ? Ptr.getNumElems() : Ptr.getIndex();
1360 } else {
1361 if (std::optional<size_t> PtrOff =
1362 Ptr.computeLayoutOffset(ASTCtx: S.getASTContext()))
1363 PtrOffset = *PtrOff;
1364 else
1365 return false;
1366 }
1367
1368 CharUnits BaseAlignment = S.getASTContext().getDeclAlign(D: PtrDecl);
1369 CharUnits PtrAlign =
1370 BaseAlignment.alignmentAtOffset(offset: CharUnits::fromQuantity(Quantity: PtrOffset));
1371
1372 if (BuiltinOp == Builtin::BI__builtin_is_aligned) {
1373 if (PtrAlign.getQuantity() >= Alignment) {
1374 S.Stk.push<Boolean>(Args: true);
1375 return true;
1376 }
1377 // If the alignment is not known to be sufficient, some cases could still
1378 // be aligned at run time. However, if the requested alignment is less or
1379 // equal to the base alignment and the offset is not aligned, we know that
1380 // the run-time value can never be aligned.
1381 if (BaseAlignment.getQuantity() >= Alignment &&
1382 PtrAlign.getQuantity() < Alignment) {
1383 S.Stk.push<Boolean>(Args: false);
1384 return true;
1385 }
1386
1387 S.FFDiag(E: Call->getArg(Arg: 0), DiagId: diag::note_constexpr_alignment_compute)
1388 << Alignment;
1389 return false;
1390 }
1391
1392 assert(BuiltinOp == Builtin::BI__builtin_align_down ||
1393 BuiltinOp == Builtin::BI__builtin_align_up);
1394
1395 // For align_up/align_down, we can return the same value if the alignment
1396 // is known to be greater or equal to the requested value.
1397 if (PtrAlign.getQuantity() >= Alignment) {
1398 S.Stk.push<Pointer>(Args: Ptr);
1399 return true;
1400 }
1401
1402 // The alignment could be greater than the minimum at run-time, so we cannot
1403 // infer much about the resulting pointer value. One case is possible:
1404 // For `_Alignas(32) char buf[N]; __builtin_align_down(&buf[idx], 32)` we
1405 // can infer the correct index if the requested alignment is smaller than
1406 // the base alignment so we can perform the computation on the offset.
1407 if (BaseAlignment.getQuantity() >= Alignment) {
1408 assert(Alignment.getBitWidth() <= 64 &&
1409 "Cannot handle > 64-bit address-space");
1410 uint64_t Alignment64 = Alignment.getZExtValue();
1411 CharUnits NewOffset =
1412 CharUnits::fromQuantity(Quantity: BuiltinOp == Builtin::BI__builtin_align_down
1413 ? llvm::alignDown(Value: PtrOffset, Align: Alignment64)
1414 : llvm::alignTo(Value: PtrOffset, Align: Alignment64));
1415
1416 if (Ptr.isBlockPointer()) {
1417 S.Stk.push<Pointer>(Args: Ptr.atIndex(Idx: NewOffset.getQuantity()));
1418 return true;
1419 }
1420
1421 assert(Ptr.isOpaquePointer());
1422
1423 APSInt APOffset =
1424 APSInt(APInt(64, NewOffset.getQuantity(), /*IsSigned=*/true),
1425 /*IsUnsigned=*/false);
1426 return arrayElemPtrOpaque(S, OpPC, Ptr, Index: std::move(APOffset),
1427 /*AllocReplace=*/AllowReplace: true);
1428 }
1429
1430 // Otherwise, we cannot constant-evaluate the result.
1431 S.FFDiag(E: Call->getArg(Arg: 0), DiagId: diag::note_constexpr_alignment_adjust) << Alignment;
1432 return false;
1433}
1434
1435/// __builtin_assume_aligned(Ptr, Alignment[, ExtraOffset])
1436static bool interp__builtin_assume_aligned(InterpState &S, CodePtr OpPC,
1437 const InterpFrame *Frame,
1438 const CallExpr *Call) {
1439 assert(Call->getNumArgs() == 2 || Call->getNumArgs() == 3);
1440
1441 std::optional<APSInt> ExtraOffset;
1442 if (Call->getNumArgs() == 3) {
1443 APSInt ExtraOffsetVal;
1444 if (!popToAPSInt(Stk&: S.Stk, T: *S.Ctx.classify(E: Call->getArg(Arg: 2)), Out&: ExtraOffsetVal))
1445 return false;
1446 ExtraOffset = ExtraOffsetVal;
1447 }
1448
1449 APSInt Alignment;
1450 if (!popToAPSInt(Stk&: S.Stk, T: *S.Ctx.classify(E: Call->getArg(Arg: 1)), Out&: Alignment))
1451 return false;
1452 const Pointer &Ptr = S.Stk.pop<Pointer>();
1453
1454 const ASTContext &ASTCtx = S.getASTContext();
1455 CharUnits Align = CharUnits::fromQuantity(Quantity: Alignment.getZExtValue());
1456
1457 // If there is a base object, then it must have the correct alignment.
1458 if (Ptr.isBlockPointer() || Ptr.isOpaquePointer()) {
1459 CharUnits BaseAlignment;
1460 if (Ptr.isBlockPointer() && Ptr.block()->isDynamic())
1461 BaseAlignment = Ptr.getDeclDesc()->computeAlignForDynamicAlloc(Ctx: ASTCtx);
1462 else if (const auto *VD = Ptr.getRootVarDecl())
1463 BaseAlignment = ASTCtx.getDeclAlign(D: VD);
1464 else if (const auto *E = Ptr.getRootExpr())
1465 BaseAlignment = GetAlignOfExpr(Ctx: ASTCtx, E, ExprKind: UETT_AlignOf);
1466
1467 if (BaseAlignment < Align) {
1468 S.CCEDiag(E: Call->getArg(Arg: 0),
1469 DiagId: diag::note_constexpr_baa_insufficient_alignment)
1470 << 0 << BaseAlignment.getQuantity() << Align.getQuantity();
1471 return false;
1472 }
1473 }
1474
1475 std::optional<size_t> LayoutOffset = Ptr.computeLayoutOffset(ASTCtx);
1476 if (!LayoutOffset)
1477 return false;
1478
1479 CharUnits AVOffset = CharUnits::fromQuantity(Quantity: *LayoutOffset);
1480 if (ExtraOffset)
1481 AVOffset -= CharUnits::fromQuantity(Quantity: ExtraOffset->getZExtValue());
1482 if (AVOffset.alignTo(Align) != AVOffset) {
1483 if (Ptr.isBlockPointer() || Ptr.isOpaquePointer())
1484 S.CCEDiag(E: Call->getArg(Arg: 0),
1485 DiagId: diag::note_constexpr_baa_insufficient_alignment)
1486 << 1 << AVOffset.getQuantity() << Align.getQuantity();
1487 else
1488 S.CCEDiag(E: Call->getArg(Arg: 0),
1489 DiagId: diag::note_constexpr_baa_value_insufficient_alignment)
1490 << AVOffset.getQuantity() << Align.getQuantity();
1491 return false;
1492 }
1493
1494 S.Stk.push<Pointer>(Args: Ptr);
1495 return true;
1496}
1497
1498/// (CarryIn, LHS, RHS, Result)
1499static bool interp__builtin_ia32_addcarry_subborrow(InterpState &S,
1500 CodePtr OpPC,
1501 const InterpFrame *Frame,
1502 const CallExpr *Call,
1503 bool IsAdd) {
1504 if (Call->getNumArgs() != 4 || !Call->getArg(Arg: 0)->getType()->isIntegerType() ||
1505 !Call->getArg(Arg: 1)->getType()->isIntegerType() ||
1506 !Call->getArg(Arg: 2)->getType()->isIntegerType())
1507 return false;
1508
1509 const Pointer &CarryOutPtr = S.Stk.pop<Pointer>();
1510
1511 APSInt RHS;
1512 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: RHS))
1513 return false;
1514 APSInt LHS;
1515 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: LHS))
1516 return false;
1517 APSInt CarryIn;
1518 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: CarryIn))
1519 return false;
1520
1521 unsigned BitWidth = LHS.getBitWidth();
1522 unsigned CarryInBit = CarryIn.ugt(RHS: 0) ? 1 : 0;
1523 APInt ExResult =
1524 IsAdd ? (LHS.zext(width: BitWidth + 1) + (RHS.zext(width: BitWidth + 1) + CarryInBit))
1525 : (LHS.zext(width: BitWidth + 1) - (RHS.zext(width: BitWidth + 1) + CarryInBit));
1526
1527 APInt Result = ExResult.extractBits(numBits: BitWidth, bitPosition: 0);
1528 APSInt CarryOut =
1529 APSInt(ExResult.extractBits(numBits: 1, bitPosition: BitWidth), /*IsUnsigned=*/true);
1530
1531 QualType CarryOutType = Call->getArg(Arg: 3)->getType()->getPointeeType();
1532 PrimType CarryOutT = *S.getContext().classify(T: CarryOutType);
1533 assignIntegral(S, Dest: CarryOutPtr, ValueT: CarryOutT, Value: APSInt(std::move(Result), true));
1534
1535 pushInteger(S, Val: CarryOut, QT: Call->getType());
1536
1537 return true;
1538}
1539
1540static bool interp__builtin_os_log_format_buffer_size(InterpState &S,
1541 CodePtr OpPC,
1542 const InterpFrame *Frame,
1543 const CallExpr *Call) {
1544 analyze_os_log::OSLogBufferLayout Layout;
1545 analyze_os_log::computeOSLogBufferLayout(Ctx&: S.getASTContext(), E: Call, layout&: Layout);
1546 pushInteger(S, Val: Layout.size().getQuantity(), QT: Call->getType());
1547 return true;
1548}
1549
1550static bool
1551interp__builtin_ptrauth_string_discriminator(InterpState &S, CodePtr OpPC,
1552 const InterpFrame *Frame,
1553 const CallExpr *Call) {
1554 const auto &Ptr = S.Stk.pop<Pointer>();
1555 if (!Ptr.isStringPointer())
1556 return false;
1557
1558 uint64_t Result = getPointerAuthStableSipHash(
1559 S: cast<StringLiteral>(Val: Ptr.getRootExpr())->getString());
1560 pushInteger(S, Val: Result, QT: Call->getType());
1561 return true;
1562}
1563
1564static bool interp__builtin_infer_alloc_token(InterpState &S, CodePtr OpPC,
1565 const InterpFrame *Frame,
1566 const CallExpr *Call) {
1567 const ASTContext &ASTCtx = S.getASTContext();
1568 uint64_t BitWidth = ASTCtx.getTypeSize(T: ASTCtx.getSizeType());
1569 auto Mode =
1570 ASTCtx.getLangOpts().AllocTokenMode.value_or(u: llvm::DefaultAllocTokenMode);
1571 auto MaxTokensOpt = ASTCtx.getLangOpts().AllocTokenMax;
1572 uint64_t MaxTokens =
1573 MaxTokensOpt.value_or(u: 0) ? *MaxTokensOpt : (~0ULL >> (64 - BitWidth));
1574
1575 // We do not read any of the arguments; discard them.
1576 for (int I = Call->getNumArgs() - 1; I >= 0; --I)
1577 discard(Stk&: S.Stk, T: S.getContext().classify(E: Call->getArg(Arg: I)).value_or(PT: PT_Ptr));
1578
1579 // Note: Type inference from a surrounding cast is not supported in
1580 // constexpr evaluation.
1581 QualType AllocType = infer_alloc::inferPossibleType(E: Call, Ctx: ASTCtx, CastE: nullptr);
1582 if (AllocType.isNull()) {
1583 S.CCEDiag(E: Call,
1584 DiagId: diag::note_constexpr_infer_alloc_token_type_inference_failed);
1585 return false;
1586 }
1587
1588 auto ATMD = infer_alloc::getAllocTokenMetadata(T: AllocType, Ctx: ASTCtx);
1589 if (!ATMD) {
1590 S.CCEDiag(E: Call, DiagId: diag::note_constexpr_infer_alloc_token_no_metadata);
1591 return false;
1592 }
1593
1594 auto MaybeToken = llvm::getAllocToken(Mode, Metadata: *ATMD, MaxTokens);
1595 if (!MaybeToken) {
1596 S.CCEDiag(E: Call, DiagId: diag::note_constexpr_infer_alloc_token_stateful_mode);
1597 return false;
1598 }
1599
1600 pushInteger(S, Val: llvm::APInt(BitWidth, *MaybeToken), QT: ASTCtx.getSizeType());
1601 return true;
1602}
1603
1604static bool interp__builtin_operator_new(InterpState &S, CodePtr OpPC,
1605 const InterpFrame *Frame,
1606 const CallExpr *Call) {
1607 // A call to __operator_new is only valid within std::allocate<>::allocate.
1608 // Walk up the call stack to find the appropriate caller and get the
1609 // element type from it.
1610 auto [NewCall, ElemType] = S.getStdAllocatorCaller(Name: "allocate");
1611
1612 if (ElemType.isNull()) {
1613 S.FFDiag(E: Call, DiagId: S.getLangOpts().CPlusPlus20
1614 ? diag::note_constexpr_new_untyped
1615 : diag::note_constexpr_new);
1616 return false;
1617 }
1618 assert(NewCall);
1619
1620 if (ElemType->isIncompleteType() || ElemType->isFunctionType()) {
1621 S.FFDiag(E: Call, DiagId: diag::note_constexpr_new_not_complete_object_type)
1622 << (ElemType->isIncompleteType() ? 0 : 1) << ElemType;
1623 return false;
1624 }
1625
1626 // We only care about the first parameter (the size), so discard all the
1627 // others.
1628 {
1629 unsigned NumArgs = Call->getNumArgs();
1630 assert(NumArgs >= 1);
1631
1632 // The std::nothrow_t arg never gets put on the stack.
1633 if (Call->getArg(Arg: NumArgs - 1)->getType()->isNothrowT())
1634 --NumArgs;
1635 auto Args = ArrayRef(Call->getArgs(), Call->getNumArgs());
1636 // First arg is needed.
1637 Args = Args.drop_front();
1638
1639 // Discard the rest.
1640 for (const Expr *Arg : Args)
1641 discard(Stk&: S.Stk, T: *S.getContext().classify(E: Arg));
1642 }
1643
1644 APSInt Bytes;
1645 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Bytes))
1646 return false;
1647 CharUnits ElemSize = S.getASTContext().getTypeSizeInChars(T: ElemType);
1648 assert(!ElemSize.isZero());
1649 // Divide the number of bytes by sizeof(ElemType), so we get the number of
1650 // elements we should allocate.
1651 APInt NumElems, Remainder;
1652 APInt ElemSizeAP(Bytes.getBitWidth(), ElemSize.getQuantity());
1653 APInt::udivrem(LHS: Bytes, RHS: ElemSizeAP, Quotient&: NumElems, Remainder);
1654 if (Remainder != 0) {
1655 // This likely indicates a bug in the implementation of 'std::allocator'.
1656 S.FFDiag(E: Call, DiagId: diag::note_constexpr_operator_new_bad_size)
1657 << Bytes << APSInt(ElemSizeAP, true) << ElemType;
1658 return false;
1659 }
1660
1661 // NB: The same check we're using in CheckArraySize()
1662 if (NumElems.getActiveBits() >
1663 ConstantArrayType::getMaxSizeBits(Context: S.getASTContext()) ||
1664 NumElems.ugt(RHS: Descriptor::MaxArrayElemBytes / ElemSize.getQuantity())) {
1665 // FIXME: NoThrow check?
1666 const SourceInfo &Loc = S.Current->getSource(PC: OpPC);
1667 S.FFDiag(SI: Loc, DiagId: diag::note_constexpr_new_too_large)
1668 << NumElems.getZExtValue();
1669 return false;
1670 }
1671
1672 if (!CheckArraySize(S, OpPC, NumElems: NumElems.getZExtValue()))
1673 return false;
1674
1675 bool IsArray = NumElems.ugt(RHS: 1);
1676 OptPrimType ElemT = S.getContext().classify(T: ElemType);
1677 DynamicAllocator &Allocator = S.getAllocator();
1678 if (ElemT) {
1679 Block *B =
1680 Allocator.allocate(Source: NewCall, T: *ElemT, NumElements: NumElems.getZExtValue(),
1681 EvalID: S.Ctx.getEvalID(), AllocForm: DynamicAllocator::Form::Operator);
1682 assert(B);
1683 S.Stk.push<Pointer>(Args: Pointer(B).atIndex(Idx: 0));
1684 return true;
1685 }
1686
1687 assert(!ElemT);
1688
1689 // Composite arrays
1690 if (IsArray) {
1691 const Descriptor *Desc =
1692 S.P.createDescriptor(D: NewCall, Ty: ElemType.getTypePtr());
1693 Block *B =
1694 Allocator.allocate(D: Desc, NumElements: NumElems.getZExtValue(), EvalID: S.Ctx.getEvalID(),
1695 AllocForm: DynamicAllocator::Form::Operator);
1696 assert(B);
1697 S.Stk.push<Pointer>(Args: Pointer(B).atIndex(Idx: 0).narrow());
1698 return true;
1699 }
1700
1701 // Records. Still allocate them as single-element arrays.
1702 QualType AllocType = S.getASTContext().getConstantArrayType(
1703 EltTy: ElemType, ArySize: NumElems, SizeExpr: nullptr, ASM: ArraySizeModifier::Normal, IndexTypeQuals: 0);
1704
1705 const Descriptor *Desc =
1706 S.P.createDescriptor(D: NewCall, Ty: AllocType.getTypePtr());
1707 Block *B = Allocator.allocate(D: Desc, EvalID: S.getContext().getEvalID(),
1708 AllocForm: DynamicAllocator::Form::Operator);
1709 assert(B);
1710 S.Stk.push<Pointer>(Args: Pointer(B).atIndex(Idx: 0).narrow());
1711 return true;
1712}
1713
1714static bool interp__builtin_operator_delete(InterpState &S, CodePtr OpPC,
1715 const InterpFrame *Frame,
1716 const CallExpr *Call) {
1717 const Expr *Source = nullptr;
1718 const Block *BlockToDelete = nullptr;
1719
1720 unsigned NumArgs = Call->getNumArgs();
1721 assert(NumArgs >= 1);
1722
1723 // Args are pushed in source order. The trailing sized/aligned delete
1724 // operands are above the pointer on the stack.
1725 for (unsigned I = NumArgs - 1; I != 0; --I)
1726 discard(Stk&: S.Stk, T: *S.getContext().classify(E: Call->getArg(Arg: I)));
1727
1728 if (S.checkingPotentialConstantExpression()) {
1729 S.Stk.discard<Pointer>();
1730 return false;
1731 }
1732
1733 // This is permitted only within a call to std::allocator<T>::deallocate.
1734 if (!S.getStdAllocatorCaller(Name: "deallocate")) {
1735 S.FFDiag(E: Call);
1736 S.Stk.discard<Pointer>();
1737 return true;
1738 }
1739
1740 {
1741 const Pointer &Ptr = S.Stk.pop<Pointer>();
1742
1743 if (Ptr.isZero()) {
1744 S.CCEDiag(E: Call, DiagId: diag::note_constexpr_deallocate_null);
1745 return true;
1746 }
1747
1748 Source = Ptr.getRootExpr();
1749 BlockToDelete = Ptr.block();
1750
1751 if (!BlockToDelete->isDynamic()) {
1752 S.FFDiag(E: Call, DiagId: diag::note_constexpr_delete_not_heap_alloc)
1753 << Ptr.toDiagnosticString(Ctx: S.getASTContext());
1754 if (const auto *D = Ptr.getFieldDesc()->asDecl())
1755 S.Note(Loc: D->getLocation(), DiagId: diag::note_declared_at);
1756 }
1757 }
1758 assert(BlockToDelete);
1759
1760 DynamicAllocator &Allocator = S.getAllocator();
1761 const Descriptor *BlockDesc = BlockToDelete->getDescriptor();
1762 std::optional<DynamicAllocator::Form> AllocForm =
1763 Allocator.getAllocationForm(Source);
1764
1765 if (!Allocator.deallocate(Source, BlockToDelete)) {
1766 // Nothing has been deallocated, this must be a double-delete.
1767 const SourceInfo &Loc = S.Current->getSource(PC: OpPC);
1768 S.FFDiag(SI: Loc, DiagId: diag::note_constexpr_double_delete);
1769 return false;
1770 }
1771 assert(AllocForm);
1772
1773 return CheckNewDeleteForms(
1774 S, OpPC, AllocForm: *AllocForm, DeleteForm: DynamicAllocator::Form::Operator, D: BlockDesc, NewExpr: Source);
1775}
1776
1777static bool interp__builtin_arithmetic_fence(InterpState &S, CodePtr OpPC,
1778 const InterpFrame *Frame,
1779 const CallExpr *Call) {
1780 const Floating &Arg0 = S.Stk.pop<Floating>();
1781 S.Stk.push<Floating>(Args: Arg0);
1782 return true;
1783}
1784
1785static bool interp__builtin_vector_reduce(InterpState &S, CodePtr OpPC,
1786 const CallExpr *Call, unsigned ID) {
1787 const Pointer &Arg = S.Stk.pop<Pointer>();
1788 assert(Arg.getFieldDesc()->isPrimitiveArray());
1789
1790 QualType ElemType = Arg.getFieldDesc()->getElemQualType();
1791 assert(Call->getType() == ElemType);
1792 PrimType ElemT = *S.getContext().classify(T: ElemType);
1793 unsigned NumElems = Arg.getNumElems();
1794
1795 if (!isIntegerType(T: ElemT))
1796 return false;
1797
1798 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
1799 T Result = Arg.elem<T>(0);
1800 unsigned BitWidth = Result.bitWidth();
1801 for (unsigned I = 1; I != NumElems; ++I) {
1802 T Elem = Arg.elem<T>(I);
1803 T PrevResult = Result;
1804
1805 if (ID == Builtin::BI__builtin_reduce_add) {
1806 if (T::add(Result, Elem, BitWidth, &Result)) {
1807 unsigned OverflowBits = BitWidth + 1;
1808 (void)handleOverflow(S, OpPC,
1809 (PrevResult.toAPSInt(OverflowBits) +
1810 Elem.toAPSInt(OverflowBits)));
1811 return false;
1812 }
1813 } else if (ID == Builtin::BI__builtin_reduce_mul) {
1814 if (T::mul(Result, Elem, BitWidth, &Result)) {
1815 unsigned OverflowBits = BitWidth * 2;
1816 (void)handleOverflow(S, OpPC,
1817 (PrevResult.toAPSInt(OverflowBits) *
1818 Elem.toAPSInt(OverflowBits)));
1819 return false;
1820 }
1821
1822 } else if (ID == Builtin::BI__builtin_reduce_and) {
1823 (void)T::bitAnd(Result, Elem, BitWidth, &Result);
1824 } else if (ID == Builtin::BI__builtin_reduce_or) {
1825 (void)T::bitOr(Result, Elem, BitWidth, &Result);
1826 } else if (ID == Builtin::BI__builtin_reduce_xor) {
1827 (void)T::bitXor(Result, Elem, BitWidth, &Result);
1828 } else if (ID == Builtin::BI__builtin_reduce_min) {
1829 if (Elem < Result)
1830 Result = Elem;
1831 } else if (ID == Builtin::BI__builtin_reduce_max) {
1832 if (Elem > Result)
1833 Result = Elem;
1834 } else {
1835 llvm_unreachable("Unhandled vector reduce builtin");
1836 }
1837 }
1838 pushInteger(S, Result.toAPSInt(), Call->getType());
1839 });
1840
1841 return true;
1842}
1843
1844static bool interp__builtin_elementwise_abs(InterpState &S, CodePtr OpPC,
1845 const InterpFrame *Frame,
1846 const CallExpr *Call,
1847 unsigned BuiltinID) {
1848 assert(Call->getNumArgs() == 1);
1849 QualType Ty = Call->getArg(Arg: 0)->getType();
1850 if (Ty->isIntegerType()) {
1851 APSInt Val;
1852 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Val))
1853 return false;
1854 pushInteger(S, Val: Val.abs(), QT: Call->getType());
1855 return true;
1856 }
1857
1858 if (Ty->isFloatingType()) {
1859 Floating Val = S.Stk.pop<Floating>();
1860 Floating Result = abs(S, In: Val);
1861 S.Stk.push<Floating>(Args&: Result);
1862 return true;
1863 }
1864
1865 // Otherwise, the argument must be a vector.
1866 assert(Call->getArg(0)->getType()->isVectorType());
1867 const Pointer &Arg = S.Stk.pop<Pointer>();
1868 assert(Arg.getFieldDesc()->isPrimitiveArray());
1869 const Pointer &Dst = S.Stk.peek<Pointer>();
1870 assert(Dst.getFieldDesc()->isPrimitiveArray());
1871 assert(Arg.getFieldDesc()->getNumElems() ==
1872 Dst.getFieldDesc()->getNumElems());
1873
1874 QualType ElemType = Arg.getFieldDesc()->getElemQualType();
1875 PrimType ElemT = *S.getContext().classify(T: ElemType);
1876 unsigned NumElems = Arg.getNumElems();
1877 // we can either have a vector of integer or a vector of floating point
1878 for (unsigned I = 0; I != NumElems; ++I) {
1879 if (ElemType->isIntegerType()) {
1880 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
1881 Dst.elem<T>(I) = T::from(static_cast<T>(
1882 APSInt(Arg.elem<T>(I).toAPSInt().abs(),
1883 ElemType->isUnsignedIntegerOrEnumerationType())));
1884 });
1885 } else {
1886 Floating Val = Arg.elem<Floating>(I);
1887 Dst.elem<Floating>(I) = abs(S, In: Val);
1888 }
1889 }
1890 Dst.initializeAllElements();
1891
1892 return true;
1893}
1894
1895/// Can be called with an integer or vector as the first and only parameter.
1896static bool interp__builtin_elementwise_countzeroes(InterpState &S,
1897 CodePtr OpPC,
1898 const InterpFrame *Frame,
1899 const CallExpr *Call,
1900 unsigned BuiltinID) {
1901 bool HasZeroArg = Call->getNumArgs() == 2;
1902 bool IsCTTZ = BuiltinID == Builtin::BI__builtin_elementwise_ctzg;
1903 assert(Call->getNumArgs() == 1 || HasZeroArg);
1904 if (Call->getArg(Arg: 0)->getType()->isIntegerType()) {
1905 PrimType ArgT = *S.getContext().classify(T: Call->getArg(Arg: 0)->getType());
1906 APSInt Val;
1907 if (!popToAPSInt(Stk&: S.Stk, T: ArgT, Out&: Val))
1908 return false;
1909 std::optional<APSInt> ZeroVal;
1910 if (HasZeroArg) {
1911 ZeroVal = Val;
1912 if (!popToAPSInt(Stk&: S.Stk, T: ArgT, Out&: Val))
1913 return false;
1914 }
1915
1916 if (Val.isZero()) {
1917 if (ZeroVal) {
1918 pushInteger(S, Val: *ZeroVal, QT: Call->getType());
1919 return true;
1920 }
1921 // If we haven't been provided the second argument, the result is
1922 // undefined
1923 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
1924 DiagId: diag::note_constexpr_countzeroes_zero)
1925 << /*IsTrailing=*/IsCTTZ;
1926 return false;
1927 }
1928
1929 if (BuiltinID == Builtin::BI__builtin_elementwise_clzg) {
1930 pushInteger(S, Val: Val.countLeadingZeros(), QT: Call->getType());
1931 } else {
1932 pushInteger(S, Val: Val.countTrailingZeros(), QT: Call->getType());
1933 }
1934 return true;
1935 }
1936 // Otherwise, the argument must be a vector.
1937 const ASTContext &ASTCtx = S.getASTContext();
1938 Pointer ZeroArg;
1939 if (HasZeroArg) {
1940 assert(Call->getArg(1)->getType()->isVectorType() &&
1941 ASTCtx.hasSameUnqualifiedType(Call->getArg(0)->getType(),
1942 Call->getArg(1)->getType()));
1943 (void)ASTCtx;
1944 ZeroArg = S.Stk.pop<Pointer>();
1945 assert(ZeroArg.getFieldDesc()->isPrimitiveArray());
1946 }
1947 assert(Call->getArg(0)->getType()->isVectorType());
1948 const Pointer &Arg = S.Stk.pop<Pointer>();
1949 assert(Arg.getFieldDesc()->isPrimitiveArray());
1950 const Pointer &Dst = S.Stk.peek<Pointer>();
1951 assert(Dst.getFieldDesc()->isPrimitiveArray());
1952 assert(Arg.getFieldDesc()->getNumElems() ==
1953 Dst.getFieldDesc()->getNumElems());
1954
1955 QualType ElemType = Arg.getFieldDesc()->getElemQualType();
1956 PrimType ElemT = *S.getContext().classify(T: ElemType);
1957 unsigned NumElems = Arg.getNumElems();
1958
1959 // FIXME: Reading from uninitialized vector elements?
1960 for (unsigned I = 0; I != NumElems; ++I) {
1961 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
1962 APInt EltVal = Arg.atIndex(I).deref<T>().toAPSInt();
1963 if (EltVal.isZero()) {
1964 if (HasZeroArg) {
1965 Dst.atIndex(I).deref<T>() = ZeroArg.atIndex(I).deref<T>();
1966 } else {
1967 // If we haven't been provided the second argument, the result is
1968 // undefined
1969 S.FFDiag(S.Current->getSource(OpPC),
1970 diag::note_constexpr_countzeroes_zero)
1971 << /*IsTrailing=*/IsCTTZ;
1972 return false;
1973 }
1974 } else if (IsCTTZ) {
1975 Dst.atIndex(I).deref<T>() = T::from(EltVal.countTrailingZeros());
1976 } else {
1977 Dst.atIndex(I).deref<T>() = T::from(EltVal.countLeadingZeros());
1978 }
1979 Dst.atIndex(I).initialize();
1980 });
1981 }
1982
1983 return true;
1984}
1985
1986static bool interp__builtin_memcpy(InterpState &S, CodePtr OpPC,
1987 const InterpFrame *Frame,
1988 const CallExpr *Call, unsigned ID) {
1989 assert(Call->getNumArgs() == 3);
1990 const ASTContext &ASTCtx = S.getASTContext();
1991 uint64_t Size;
1992 if (!popToUInt64(S, E: Call->getArg(Arg: 2), Out&: Size))
1993 return false;
1994 Pointer SrcPtr = S.Stk.pop<Pointer>().expand();
1995 Pointer DestPtr = S.Stk.pop<Pointer>().expand();
1996
1997 if (ID == Builtin::BImemcpy || ID == Builtin::BImemmove)
1998 diagnoseNonConstexprBuiltin(S, OpPC, ID);
1999
2000 bool Move =
2001 (ID == Builtin::BI__builtin_memmove || ID == Builtin::BImemmove ||
2002 ID == Builtin::BI__builtin_wmemmove || ID == Builtin::BIwmemmove);
2003 bool WChar = ID == Builtin::BIwmemcpy || ID == Builtin::BIwmemmove ||
2004 ID == Builtin::BI__builtin_wmemcpy ||
2005 ID == Builtin::BI__builtin_wmemmove;
2006
2007 // If the size is zero, we treat this as always being a valid no-op.
2008 if (Size == 0) {
2009 S.Stk.push<Pointer>(Args&: DestPtr);
2010 return true;
2011 }
2012
2013 if (SrcPtr.isZero() || DestPtr.isZero()) {
2014 Pointer DiagPtr = (SrcPtr.isZero() ? SrcPtr : DestPtr);
2015 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_memcpy_null)
2016 << /*IsMove=*/Move << /*IsWchar=*/WChar << !SrcPtr.isZero()
2017 << DiagPtr.toDiagnosticString(Ctx: ASTCtx);
2018 return false;
2019 }
2020
2021 // Diagnose integral src/dest pointers specially.
2022 if (SrcPtr.isIntegralPointer() || DestPtr.isIntegralPointer()) {
2023 std::string DiagVal = "(void *)";
2024 DiagVal += SrcPtr.isIntegralPointer()
2025 ? std::to_string(val: SrcPtr.getIntegerRepresentation())
2026 : std::to_string(val: DestPtr.getIntegerRepresentation());
2027 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_memcpy_null)
2028 << Move << WChar << DestPtr.isIntegralPointer() << DiagVal;
2029 return false;
2030 }
2031
2032 if (!isReadable(P: DestPtr) || !isReadable(P: SrcPtr))
2033 return false;
2034
2035 if (DestPtr.getType()->isIncompleteType()) {
2036 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
2037 DiagId: diag::note_constexpr_memcpy_incomplete_type)
2038 << Move << DestPtr.getType();
2039 return false;
2040 }
2041 if (SrcPtr.getType()->isIncompleteType()) {
2042 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
2043 DiagId: diag::note_constexpr_memcpy_incomplete_type)
2044 << Move << SrcPtr.getType();
2045 return false;
2046 }
2047
2048 QualType DestElemType = getElemType(P: DestPtr);
2049 if (DestElemType->isIncompleteType()) {
2050 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
2051 DiagId: diag::note_constexpr_memcpy_incomplete_type)
2052 << Move << DestElemType;
2053 return false;
2054 }
2055
2056 size_t RemainingDestElems;
2057 if (DestPtr.inArray()) {
2058 RemainingDestElems = DestPtr.isUnknownSizeArray()
2059 ? 0
2060 : (DestPtr.getNumElems() - DestPtr.getIndex());
2061 } else {
2062 RemainingDestElems = 1;
2063 }
2064 unsigned DestElemSize = ASTCtx.getTypeSizeInChars(T: DestElemType).getQuantity();
2065
2066 if (WChar) {
2067 uint64_t WCharSize =
2068 ASTCtx.getTypeSizeInChars(T: ASTCtx.getWCharType()).getQuantity();
2069 Size *= WCharSize;
2070 }
2071
2072 if (Size % DestElemSize != 0) {
2073 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
2074 DiagId: diag::note_constexpr_memcpy_unsupported)
2075 << Move << WChar << 0 << DestElemType << Size << DestElemSize;
2076 return false;
2077 }
2078
2079 QualType SrcElemType = getElemType(P: SrcPtr);
2080 size_t RemainingSrcElems;
2081 if (SrcPtr.inArray()) {
2082 RemainingSrcElems = SrcPtr.isUnknownSizeArray()
2083 ? 0
2084 : (SrcPtr.getNumElems() - SrcPtr.getIndex());
2085 } else {
2086 RemainingSrcElems = 1;
2087 }
2088 unsigned SrcElemSize = ASTCtx.getTypeSizeInChars(T: SrcElemType).getQuantity();
2089
2090 if (!ASTCtx.hasSameUnqualifiedType(T1: DestElemType, T2: SrcElemType)) {
2091 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_memcpy_type_pun)
2092 << Move << SrcElemType << DestElemType;
2093 return false;
2094 }
2095
2096 if (!DestElemType.isTriviallyCopyableType(Context: ASTCtx)) {
2097 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_memcpy_nontrivial)
2098 << Move << DestElemType;
2099 return false;
2100 }
2101
2102 // Check if we have enough elements to read from and write to.
2103 size_t RemainingDestBytes = RemainingDestElems * DestElemSize;
2104 size_t RemainingSrcBytes = RemainingSrcElems * SrcElemSize;
2105 if (Size > RemainingDestBytes || Size > RemainingSrcBytes) {
2106 APInt N = APInt(64, Size / DestElemSize);
2107 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
2108 DiagId: diag::note_constexpr_memcpy_unsupported)
2109 << Move << WChar << (Size > RemainingSrcBytes ? 1 : 2) << DestElemType
2110 << toString(I: N, Radix: 10, /*Signed=*/false);
2111 return false;
2112 }
2113
2114 // Check for overlapping memory regions.
2115 if (!Move && Pointer::pointToSameBlock(A: SrcPtr, B: DestPtr)) {
2116 // Remove base casts.
2117 Pointer SrcP = SrcPtr.stripBaseCasts();
2118 Pointer DestP = DestPtr.stripBaseCasts();
2119
2120 unsigned SrcIndex = SrcP.expand().getIndex() * SrcElemSize;
2121 unsigned DstIndex = DestP.expand().getIndex() * DestElemSize;
2122
2123 if ((SrcIndex <= DstIndex && (SrcIndex + Size) > DstIndex) ||
2124 (DstIndex <= SrcIndex && (DstIndex + Size) > SrcIndex)) {
2125 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_memcpy_overlap)
2126 << /*IsWChar=*/false;
2127 return false;
2128 }
2129 }
2130
2131 assert(Size % DestElemSize == 0);
2132 if (!DoMemcpy(S, OpPC, SrcPtr, DestPtr, Size: Bytes(Size).toBits()))
2133 return false;
2134
2135 S.Stk.push<Pointer>(Args&: DestPtr);
2136 return true;
2137}
2138
2139/// Determine if T is a character type for which we guarantee that
2140/// sizeof(T) == 1.
2141static bool isOneByteCharacterType(QualType T) {
2142 return T->isCharType() || T->isChar8Type();
2143}
2144
2145// stdc_memreverse8(size_t N, unsigned char *P)
2146static bool interp__builtin_stdc_memreverse8(InterpState &S, CodePtr OpPC,
2147 const InterpFrame *Frame,
2148 const CallExpr *Call) {
2149 Pointer Ptr = S.Stk.pop<Pointer>();
2150
2151 uint64_t NElems;
2152 if (!popToUInt64(S, E: Call->getArg(Arg: 0), Out&: NElems))
2153 return false;
2154
2155 if (Ptr.isZero()) {
2156 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_access_null)
2157 << AK_Assign;
2158 return false;
2159 }
2160
2161 if (!isReadable(P: Ptr) && !Ptr.isOnePastEnd())
2162 return false;
2163
2164 const Descriptor *Desc = Ptr.getFieldDesc();
2165 bool IsArray = Desc->isArray();
2166 QualType ElemTy = IsArray ? Desc->getElemQualType() : Desc->getType();
2167
2168 if (IsArray)
2169 Ptr = Ptr.expand();
2170
2171 uint64_t BaseIdx = Ptr.getIndex();
2172 uint64_t ArraySize = Ptr.getNumElems();
2173 uint64_t RemainingElems = ArraySize - BaseIdx;
2174 if (NElems > RemainingElems) {
2175 uint64_t LastIndex = llvm::SaturatingAdd(X: BaseIdx, Y: NElems - 1);
2176 if (IsArray)
2177 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_array_index)
2178 << LastIndex << /*array*/ 0 << ArraySize;
2179 else
2180 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_array_index)
2181 << LastIndex << /*non-array*/ 1;
2182 return false;
2183 }
2184
2185 if (NElems <= 1)
2186 return true;
2187
2188 PrimType ElemT = *S.getContext().classify(T: ElemTy);
2189
2190 for (uint64_t I = 0, Half = NElems / 2; I < Half; ++I) {
2191 Pointer LoPtr = Ptr.atIndex(Idx: BaseIdx + I);
2192 Pointer HiPtr = Ptr.atIndex(Idx: BaseIdx + NElems - 1 - I);
2193
2194 if (!CheckLoad(S, OpPC, Ptr: LoPtr, AK: AK_Read) ||
2195 !CheckLoad(S, OpPC, Ptr: HiPtr, AK: AK_Read) || !CheckStore(S, OpPC, Ptr: LoPtr) ||
2196 !CheckStore(S, OpPC, Ptr: HiPtr))
2197 return false;
2198
2199 INT_TYPE_SWITCH_NO_BOOL(ElemT,
2200 { std::swap(LoPtr.deref<T>(), HiPtr.deref<T>()); });
2201 LoPtr.initialize();
2202 HiPtr.initialize();
2203 }
2204 return true;
2205}
2206
2207static bool interp__builtin_memcmp(InterpState &S, CodePtr OpPC,
2208 const InterpFrame *Frame,
2209 const CallExpr *Call, unsigned ID) {
2210 assert(Call->getNumArgs() == 3);
2211 uint64_t Size;
2212 if (!popToUInt64(S, E: Call->getArg(Arg: 2), Out&: Size))
2213 return false;
2214 const Pointer &PtrB = S.Stk.pop<Pointer>();
2215 const Pointer &PtrA = S.Stk.pop<Pointer>();
2216
2217 if (ID == Builtin::BImemcmp || ID == Builtin::BIbcmp ||
2218 ID == Builtin::BIwmemcmp)
2219 diagnoseNonConstexprBuiltin(S, OpPC, ID);
2220
2221 if (Size == 0) {
2222 pushInteger(S, Val: 0, QT: Call->getType());
2223 return true;
2224 }
2225 bool IsWide =
2226 (ID == Builtin::BIwmemcmp || ID == Builtin::BI__builtin_wmemcmp);
2227
2228 const ASTContext &ASTCtx = S.getASTContext();
2229 QualType ElemTypeA = getElemType(P: PtrA);
2230 QualType ElemTypeB = getElemType(P: PtrB);
2231 // FIXME: This is an arbitrary limitation the current constant interpreter
2232 // had. We could remove this.
2233 if (!IsWide && (!isOneByteCharacterType(T: ElemTypeA) ||
2234 !isOneByteCharacterType(T: ElemTypeB))) {
2235 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
2236 DiagId: diag::note_constexpr_memcmp_unsupported)
2237 << ASTCtx.BuiltinInfo.getQuotedName(ID) << PtrA.getType()
2238 << PtrB.getType();
2239 return false;
2240 }
2241
2242 if (!PtrA.isReadablePointerType() || !PtrB.isReadablePointerType())
2243 return false;
2244
2245 if (!CheckLoad(S, OpPC, Ptr: PtrA, AK: AK_Read) || !CheckLoad(S, OpPC, Ptr: PtrB, AK: AK_Read))
2246 return false;
2247
2248 // Now, read both pointers to a buffer and compare those.
2249 BitcastBuffer BufferA(
2250 Bits(ASTCtx.getTypeSize(T: ElemTypeA) * PtrA.getNumElems()));
2251 readPointerToBuffer(Ctx: S.getContext(), FromPtr: PtrA, Buffer&: BufferA, /*ReturnOnUninit=*/false);
2252
2253 // FIXME: The swapping here is UNDOING something we do when reading the
2254 // data into the buffer.
2255 if (ASTCtx.getTargetInfo().isBigEndian())
2256 swapBytes(M: BufferA.Data.get(), N: BufferA.byteSize().getQuantity());
2257
2258 BitcastBuffer BufferB(
2259 Bits(ASTCtx.getTypeSize(T: ElemTypeB) * PtrB.getNumElems()));
2260 readPointerToBuffer(Ctx: S.getContext(), FromPtr: PtrB, Buffer&: BufferB, /*ReturnOnUninit=*/false);
2261 // FIXME: The swapping here is UNDOING something we do when reading the
2262 // data into the buffer.
2263 if (ASTCtx.getTargetInfo().isBigEndian())
2264 swapBytes(M: BufferB.Data.get(), N: BufferB.byteSize().getQuantity());
2265
2266 size_t MinBufferSize = std::min(a: BufferA.byteSize().getQuantity(),
2267 b: BufferB.byteSize().getQuantity());
2268
2269 unsigned ElemSize = 1;
2270 if (IsWide)
2271 ElemSize = ASTCtx.getTypeSizeInChars(T: ASTCtx.getWCharType()).getQuantity();
2272 // The Size given for the wide variants is in wide-char units. Convert it
2273 // to bytes.
2274 size_t ByteSize = Size * ElemSize;
2275 size_t CmpSize = std::min(a: MinBufferSize, b: ByteSize);
2276
2277 for (size_t I = 0; I != CmpSize; I += ElemSize) {
2278 if (IsWide) {
2279 FIXED_SIZE_INT_TYPE_SWITCH(
2280 *S.getContext().classify(ASTCtx.getWCharType()), {
2281 T A = T::bitcastFromMemory(BufferA.atByte(I), T::bitWidth());
2282 T B = T::bitcastFromMemory(BufferB.atByte(I), T::bitWidth());
2283 if (A < B) {
2284 pushInteger(S, -1, Call->getType());
2285 return true;
2286 }
2287 if (A > B) {
2288 pushInteger(S, 1, Call->getType());
2289 return true;
2290 }
2291 });
2292 } else {
2293 auto A = BufferA.deref<std::byte>(Offset: Bytes(I));
2294 auto B = BufferB.deref<std::byte>(Offset: Bytes(I));
2295
2296 if (A < B) {
2297 pushInteger(S, Val: -1, QT: Call->getType());
2298 return true;
2299 }
2300 if (A > B) {
2301 pushInteger(S, Val: 1, QT: Call->getType());
2302 return true;
2303 }
2304 }
2305 }
2306
2307 // We compared CmpSize bytes above. If the limiting factor was the Size
2308 // passed, we're done and the result is equality (0).
2309 if (ByteSize <= CmpSize) {
2310 pushInteger(S, Val: 0, QT: Call->getType());
2311 return true;
2312 }
2313
2314 // However, if we read all the available bytes but were instructed to read
2315 // even more, diagnose this as a "read of dereferenced one-past-the-end
2316 // pointer". This is what would happen if we called CheckLoad() on every array
2317 // element.
2318 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_access_past_end)
2319 << AK_Read << S.Current->getRange(PC: OpPC);
2320 return false;
2321}
2322
2323// __builtin_memchr(ptr, int, int)
2324// __builtin_strchr(ptr, int)
2325static bool interp__builtin_memchr(InterpState &S, CodePtr OpPC,
2326 const CallExpr *Call, unsigned ID) {
2327 if (ID == Builtin::BImemchr || ID == Builtin::BIwcschr ||
2328 ID == Builtin::BIstrchr || ID == Builtin::BIwmemchr)
2329 diagnoseNonConstexprBuiltin(S, OpPC, ID);
2330
2331 std::optional<APSInt> MaxLength;
2332 if (Call->getNumArgs() == 3) {
2333 APSInt MaxLengthVal;
2334 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: MaxLengthVal))
2335 return false;
2336 MaxLength = MaxLengthVal;
2337 }
2338
2339 APSInt Desired;
2340 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: Desired))
2341 return false;
2342 const Pointer &Ptr = S.Stk.pop<Pointer>();
2343
2344 if (MaxLength && MaxLength->isZero()) {
2345 S.Stk.push<Pointer>();
2346 return true;
2347 }
2348
2349 if (Ptr.isDummy()) {
2350 if (Ptr.getType()->isIncompleteType())
2351 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
2352 DiagId: diag::note_constexpr_ltor_incomplete_type)
2353 << Ptr.getType();
2354 return false;
2355 }
2356
2357 // Null is only okay if the given size is 0.
2358 if (Ptr.isZero()) {
2359 S.FFDiag(SI: S.Current->getSource(PC: OpPC), DiagId: diag::note_constexpr_access_null)
2360 << AK_Read;
2361 return false;
2362 }
2363
2364 if (!Ptr.isReadablePointerType())
2365 return false;
2366
2367 QualType ElemTy = getElemType(P: Ptr);
2368 bool IsRawByte = ID == Builtin::BImemchr || ID == Builtin::BI__builtin_memchr;
2369
2370 // Give up on byte-oriented matching against multibyte elements.
2371 if (IsRawByte && !isOneByteCharacterType(T: ElemTy)) {
2372 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
2373 DiagId: diag::note_constexpr_memchr_unsupported)
2374 << S.getASTContext().BuiltinInfo.getQuotedName(ID) << ElemTy;
2375 return false;
2376 }
2377
2378 if (!isReadable(P: Ptr))
2379 return false;
2380
2381 if (ID == Builtin::BIstrchr || ID == Builtin::BI__builtin_strchr) {
2382 int64_t DesiredTrunc;
2383 if (S.getASTContext().CharTy->isSignedIntegerType())
2384 DesiredTrunc =
2385 Desired.trunc(width: S.getASTContext().getCharWidth()).getSExtValue();
2386 else
2387 DesiredTrunc =
2388 Desired.trunc(width: S.getASTContext().getCharWidth()).getZExtValue();
2389 // strchr compares directly to the passed integer, and therefore
2390 // always fails if given an int that is not a char.
2391 if (Desired != DesiredTrunc) {
2392 S.Stk.push<Pointer>();
2393 return true;
2394 }
2395 }
2396
2397 uint64_t DesiredVal;
2398 if (ID == Builtin::BIwmemchr || ID == Builtin::BI__builtin_wmemchr ||
2399 ID == Builtin::BIwcschr || ID == Builtin::BI__builtin_wcschr) {
2400 // wcschr and wmemchr are given a wchar_t to look for. Just use it.
2401 DesiredVal = Desired.getZExtValue();
2402 } else {
2403 DesiredVal = Desired.trunc(width: S.getASTContext().getCharWidth()).getZExtValue();
2404 }
2405
2406 bool StopAtZero =
2407 (ID == Builtin::BIstrchr || ID == Builtin::BI__builtin_strchr ||
2408 ID == Builtin::BIwcschr || ID == Builtin::BI__builtin_wcschr);
2409
2410 PrimType ElemT =
2411 IsRawByte ? PT_Sint8 : *S.getContext().classify(T: getElemType(P: Ptr));
2412
2413 size_t Index = Ptr.getIndex();
2414 size_t Step = 0;
2415 for (;;) {
2416 const Pointer &ElemPtr =
2417 (Index + Step) > 0 ? Ptr.atIndex(Idx: Index + Step) : Ptr;
2418
2419 if (!CheckLoad(S, OpPC, Ptr: ElemPtr))
2420 return false;
2421
2422 uint64_t V;
2423 INT_TYPE_SWITCH_NO_BOOL(
2424 ElemT, { V = static_cast<uint64_t>(ElemPtr.load<T>().toUnsigned()); });
2425
2426 if (V == DesiredVal) {
2427 S.Stk.push<Pointer>(Args: ElemPtr);
2428 return true;
2429 }
2430
2431 if (StopAtZero && V == 0)
2432 break;
2433
2434 ++Step;
2435 if (MaxLength && Step == MaxLength->getZExtValue())
2436 break;
2437 }
2438
2439 S.Stk.push<Pointer>();
2440 return true;
2441}
2442
2443static bool interp__builtin_object_size(InterpState &S, CodePtr OpPC,
2444 const InterpFrame *Frame,
2445 const CallExpr *Call, bool IsDynamic) {
2446 const ASTContext &ASTCtx = S.getASTContext();
2447 // From the GCC docs:
2448 // Kind is an integer constant from 0 to 3. If the least significant bit is
2449 // clear, objects are whole variables. If it is set, a closest surrounding
2450 // subobject is considered the object a pointer points to. The second bit
2451 // determines if maximum or minimum of remaining bytes is computed.
2452 uint64_t Kind;
2453 if (!popToUInt64(S, E: Call->getArg(Arg: 1), Out&: Kind))
2454 return false;
2455 assert(Kind <= 3 && "unexpected kind");
2456 Pointer Ptr = S.Stk.pop<Pointer>();
2457
2458 if (auto Result = evaluateBuiltinObjectSize(ASTCtx, Kind, Ptr,
2459 E: Call->getArg(Arg: 0), IsDynamic)) {
2460 pushInteger(S, Val: *Result, QT: Call->getType());
2461 return true;
2462 }
2463
2464 if (Call->getArg(Arg: 0)->HasSideEffects(Ctx: ASTCtx)) {
2465 // "If there are any side effects in them, it returns (size_t) -1
2466 // for type 0 or 1 and (size_t) 0 for type 2 or 3."
2467 pushInteger(S, Val: Kind <= 1 ? (size_t)-1 : (size_t)0, QT: Call->getType());
2468 return true;
2469 }
2470
2471 switch (S.EvalMode) {
2472 case EvaluationMode::ConstantExpression:
2473 case EvaluationMode::ConstantFold:
2474 case EvaluationMode::IgnoreSideEffects:
2475 // Leave it to IR generation.
2476 return Invalid(S, OpPC);
2477 case EvaluationMode::ConstantExpressionUnevaluated:
2478 // Reduce it to a constant now.
2479 pushInteger(S, Val: ((Kind & 2u) ? (size_t)0 : (size_t)-1), QT: Call->getType());
2480 return true;
2481 }
2482
2483 return false;
2484}
2485
2486static bool interp__builtin_is_within_lifetime(InterpState &S, CodePtr OpPC,
2487 const CallExpr *Call) {
2488
2489 if (!S.inConstantContext())
2490 return false;
2491
2492 const Pointer &Ptr = S.Stk.pop<Pointer>();
2493
2494 auto Error = [&](int Diag) {
2495 bool CalledFromStd = false;
2496 const auto *Callee = S.Current->getCallee();
2497 if (Callee && Callee->isInStdNamespace()) {
2498 const IdentifierInfo *Identifier = Callee->getIdentifier();
2499 CalledFromStd = Identifier && Identifier->isStr(Str: "is_within_lifetime");
2500 }
2501 S.CCEDiag(SI: CalledFromStd
2502 ? S.Current->Caller->getSource(PC: S.Current->getRetPC())
2503 : S.Current->getSource(PC: OpPC),
2504 DiagId: diag::err_invalid_is_within_lifetime)
2505 << (CalledFromStd ? "std::is_within_lifetime"
2506 : "__builtin_is_within_lifetime")
2507 << Diag;
2508 return false;
2509 };
2510
2511 if (Ptr.isZero())
2512 return Error(0);
2513 if (Ptr.isOnePastEnd())
2514 return Error(1);
2515
2516 bool Result = Ptr.getLifetime() != Lifetime::Ended;
2517 if (!Ptr.isActive()) {
2518 Result = false;
2519 } else {
2520 if (!CheckLive(S, OpPC, Ptr, AK: AK_Read))
2521 return false;
2522 if (!CheckMutable(S, OpPC, Ptr))
2523 return false;
2524 if (!CheckDummy(S, OpPC, Ptr, AK: AK_Read))
2525 return false;
2526 }
2527
2528 // Check if we're currently running an initializer.
2529 if (S.initializingBlock(B: Ptr.block()))
2530 return Error(2);
2531 if (S.EvaluatingDecl && Ptr.getRootVarDecl() == S.EvaluatingDecl)
2532 return Error(2);
2533
2534 pushInteger(S, Val: Result, QT: Call->getType());
2535 return true;
2536}
2537
2538static bool interp__builtin_elementwise_int_unaryop(
2539 InterpState &S, CodePtr OpPC, const CallExpr *Call,
2540 llvm::function_ref<APInt(const APSInt &)> Fn) {
2541 assert(Call->getNumArgs() == 1);
2542
2543 // Single integer case.
2544 if (!Call->getArg(Arg: 0)->getType()->isVectorType()) {
2545 assert(Call->getType()->isIntegerType());
2546 APSInt Src;
2547 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Src))
2548 return false;
2549 APInt Result = Fn(Src);
2550 pushInteger(S, Val: APSInt(std::move(Result), !Src.isSigned()), QT: Call->getType());
2551 return true;
2552 }
2553
2554 // Vector case.
2555 const Pointer &Arg = S.Stk.pop<Pointer>();
2556 assert(Arg.getFieldDesc()->isPrimitiveArray());
2557 const Pointer &Dst = S.Stk.peek<Pointer>();
2558 assert(Dst.getFieldDesc()->isPrimitiveArray());
2559 assert(Arg.getFieldDesc()->getNumElems() ==
2560 Dst.getFieldDesc()->getNumElems());
2561
2562 QualType ElemType = Arg.getFieldDesc()->getElemQualType();
2563 PrimType ElemT = *S.getContext().classify(T: ElemType);
2564 unsigned NumElems = Arg.getNumElems();
2565 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
2566
2567 for (unsigned I = 0; I != NumElems; ++I) {
2568 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
2569 APSInt Src = Arg.elem<T>(I).toAPSInt();
2570 APInt Result = Fn(Src);
2571 Dst.elem<T>(I) = static_cast<T>(APSInt(std::move(Result), DestUnsigned));
2572 });
2573 }
2574 Dst.initializeAllElements();
2575
2576 return true;
2577}
2578
2579static bool interp__builtin_elementwise_fp_binop(
2580 InterpState &S, CodePtr OpPC, const CallExpr *Call,
2581 llvm::function_ref<std::optional<APFloat>(
2582 const APFloat &, const APFloat &, std::optional<APSInt> RoundingMode)>
2583 Fn,
2584 bool IsScalar = false) {
2585 assert((Call->getNumArgs() == 2) || (Call->getNumArgs() == 3));
2586 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
2587 assert(VT->getElementType()->isFloatingType());
2588 unsigned NumElems = VT->getNumElements();
2589
2590 // Vector case.
2591 assert(Call->getArg(0)->getType()->isVectorType() &&
2592 Call->getArg(1)->getType()->isVectorType());
2593 assert(VT->getElementType() ==
2594 Call->getArg(1)->getType()->castAs<VectorType>()->getElementType());
2595 assert(VT->getNumElements() ==
2596 Call->getArg(1)->getType()->castAs<VectorType>()->getNumElements());
2597
2598 std::optional<APSInt> RoundingMode = std::nullopt;
2599 if (Call->getNumArgs() == 3) {
2600 APSInt RoundingModeVal;
2601 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: RoundingModeVal))
2602 return false;
2603 RoundingMode = RoundingModeVal;
2604 }
2605
2606 const Pointer &BPtr = S.Stk.pop<Pointer>();
2607 const Pointer &APtr = S.Stk.pop<Pointer>();
2608 const Pointer &Dst = S.Stk.peek<Pointer>();
2609 for (unsigned ElemIdx = 0; ElemIdx != NumElems; ++ElemIdx) {
2610 using T = PrimConv<PT_Float>::T;
2611 if (IsScalar && ElemIdx > 0) {
2612 Dst.elem<T>(I: ElemIdx) = APtr.elem<T>(I: ElemIdx);
2613 continue;
2614 }
2615 APFloat ElemA = APtr.elem<T>(I: ElemIdx).getAPFloat();
2616 APFloat ElemB = BPtr.elem<T>(I: ElemIdx).getAPFloat();
2617 std::optional<APFloat> Result = Fn(ElemA, ElemB, RoundingMode);
2618 if (!Result)
2619 return false;
2620 Dst.elem<T>(I: ElemIdx) = static_cast<T>(*Result);
2621 }
2622
2623 Dst.initializeAllElements();
2624
2625 return true;
2626}
2627
2628static bool interp__builtin_scalar_fp_round_mask_binop(
2629 InterpState &S, CodePtr OpPC, const CallExpr *Call,
2630 llvm::function_ref<std::optional<APFloat>(const APFloat &, const APFloat &,
2631 std::optional<APSInt>)>
2632 Fn) {
2633 assert(Call->getNumArgs() == 5);
2634 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
2635 unsigned NumElems = VT->getNumElements();
2636
2637 APSInt RoundingMode;
2638 if (!popToAPSInt(S, E: Call->getArg(Arg: 4), Out&: RoundingMode))
2639 return false;
2640 uint64_t MaskVal;
2641 if (!popToUInt64(S, E: Call->getArg(Arg: 3), Out&: MaskVal))
2642 return false;
2643 const Pointer &SrcPtr = S.Stk.pop<Pointer>();
2644 const Pointer &BPtr = S.Stk.pop<Pointer>();
2645 const Pointer &APtr = S.Stk.pop<Pointer>();
2646 const Pointer &Dst = S.Stk.peek<Pointer>();
2647
2648 using T = PrimConv<PT_Float>::T;
2649
2650 if (MaskVal & 1) {
2651 APFloat ElemA = APtr.elem<T>(I: 0).getAPFloat();
2652 APFloat ElemB = BPtr.elem<T>(I: 0).getAPFloat();
2653 std::optional<APFloat> Result = Fn(ElemA, ElemB, RoundingMode);
2654 if (!Result)
2655 return false;
2656 Dst.elem<T>(I: 0) = static_cast<T>(*Result);
2657 } else {
2658 Dst.elem<T>(I: 0) = SrcPtr.elem<T>(I: 0);
2659 }
2660
2661 for (unsigned I = 1; I < NumElems; ++I)
2662 Dst.elem<T>(I) = APtr.elem<T>(I);
2663
2664 Dst.initializeAllElements();
2665
2666 return true;
2667}
2668
2669static bool interp__builtin_elementwise_int_binop(
2670 InterpState &S, CodePtr OpPC, const CallExpr *Call,
2671 llvm::function_ref<APInt(const APSInt &, const APSInt &)> Fn) {
2672 assert(Call->getNumArgs() == 2);
2673
2674 // Single integer case.
2675 if (!Call->getArg(Arg: 0)->getType()->isVectorType()) {
2676 assert(!Call->getArg(1)->getType()->isVectorType());
2677 APSInt RHS;
2678 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: RHS))
2679 return false;
2680 APSInt LHS;
2681 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: LHS))
2682 return false;
2683 APInt Result = Fn(LHS, RHS);
2684 pushInteger(S, Val: APSInt(std::move(Result), !LHS.isSigned()), QT: Call->getType());
2685 return true;
2686 }
2687
2688 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
2689 assert(VT->getElementType()->isIntegralOrEnumerationType());
2690 PrimType ElemT = *S.getContext().classify(T: VT->getElementType());
2691 unsigned NumElems = VT->getNumElements();
2692 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
2693
2694 // Vector + Scalar case.
2695 if (!Call->getArg(Arg: 1)->getType()->isVectorType()) {
2696 assert(Call->getArg(1)->getType()->isIntegralOrEnumerationType());
2697
2698 APSInt RHS;
2699 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: RHS))
2700 return false;
2701 const Pointer &LHS = S.Stk.pop<Pointer>();
2702 const Pointer &Dst = S.Stk.peek<Pointer>();
2703
2704 for (unsigned I = 0; I != NumElems; ++I) {
2705 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
2706 Dst.elem<T>(I) = static_cast<T>(
2707 APSInt(Fn(LHS.elem<T>(I).toAPSInt(), RHS), DestUnsigned));
2708 });
2709 }
2710 Dst.initializeAllElements();
2711 return true;
2712 }
2713
2714 // Vector case.
2715 assert(Call->getArg(0)->getType()->isVectorType() &&
2716 Call->getArg(1)->getType()->isVectorType());
2717 assert(VT->getElementType() ==
2718 Call->getArg(1)->getType()->castAs<VectorType>()->getElementType());
2719 assert(VT->getNumElements() ==
2720 Call->getArg(1)->getType()->castAs<VectorType>()->getNumElements());
2721 assert(VT->getElementType()->isIntegralOrEnumerationType());
2722
2723 const Pointer &RHS = S.Stk.pop<Pointer>();
2724 const Pointer &LHS = S.Stk.pop<Pointer>();
2725 const Pointer &Dst = S.Stk.peek<Pointer>();
2726 for (unsigned I = 0; I != NumElems; ++I) {
2727 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
2728 APSInt Elem1 = LHS.elem<T>(I).toAPSInt();
2729 APSInt Elem2 = RHS.elem<T>(I).toAPSInt();
2730 Dst.elem<T>(I) = static_cast<T>(APSInt(Fn(Elem1, Elem2), DestUnsigned));
2731 });
2732 }
2733 Dst.initializeAllElements();
2734
2735 return true;
2736}
2737
2738static bool
2739interp__builtin_ia32_pack(InterpState &S, CodePtr, const CallExpr *E,
2740 llvm::function_ref<APInt(const APSInt &)> PackFn) {
2741 const auto *VT0 = E->getArg(Arg: 0)->getType()->castAs<VectorType>();
2742 [[maybe_unused]] const auto *VT1 =
2743 E->getArg(Arg: 1)->getType()->castAs<VectorType>();
2744 assert(VT0 && VT1 && "pack builtin VT0 and VT1 must be VectorType");
2745 assert(VT0->getElementType() == VT1->getElementType() &&
2746 VT0->getNumElements() == VT1->getNumElements() &&
2747 "pack builtin VT0 and VT1 ElementType must be same");
2748
2749 const Pointer &RHS = S.Stk.pop<Pointer>();
2750 const Pointer &LHS = S.Stk.pop<Pointer>();
2751 const Pointer &Dst = S.Stk.peek<Pointer>();
2752
2753 const ASTContext &ASTCtx = S.getASTContext();
2754 unsigned SrcBits = ASTCtx.getIntWidth(T: VT0->getElementType());
2755 unsigned LHSVecLen = VT0->getNumElements();
2756 unsigned SrcPerLane = 128 / SrcBits;
2757 unsigned Lanes = LHSVecLen * SrcBits / 128;
2758
2759 PrimType SrcT = *S.getContext().classify(T: VT0->getElementType());
2760 PrimType DstT = *S.getContext().classify(T: getElemType(P: Dst));
2761 bool IsUnsigend = getElemType(P: Dst)->isUnsignedIntegerType();
2762
2763 for (unsigned Lane = 0; Lane != Lanes; ++Lane) {
2764 unsigned BaseSrc = Lane * SrcPerLane;
2765 unsigned BaseDst = Lane * (2 * SrcPerLane);
2766
2767 for (unsigned I = 0; I != SrcPerLane; ++I) {
2768 INT_TYPE_SWITCH_NO_BOOL(SrcT, {
2769 APSInt A = LHS.elem<T>(BaseSrc + I).toAPSInt();
2770 APSInt B = RHS.elem<T>(BaseSrc + I).toAPSInt();
2771
2772 assignIntegral(S, Dst.atIndex(BaseDst + I), DstT,
2773 APSInt(PackFn(A), IsUnsigend));
2774 assignIntegral(S, Dst.atIndex(BaseDst + SrcPerLane + I), DstT,
2775 APSInt(PackFn(B), IsUnsigend));
2776 });
2777 }
2778 }
2779
2780 Dst.initializeAllElements();
2781 return true;
2782}
2783
2784static bool interp__builtin_elementwise_maxmin(InterpState &S, CodePtr OpPC,
2785 const CallExpr *Call,
2786 unsigned BuiltinID) {
2787 assert(Call->getNumArgs() == 2);
2788
2789 QualType Arg0Type = Call->getArg(Arg: 0)->getType();
2790
2791 // TODO: Support floating-point types.
2792 if (!(Arg0Type->isIntegerType() ||
2793 (Arg0Type->isVectorType() &&
2794 Arg0Type->castAs<VectorType>()->getElementType()->isIntegerType())))
2795 return false;
2796
2797 if (!Arg0Type->isVectorType()) {
2798 assert(!Call->getArg(1)->getType()->isVectorType());
2799 APSInt RHS;
2800 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: RHS))
2801 return false;
2802 APSInt LHS;
2803 if (!popToAPSInt(S, T: Arg0Type, Out&: LHS))
2804 return false;
2805 APInt Result;
2806 if (BuiltinID == Builtin::BI__builtin_elementwise_max) {
2807 Result = std::max(a: LHS, b: RHS);
2808 } else if (BuiltinID == Builtin::BI__builtin_elementwise_min) {
2809 Result = std::min(a: LHS, b: RHS);
2810 } else {
2811 llvm_unreachable("Wrong builtin ID");
2812 }
2813
2814 pushInteger(S, Val: APSInt(Result, !LHS.isSigned()), QT: Call->getType());
2815 return true;
2816 }
2817
2818 // Vector case.
2819 assert(Call->getArg(0)->getType()->isVectorType() &&
2820 Call->getArg(1)->getType()->isVectorType());
2821 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
2822 assert(VT->getElementType() ==
2823 Call->getArg(1)->getType()->castAs<VectorType>()->getElementType());
2824 assert(VT->getNumElements() ==
2825 Call->getArg(1)->getType()->castAs<VectorType>()->getNumElements());
2826 assert(VT->getElementType()->isIntegralOrEnumerationType());
2827
2828 const Pointer &RHS = S.Stk.pop<Pointer>();
2829 const Pointer &LHS = S.Stk.pop<Pointer>();
2830 const Pointer &Dst = S.Stk.peek<Pointer>();
2831 PrimType ElemT = *S.getContext().classify(T: VT->getElementType());
2832 unsigned NumElems = VT->getNumElements();
2833 for (unsigned I = 0; I != NumElems; ++I) {
2834 APSInt Elem1;
2835 APSInt Elem2;
2836 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
2837 Elem1 = LHS.elem<T>(I).toAPSInt();
2838 Elem2 = RHS.elem<T>(I).toAPSInt();
2839 });
2840
2841 APSInt Result;
2842 if (BuiltinID == Builtin::BI__builtin_elementwise_max) {
2843 Result = APSInt(std::max(a: Elem1, b: Elem2),
2844 Call->getType()->isUnsignedIntegerOrEnumerationType());
2845 } else if (BuiltinID == Builtin::BI__builtin_elementwise_min) {
2846 Result = APSInt(std::min(a: Elem1, b: Elem2),
2847 Call->getType()->isUnsignedIntegerOrEnumerationType());
2848 } else {
2849 llvm_unreachable("Wrong builtin ID");
2850 }
2851
2852 INT_TYPE_SWITCH_NO_BOOL(ElemT,
2853 { Dst.elem<T>(I) = static_cast<T>(Result); });
2854 }
2855 Dst.initializeAllElements();
2856
2857 return true;
2858}
2859
2860static bool interp__builtin_ia32_pmul(
2861 InterpState &S, CodePtr OpPC, const CallExpr *Call,
2862 llvm::function_ref<APInt(const APSInt &, const APSInt &, const APSInt &,
2863 const APSInt &)>
2864 Fn) {
2865 assert(Call->getArg(0)->getType()->isVectorType() &&
2866 Call->getArg(1)->getType()->isVectorType());
2867 const Pointer &RHS = S.Stk.pop<Pointer>();
2868 const Pointer &LHS = S.Stk.pop<Pointer>();
2869 const Pointer &Dst = S.Stk.peek<Pointer>();
2870
2871 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
2872 PrimType ElemT = *S.getContext().classify(T: VT->getElementType());
2873 unsigned NumElems = VT->getNumElements();
2874 const auto *DestVT = Call->getType()->castAs<VectorType>();
2875 PrimType DestElemT = *S.getContext().classify(T: DestVT->getElementType());
2876 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
2877
2878 unsigned DstElem = 0;
2879 for (unsigned I = 0; I != NumElems; I += 2) {
2880 APSInt Result;
2881 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
2882 APSInt LoLHS = LHS.elem<T>(I).toAPSInt();
2883 APSInt HiLHS = LHS.elem<T>(I + 1).toAPSInt();
2884 APSInt LoRHS = RHS.elem<T>(I).toAPSInt();
2885 APSInt HiRHS = RHS.elem<T>(I + 1).toAPSInt();
2886 Result = APSInt(Fn(LoLHS, HiLHS, LoRHS, HiRHS), DestUnsigned);
2887 });
2888
2889 INT_TYPE_SWITCH_NO_BOOL(DestElemT,
2890 { Dst.elem<T>(DstElem) = static_cast<T>(Result); });
2891 ++DstElem;
2892 }
2893
2894 Dst.initializeAllElements();
2895 return true;
2896}
2897
2898static bool interp__builtin_ia32_psadbw(InterpState &S, CodePtr OpPC,
2899 const CallExpr *Call) {
2900 assert(Call->getNumArgs() == 2);
2901
2902 const Pointer &RHS = S.Stk.pop<Pointer>();
2903 const Pointer &LHS = S.Stk.pop<Pointer>();
2904 const Pointer &Dst = S.Stk.peek<Pointer>();
2905
2906 const auto *SrcVT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
2907 PrimType SrcElemT = *S.getContext().classify(T: SrcVT->getElementType());
2908 unsigned SourceLen = SrcVT->getNumElements();
2909 assert((SourceLen % 8) == 0);
2910
2911 const auto *DestVT = Call->getType()->castAs<VectorType>();
2912 PrimType DestElemT = *S.getContext().classify(T: DestVT->getElementType());
2913 bool DestUnsigned =
2914 DestVT->getElementType()->isUnsignedIntegerOrEnumerationType();
2915
2916 unsigned DstElem = 0;
2917 for (unsigned Lane = 0; Lane != SourceLen; Lane += 8) {
2918 APInt Sum(64, 0);
2919 for (unsigned I = 0; I != 8; ++I) {
2920 INT_TYPE_SWITCH_NO_BOOL(SrcElemT, {
2921 APSInt L = LHS.elem<T>(Lane + I).toAPSInt();
2922 APSInt R = RHS.elem<T>(Lane + I).toAPSInt();
2923 Sum += llvm::APIntOps::abdu(L.extOrTrunc(8), R.extOrTrunc(8)).zext(64);
2924 });
2925 }
2926
2927 INT_TYPE_SWITCH_NO_BOOL(DestElemT, {
2928 Dst.elem<T>(DstElem) = static_cast<T>(APSInt(Sum, DestUnsigned));
2929 });
2930 ++DstElem;
2931 }
2932
2933 Dst.initializeAllElements();
2934 return true;
2935}
2936
2937static bool interp__builtin_ia32_dbpsadbw(InterpState &S, CodePtr OpPC,
2938 const CallExpr *Call) {
2939 assert(Call->getNumArgs() == 3);
2940 uint64_t Imm;
2941 if (!popToUInt64(S, E: Call->getArg(Arg: 2), Out&: Imm))
2942 return false;
2943
2944 const Pointer &Src2 = S.Stk.pop<Pointer>();
2945 const Pointer &Src1 = S.Stk.pop<Pointer>();
2946 const Pointer &Dst = S.Stk.peek<Pointer>();
2947
2948 const auto *SrcVT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
2949 PrimType SrcElemT = *S.getContext().classify(T: SrcVT->getElementType());
2950 unsigned SourceLen = SrcVT->getNumElements();
2951
2952 const auto *DestVT = Call->getType()->castAs<VectorType>();
2953 PrimType DestElemT = *S.getContext().classify(T: DestVT->getElementType());
2954 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
2955
2956 constexpr unsigned LaneSize = 16; // 128-bit lane = 16 bytes
2957
2958 // Phase 1: Shuffle Src2 using all four 2-bit fields of imm8.
2959 // Within each 128-bit lane, for group j (0..3), select a 4-byte block
2960 // from Src2 based on bits [2*j+1:2*j] of imm8.
2961 SmallVector<uint8_t, 64> Shuffled(SourceLen);
2962 for (unsigned I = 0; I < SourceLen; I += LaneSize) {
2963 for (unsigned J = 0; J < 4; ++J) {
2964 unsigned Part = (Imm >> (2 * J)) & 3;
2965 for (unsigned K = 0; K < 4; ++K) {
2966 INT_TYPE_SWITCH_NO_BOOL(SrcElemT, {
2967 Shuffled[I + 4 * J + K] =
2968 static_cast<uint8_t>(Src2.elem<T>(I + 4 * Part + K));
2969 });
2970 }
2971 }
2972 }
2973
2974 // Phase 2: Sliding SAD computation.
2975 // For every group of 4 output u16 values, compute absolute differences
2976 // using overlapping windows into Src1 and the shuffled array.
2977 unsigned Size = SourceLen / 2; // number of output u16 elements
2978 for (unsigned I = 0; I < Size; I += 4) {
2979 unsigned Sad[4] = {0, 0, 0, 0};
2980 for (unsigned J = 0; J < 4; ++J) {
2981 uint8_t A1, A2;
2982 INT_TYPE_SWITCH_NO_BOOL(SrcElemT, {
2983 A1 = static_cast<uint8_t>(Src1.elem<T>(2 * I + J));
2984 A2 = static_cast<uint8_t>(Src1.elem<T>(2 * I + J + 4));
2985 });
2986 uint8_t B0 = Shuffled[2 * I + J];
2987 uint8_t B1 = Shuffled[2 * I + J + 1];
2988 uint8_t B2 = Shuffled[2 * I + J + 2];
2989 uint8_t B3 = Shuffled[2 * I + J + 3];
2990 Sad[0] += (A1 > B0) ? (A1 - B0) : (B0 - A1);
2991 Sad[1] += (A1 > B1) ? (A1 - B1) : (B1 - A1);
2992 Sad[2] += (A2 > B2) ? (A2 - B2) : (B2 - A2);
2993 Sad[3] += (A2 > B3) ? (A2 - B3) : (B3 - A2);
2994 }
2995 for (unsigned R = 0; R < 4; ++R) {
2996 INT_TYPE_SWITCH_NO_BOOL(DestElemT, {
2997 Dst.elem<T>(I + R) =
2998 static_cast<T>(APSInt(APInt(16, Sad[R]), DestUnsigned));
2999 });
3000 }
3001 }
3002
3003 Dst.initializeAllElements();
3004 return true;
3005}
3006
3007static bool interp__builtin_ia32_mpsadbw(InterpState &S, CodePtr OpPC,
3008 const CallExpr *Call) {
3009 assert(Call->getNumArgs() == 3);
3010 uint64_t Imm;
3011 if (!popToUInt64(S, E: Call->getArg(Arg: 2), Out&: Imm))
3012 return false;
3013
3014 const Pointer &Src2 = S.Stk.pop<Pointer>();
3015 const Pointer &Src1 = S.Stk.pop<Pointer>();
3016 const Pointer &Dst = S.Stk.peek<Pointer>();
3017
3018 const auto *SrcVT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
3019 PrimType SrcElemT = *S.getContext().classify(T: SrcVT->getElementType());
3020 unsigned SourceLen = SrcVT->getNumElements();
3021 assert((SourceLen == 16 || SourceLen == 32) &&
3022 "MPSADBW operates on 128-bit or 256-bit vectors");
3023
3024 const auto *DestVT = Call->getType()->castAs<VectorType>();
3025 PrimType DestElemT = *S.getContext().classify(T: DestVT->getElementType());
3026 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
3027
3028 constexpr unsigned LaneSize = 16; // 128-bit lane = 16 bytes
3029 unsigned NumLanes = SourceLen / LaneSize;
3030
3031 for (unsigned Lane = 0; Lane != NumLanes; ++Lane) {
3032 unsigned Ctrl = (Imm >> (3 * Lane)) & 0x7;
3033 unsigned AOff = ((Ctrl >> 2) & 1) * 4;
3034 unsigned BOff = (Ctrl & 3) * 4;
3035 for (unsigned J = 0; J != 8; ++J) {
3036 uint16_t Sad = 0;
3037 for (unsigned K = 0; K != 4; ++K) {
3038 uint8_t A, B;
3039 INT_TYPE_SWITCH_NO_BOOL(SrcElemT, {
3040 A = static_cast<uint8_t>(
3041 Src1.elem<T>(Lane * LaneSize + AOff + J + K));
3042 B = static_cast<uint8_t>(Src2.elem<T>(Lane * LaneSize + BOff + K));
3043 });
3044 Sad += (A > B) ? (A - B) : (B - A);
3045 }
3046 INT_TYPE_SWITCH_NO_BOOL(DestElemT, {
3047 Dst.elem<T>(Lane * 8 + J) =
3048 static_cast<T>(APSInt(APInt(16, Sad), DestUnsigned));
3049 });
3050 }
3051 }
3052
3053 Dst.initializeAllElements();
3054 return true;
3055}
3056
3057static bool interp_builtin_horizontal_int_binop(
3058 InterpState &S, CodePtr OpPC, const CallExpr *Call,
3059 llvm::function_ref<APInt(const APSInt &, const APSInt &)> Fn) {
3060 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
3061 PrimType ElemT = *S.getContext().classify(T: VT->getElementType());
3062 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
3063
3064 const Pointer &RHS = S.Stk.pop<Pointer>();
3065 const Pointer &LHS = S.Stk.pop<Pointer>();
3066 const Pointer &Dst = S.Stk.peek<Pointer>();
3067 unsigned NumElts = VT->getNumElements();
3068 unsigned EltBits = S.getASTContext().getIntWidth(T: VT->getElementType());
3069 unsigned EltsPerLane = 128 / EltBits;
3070 unsigned Lanes = NumElts * EltBits / 128;
3071 unsigned DestIndex = 0;
3072
3073 for (unsigned Lane = 0; Lane < Lanes; ++Lane) {
3074 unsigned LaneStart = Lane * EltsPerLane;
3075 for (unsigned I = 0; I < EltsPerLane; I += 2) {
3076 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3077 APSInt Elem1 = LHS.elem<T>(LaneStart + I).toAPSInt();
3078 APSInt Elem2 = LHS.elem<T>(LaneStart + I + 1).toAPSInt();
3079 APSInt ResL = APSInt(Fn(Elem1, Elem2), DestUnsigned);
3080 Dst.elem<T>(DestIndex++) = static_cast<T>(ResL);
3081 });
3082 }
3083
3084 for (unsigned I = 0; I < EltsPerLane; I += 2) {
3085 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3086 APSInt Elem1 = RHS.elem<T>(LaneStart + I).toAPSInt();
3087 APSInt Elem2 = RHS.elem<T>(LaneStart + I + 1).toAPSInt();
3088 APSInt ResR = APSInt(Fn(Elem1, Elem2), DestUnsigned);
3089 Dst.elem<T>(DestIndex++) = static_cast<T>(ResR);
3090 });
3091 }
3092 }
3093 Dst.initializeAllElements();
3094 return true;
3095}
3096
3097static bool interp_builtin_horizontal_fp_binop(
3098 InterpState &S, CodePtr OpPC, const CallExpr *Call,
3099 llvm::function_ref<APFloat(const APFloat &, const APFloat &,
3100 llvm::RoundingMode)>
3101 Fn) {
3102 const Pointer &RHS = S.Stk.pop<Pointer>();
3103 const Pointer &LHS = S.Stk.pop<Pointer>();
3104 const Pointer &Dst = S.Stk.peek<Pointer>();
3105 FPOptions FPO = Call->getFPFeaturesInEffect(LO: S.Ctx.getLangOpts());
3106 llvm::RoundingMode RM = getRoundingMode(FPO);
3107 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
3108
3109 unsigned NumElts = VT->getNumElements();
3110 unsigned EltBits = S.getASTContext().getTypeSize(T: VT->getElementType());
3111 unsigned NumLanes = NumElts * EltBits / 128;
3112 unsigned NumElemsPerLane = NumElts / NumLanes;
3113 unsigned HalfElemsPerLane = NumElemsPerLane / 2;
3114
3115 for (unsigned L = 0; L != NumElts; L += NumElemsPerLane) {
3116 using T = PrimConv<PT_Float>::T;
3117 for (unsigned E = 0; E != HalfElemsPerLane; ++E) {
3118 APFloat Elem1 = LHS.elem<T>(I: L + (2 * E) + 0).getAPFloat();
3119 APFloat Elem2 = LHS.elem<T>(I: L + (2 * E) + 1).getAPFloat();
3120 Dst.elem<T>(I: L + E) = static_cast<T>(Fn(Elem1, Elem2, RM));
3121 }
3122 for (unsigned E = 0; E != HalfElemsPerLane; ++E) {
3123 APFloat Elem1 = RHS.elem<T>(I: L + (2 * E) + 0).getAPFloat();
3124 APFloat Elem2 = RHS.elem<T>(I: L + (2 * E) + 1).getAPFloat();
3125 Dst.elem<T>(I: L + E + HalfElemsPerLane) =
3126 static_cast<T>(Fn(Elem1, Elem2, RM));
3127 }
3128 }
3129 Dst.initializeAllElements();
3130 return true;
3131}
3132
3133static bool interp__builtin_ia32_addsub(InterpState &S, CodePtr OpPC,
3134 const CallExpr *Call) {
3135 // Addsub: alternates between subtraction and addition
3136 // Result[i] = (i % 2 == 0) ? (a[i] - b[i]) : (a[i] + b[i])
3137 const Pointer &RHS = S.Stk.pop<Pointer>();
3138 const Pointer &LHS = S.Stk.pop<Pointer>();
3139 const Pointer &Dst = S.Stk.peek<Pointer>();
3140 FPOptions FPO = Call->getFPFeaturesInEffect(LO: S.Ctx.getLangOpts());
3141 llvm::RoundingMode RM = getRoundingMode(FPO);
3142 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
3143 unsigned NumElems = VT->getNumElements();
3144
3145 using T = PrimConv<PT_Float>::T;
3146 for (unsigned I = 0; I != NumElems; ++I) {
3147 APFloat LElem = LHS.elem<T>(I).getAPFloat();
3148 APFloat RElem = RHS.elem<T>(I).getAPFloat();
3149 if (I % 2 == 0) {
3150 // Even indices: subtract
3151 LElem.subtract(RHS: RElem, RM);
3152 } else {
3153 // Odd indices: add
3154 LElem.add(RHS: RElem, RM);
3155 }
3156 Dst.elem<T>(I) = static_cast<T>(LElem);
3157 }
3158 Dst.initializeAllElements();
3159 return true;
3160}
3161
3162static bool interp__builtin_ia32_pclmulqdq(InterpState &S, CodePtr OpPC,
3163 const CallExpr *Call) {
3164 // PCLMULQDQ: carry-less multiplication of selected 64-bit halves
3165 // imm8 bit 0: selects lower (0) or upper (1) 64 bits of first operand
3166 // imm8 bit 4: selects lower (0) or upper (1) 64 bits of second operand
3167 assert(Call->getArg(0)->getType()->isVectorType() &&
3168 Call->getArg(1)->getType()->isVectorType());
3169
3170 // Extract imm8 argument
3171 APSInt Imm8;
3172 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: Imm8))
3173 return false;
3174 bool SelectUpperA = (Imm8 & 0x01) != 0;
3175 bool SelectUpperB = (Imm8 & 0x10) != 0;
3176
3177 const Pointer &RHS = S.Stk.pop<Pointer>();
3178 const Pointer &LHS = S.Stk.pop<Pointer>();
3179 const Pointer &Dst = S.Stk.peek<Pointer>();
3180
3181 const auto *VT = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
3182 PrimType ElemT = *S.getContext().classify(T: VT->getElementType());
3183 unsigned NumElems = VT->getNumElements();
3184 const auto *DestVT = Call->getType()->castAs<VectorType>();
3185 PrimType DestElemT = *S.getContext().classify(T: DestVT->getElementType());
3186 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
3187
3188 // Process each 128-bit lane (2 elements at a time)
3189 for (unsigned Lane = 0; Lane < NumElems; Lane += 2) {
3190 APSInt A0, A1, B0, B1;
3191 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3192 A0 = LHS.elem<T>(Lane + 0).toAPSInt();
3193 A1 = LHS.elem<T>(Lane + 1).toAPSInt();
3194 B0 = RHS.elem<T>(Lane + 0).toAPSInt();
3195 B1 = RHS.elem<T>(Lane + 1).toAPSInt();
3196 });
3197
3198 // Select the appropriate 64-bit values based on imm8
3199 APInt A = SelectUpperA ? A1 : A0;
3200 APInt B = SelectUpperB ? B1 : B0;
3201
3202 // Extend both operands to 128 bits for carry-less multiplication
3203 APInt A128 = A.zext(width: 128);
3204 APInt B128 = B.zext(width: 128);
3205
3206 // Use APIntOps::clmul for carry-less multiplication
3207 APInt Result = llvm::APIntOps::clmul(LHS: A128, RHS: B128);
3208
3209 // Split the 128-bit result into two 64-bit halves
3210 APSInt ResultLow(Result.extractBits(numBits: 64, bitPosition: 0), DestUnsigned);
3211 APSInt ResultHigh(Result.extractBits(numBits: 64, bitPosition: 64), DestUnsigned);
3212
3213 INT_TYPE_SWITCH_NO_BOOL(DestElemT, {
3214 Dst.elem<T>(Lane + 0) = static_cast<T>(ResultLow);
3215 Dst.elem<T>(Lane + 1) = static_cast<T>(ResultHigh);
3216 });
3217 }
3218
3219 Dst.initializeAllElements();
3220 return true;
3221}
3222
3223static bool interp__builtin_elementwise_triop_fp(
3224 InterpState &S, CodePtr OpPC, const CallExpr *Call,
3225 llvm::function_ref<APFloat(const APFloat &, const APFloat &,
3226 const APFloat &, llvm::RoundingMode)>
3227 Fn) {
3228 assert(Call->getNumArgs() == 3);
3229
3230 FPOptions FPO = Call->getFPFeaturesInEffect(LO: S.Ctx.getLangOpts());
3231 llvm::RoundingMode RM = getRoundingMode(FPO);
3232 QualType Arg1Type = Call->getArg(Arg: 0)->getType();
3233 QualType Arg2Type = Call->getArg(Arg: 1)->getType();
3234 QualType Arg3Type = Call->getArg(Arg: 2)->getType();
3235
3236 // Non-vector floating point types.
3237 if (!Arg1Type->isVectorType()) {
3238 assert(!Arg2Type->isVectorType());
3239 assert(!Arg3Type->isVectorType());
3240 (void)Arg2Type;
3241 (void)Arg3Type;
3242
3243 const Floating &Z = S.Stk.pop<Floating>();
3244 const Floating &Y = S.Stk.pop<Floating>();
3245 const Floating &X = S.Stk.pop<Floating>();
3246 APFloat F = Fn(X.getAPFloat(), Y.getAPFloat(), Z.getAPFloat(), RM);
3247 Floating Result = S.allocFloat(Sem: X.getSemantics());
3248 Result.copy(F);
3249 S.Stk.push<Floating>(Args&: Result);
3250 return true;
3251 }
3252
3253 // Vector type.
3254 assert(Arg1Type->isVectorType() && Arg2Type->isVectorType() &&
3255 Arg3Type->isVectorType());
3256
3257 const VectorType *VecTy = Arg1Type->castAs<VectorType>();
3258 QualType ElemQT = VecTy->getElementType();
3259 unsigned NumElems = VecTy->getNumElements();
3260
3261 assert(ElemQT == Arg2Type->castAs<VectorType>()->getElementType() &&
3262 ElemQT == Arg3Type->castAs<VectorType>()->getElementType());
3263 assert(NumElems == Arg2Type->castAs<VectorType>()->getNumElements() &&
3264 NumElems == Arg3Type->castAs<VectorType>()->getNumElements());
3265 assert(ElemQT->isRealFloatingType());
3266 (void)ElemQT;
3267
3268 const Pointer &VZ = S.Stk.pop<Pointer>();
3269 const Pointer &VY = S.Stk.pop<Pointer>();
3270 const Pointer &VX = S.Stk.pop<Pointer>();
3271 const Pointer &Dst = S.Stk.peek<Pointer>();
3272 for (unsigned I = 0; I != NumElems; ++I) {
3273 using T = PrimConv<PT_Float>::T;
3274 APFloat X = VX.elem<T>(I).getAPFloat();
3275 APFloat Y = VY.elem<T>(I).getAPFloat();
3276 APFloat Z = VZ.elem<T>(I).getAPFloat();
3277 APFloat F = Fn(X, Y, Z, RM);
3278 Dst.elem<Floating>(I) = Floating(F);
3279 }
3280 Dst.initializeAllElements();
3281 return true;
3282}
3283
3284/// AVX512 predicated move: "Result = Mask[] ? LHS[] : RHS[]".
3285static bool interp__builtin_ia32_select(InterpState &S, CodePtr OpPC,
3286 const CallExpr *Call) {
3287 const Pointer &RHS = S.Stk.pop<Pointer>();
3288 const Pointer &LHS = S.Stk.pop<Pointer>();
3289 APSInt Mask;
3290 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Mask))
3291 return false;
3292 const Pointer &Dst = S.Stk.peek<Pointer>();
3293
3294 assert(LHS.getNumElems() == RHS.getNumElems());
3295 assert(LHS.getNumElems() == Dst.getNumElems());
3296 unsigned NumElems = LHS.getNumElems();
3297 PrimType ElemT = LHS.getFieldDesc()->getPrimType();
3298 PrimType DstElemT = Dst.getFieldDesc()->getPrimType();
3299
3300 for (unsigned I = 0; I != NumElems; ++I) {
3301 if (ElemT == PT_Float) {
3302 assert(DstElemT == PT_Float);
3303 Dst.elem<Floating>(I) =
3304 Mask[I] ? LHS.elem<Floating>(I) : RHS.elem<Floating>(I);
3305 } else {
3306 APSInt Elem;
3307 INT_TYPE_SWITCH(ElemT, {
3308 Elem = Mask[I] ? LHS.elem<T>(I).toAPSInt() : RHS.elem<T>(I).toAPSInt();
3309 });
3310 INT_TYPE_SWITCH_NO_BOOL(DstElemT,
3311 { Dst.elem<T>(I) = static_cast<T>(Elem); });
3312 }
3313 }
3314 Dst.initializeAllElements();
3315
3316 return true;
3317}
3318
3319/// Scalar variant of AVX512 predicated select:
3320/// Result[i] = (Mask bit 0) ? LHS[i] : RHS[i], but only element 0 may change.
3321/// All other elements are taken from RHS.
3322static bool interp__builtin_ia32_select_scalar(InterpState &S,
3323 const CallExpr *Call) {
3324 unsigned N =
3325 Call->getArg(Arg: 1)->getType()->castAs<VectorType>()->getNumElements();
3326
3327 const Pointer &W = S.Stk.pop<Pointer>();
3328 const Pointer &A = S.Stk.pop<Pointer>();
3329 APSInt U;
3330 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: U))
3331 return false;
3332 const Pointer &Dst = S.Stk.peek<Pointer>();
3333
3334 bool TakeA0 = U.getZExtValue() & 1ULL;
3335
3336 for (unsigned I = TakeA0; I != N; ++I)
3337 Dst.elem<Floating>(I) = W.elem<Floating>(I);
3338 if (TakeA0)
3339 Dst.elem<Floating>(I: 0) = A.elem<Floating>(I: 0);
3340
3341 Dst.initializeAllElements();
3342 return true;
3343}
3344
3345static bool interp__builtin_ia32_test_op(
3346 InterpState &S, CodePtr OpPC, const CallExpr *Call,
3347 llvm::function_ref<bool(const APInt &A, const APInt &B)> Fn) {
3348 const Pointer &RHS = S.Stk.pop<Pointer>();
3349 const Pointer &LHS = S.Stk.pop<Pointer>();
3350
3351 assert(LHS.getNumElems() == RHS.getNumElems());
3352
3353 unsigned SourceLen = LHS.getNumElems();
3354 QualType ElemQT = getElemType(P: LHS);
3355 OptPrimType ElemPT = S.getContext().classify(T: ElemQT);
3356 unsigned LaneWidth = S.getASTContext().getTypeSize(T: ElemQT);
3357
3358 APInt AWide(LaneWidth * SourceLen, 0);
3359 APInt BWide(LaneWidth * SourceLen, 0);
3360
3361 for (unsigned I = 0; I != SourceLen; ++I) {
3362 APInt ALane;
3363 APInt BLane;
3364
3365 if (ElemQT->isIntegerType()) { // Get value.
3366 INT_TYPE_SWITCH_NO_BOOL(*ElemPT, {
3367 ALane = LHS.elem<T>(I).toAPSInt();
3368 BLane = RHS.elem<T>(I).toAPSInt();
3369 });
3370 } else if (ElemQT->isFloatingType()) { // Get only sign bit.
3371 using T = PrimConv<PT_Float>::T;
3372 ALane = LHS.elem<T>(I).getAPFloat().bitcastToAPInt().isNegative();
3373 BLane = RHS.elem<T>(I).getAPFloat().bitcastToAPInt().isNegative();
3374 } else { // Must be integer or floating type.
3375 return false;
3376 }
3377 AWide.insertBits(SubBits: ALane, bitPosition: I * LaneWidth);
3378 BWide.insertBits(SubBits: BLane, bitPosition: I * LaneWidth);
3379 }
3380 pushInteger(S, Val: Fn(AWide, BWide), QT: Call->getType());
3381 return true;
3382}
3383
3384static bool interp__builtin_ia32_movmsk_op(InterpState &S, CodePtr OpPC,
3385 const CallExpr *Call) {
3386 assert(Call->getNumArgs() == 1);
3387
3388 const Pointer &Source = S.Stk.pop<Pointer>();
3389
3390 unsigned SourceLen = Source.getNumElems();
3391 QualType ElemQT = getElemType(P: Source);
3392 OptPrimType ElemT = S.getContext().classify(T: ElemQT);
3393 unsigned ResultLen =
3394 S.getASTContext().getTypeSize(T: Call->getType()); // Always 32-bit integer.
3395 APInt Result(ResultLen, 0);
3396
3397 for (unsigned I = 0; I != SourceLen; ++I) {
3398 APInt Elem;
3399 if (ElemQT->isIntegerType()) {
3400 INT_TYPE_SWITCH_NO_BOOL(*ElemT, { Elem = Source.elem<T>(I).toAPSInt(); });
3401 } else if (ElemQT->isRealFloatingType()) {
3402 using T = PrimConv<PT_Float>::T;
3403 Elem = Source.elem<T>(I).getAPFloat().bitcastToAPInt();
3404 } else {
3405 return false;
3406 }
3407 Result.setBitVal(BitPosition: I, BitValue: Elem.isNegative());
3408 }
3409 pushInteger(S, Val: Result, QT: Call->getType());
3410 return true;
3411}
3412
3413static bool interp__builtin_elementwise_triop(
3414 InterpState &S, CodePtr OpPC, const CallExpr *Call,
3415 llvm::function_ref<APInt(const APSInt &, const APSInt &, const APSInt &)>
3416 Fn) {
3417 assert(Call->getNumArgs() == 3);
3418
3419 QualType Arg0Type = Call->getArg(Arg: 0)->getType();
3420 QualType Arg2Type = Call->getArg(Arg: 2)->getType();
3421 // Non-vector integer types.
3422 if (!Arg0Type->isVectorType()) {
3423 APSInt Op2;
3424 if (!popToAPSInt(S, T: Arg2Type, Out&: Op2))
3425 return false;
3426 APSInt Op1;
3427 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: Op1))
3428 return false;
3429 APSInt Op0;
3430 if (!popToAPSInt(S, T: Arg0Type, Out&: Op0))
3431 return false;
3432 APSInt Result = APSInt(Fn(Op0, Op1, Op2), Op0.isUnsigned());
3433 pushInteger(S, Val: Result, QT: Call->getType());
3434 return true;
3435 }
3436
3437 const auto *VecT = Arg0Type->castAs<VectorType>();
3438 PrimType ElemT = *S.getContext().classify(T: VecT->getElementType());
3439 unsigned NumElems = VecT->getNumElements();
3440 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
3441
3442 // Vector + Vector + Scalar case.
3443 if (!Arg2Type->isVectorType()) {
3444 APSInt Op2;
3445 if (!popToAPSInt(S, T: Arg2Type, Out&: Op2))
3446 return false;
3447
3448 const Pointer &Op1 = S.Stk.pop<Pointer>();
3449 const Pointer &Op0 = S.Stk.pop<Pointer>();
3450 const Pointer &Dst = S.Stk.peek<Pointer>();
3451 for (unsigned I = 0; I != NumElems; ++I) {
3452 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3453 Dst.elem<T>(I) = static_cast<T>(APSInt(
3454 Fn(Op0.elem<T>(I).toAPSInt(), Op1.elem<T>(I).toAPSInt(), Op2),
3455 DestUnsigned));
3456 });
3457 }
3458 Dst.initializeAllElements();
3459
3460 return true;
3461 }
3462
3463 // Vector type.
3464 const Pointer &Op2 = S.Stk.pop<Pointer>();
3465 const Pointer &Op1 = S.Stk.pop<Pointer>();
3466 const Pointer &Op0 = S.Stk.pop<Pointer>();
3467 const Pointer &Dst = S.Stk.peek<Pointer>();
3468 for (unsigned I = 0; I != NumElems; ++I) {
3469 APSInt Val0, Val1, Val2;
3470 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3471 Val0 = Op0.elem<T>(I).toAPSInt();
3472 Val1 = Op1.elem<T>(I).toAPSInt();
3473 Val2 = Op2.elem<T>(I).toAPSInt();
3474 });
3475 APSInt Result = APSInt(Fn(Val0, Val1, Val2), Val0.isUnsigned());
3476 INT_TYPE_SWITCH_NO_BOOL(ElemT,
3477 { Dst.elem<T>(I) = static_cast<T>(Result); });
3478 }
3479 Dst.initializeAllElements();
3480
3481 return true;
3482}
3483
3484static bool interp__builtin_ia32_extract_vector(InterpState &S, CodePtr OpPC,
3485 const CallExpr *Call,
3486 unsigned ID) {
3487 assert(Call->getNumArgs() == 2);
3488
3489 APSInt ImmAPS;
3490 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: ImmAPS))
3491 return false;
3492 uint64_t Index = ImmAPS.getZExtValue();
3493
3494 const Pointer &Src = S.Stk.pop<Pointer>();
3495 if (!Src.getFieldDesc()->isPrimitiveArray())
3496 return false;
3497
3498 const Pointer &Dst = S.Stk.peek<Pointer>();
3499 if (!Dst.getFieldDesc()->isPrimitiveArray())
3500 return false;
3501
3502 unsigned SrcElems = Src.getNumElems();
3503 unsigned DstElems = Dst.getNumElems();
3504
3505 unsigned NumLanes = SrcElems / DstElems;
3506 unsigned Lane = static_cast<unsigned>(Index % NumLanes);
3507 unsigned ExtractPos = Lane * DstElems;
3508
3509 PrimType ElemT = Src.getFieldDesc()->getPrimType();
3510
3511 TYPE_SWITCH(ElemT, {
3512 for (unsigned I = 0; I != DstElems; ++I) {
3513 Dst.elem<T>(I) = Src.elem<T>(ExtractPos + I);
3514 }
3515 });
3516
3517 Dst.initializeAllElements();
3518 return true;
3519}
3520
3521static bool interp__builtin_ia32_extract_vector_masked(InterpState &S,
3522 CodePtr OpPC,
3523 const CallExpr *Call,
3524 unsigned ID) {
3525 assert(Call->getNumArgs() == 4);
3526
3527 APSInt MaskAPS;
3528 if (!popToAPSInt(S, E: Call->getArg(Arg: 3), Out&: MaskAPS))
3529 return false;
3530 const Pointer &Merge = S.Stk.pop<Pointer>();
3531 APSInt ImmAPS;
3532 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: ImmAPS))
3533 return false;
3534 const Pointer &Src = S.Stk.pop<Pointer>();
3535
3536 if (!Src.getFieldDesc()->isPrimitiveArray() ||
3537 !Merge.getFieldDesc()->isPrimitiveArray())
3538 return false;
3539
3540 const Pointer &Dst = S.Stk.peek<Pointer>();
3541 if (!Dst.getFieldDesc()->isPrimitiveArray())
3542 return false;
3543
3544 unsigned SrcElems = Src.getNumElems();
3545 unsigned DstElems = Dst.getNumElems();
3546
3547 unsigned NumLanes = SrcElems / DstElems;
3548 unsigned Lane = static_cast<unsigned>(ImmAPS.getZExtValue() % NumLanes);
3549 unsigned Base = Lane * DstElems;
3550
3551 PrimType ElemT = Src.getFieldDesc()->getPrimType();
3552
3553 TYPE_SWITCH(ElemT, {
3554 for (unsigned I = 0; I != DstElems; ++I) {
3555 if (MaskAPS[I])
3556 Dst.elem<T>(I) = Src.elem<T>(Base + I);
3557 else
3558 Dst.elem<T>(I) = Merge.elem<T>(I);
3559 }
3560 });
3561
3562 Dst.initializeAllElements();
3563 return true;
3564}
3565
3566static bool interp__builtin_ia32_insert_subvector(InterpState &S, CodePtr OpPC,
3567 const CallExpr *Call,
3568 unsigned ID) {
3569 assert(Call->getNumArgs() == 3);
3570
3571 APSInt ImmAPS;
3572 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: ImmAPS))
3573 return false;
3574 uint64_t Index = ImmAPS.getZExtValue();
3575
3576 const Pointer &SubVec = S.Stk.pop<Pointer>();
3577 if (!SubVec.getFieldDesc()->isPrimitiveArray())
3578 return false;
3579
3580 const Pointer &BaseVec = S.Stk.pop<Pointer>();
3581 if (!BaseVec.getFieldDesc()->isPrimitiveArray())
3582 return false;
3583
3584 const Pointer &Dst = S.Stk.peek<Pointer>();
3585
3586 unsigned BaseElements = BaseVec.getNumElems();
3587 unsigned SubElements = SubVec.getNumElems();
3588
3589 assert(SubElements != 0 && BaseElements != 0 &&
3590 (BaseElements % SubElements) == 0);
3591
3592 unsigned NumLanes = BaseElements / SubElements;
3593 unsigned Lane = static_cast<unsigned>(Index % NumLanes);
3594 unsigned InsertPos = Lane * SubElements;
3595
3596 PrimType ElemT = BaseVec.getFieldDesc()->getPrimType();
3597
3598 TYPE_SWITCH(ElemT, {
3599 for (unsigned I = 0; I != BaseElements; ++I)
3600 Dst.elem<T>(I) = BaseVec.elem<T>(I);
3601 for (unsigned I = 0; I != SubElements; ++I)
3602 Dst.elem<T>(InsertPos + I) = SubVec.elem<T>(I);
3603 });
3604
3605 Dst.initializeAllElements();
3606 return true;
3607}
3608
3609static bool interp__builtin_ia32_phminposuw(InterpState &S, CodePtr OpPC,
3610 const CallExpr *Call) {
3611 assert(Call->getNumArgs() == 1);
3612
3613 const Pointer &Source = S.Stk.pop<Pointer>();
3614 const Pointer &Dest = S.Stk.peek<Pointer>();
3615
3616 unsigned SourceLen = Source.getNumElems();
3617 QualType ElemQT = getElemType(P: Source);
3618 OptPrimType ElemT = S.getContext().classify(T: ElemQT);
3619 unsigned ElemBitWidth = S.getASTContext().getTypeSize(T: ElemQT);
3620
3621 bool DestUnsigned = Call->getCallReturnType(Ctx: S.getASTContext())
3622 ->castAs<VectorType>()
3623 ->getElementType()
3624 ->isUnsignedIntegerOrEnumerationType();
3625
3626 INT_TYPE_SWITCH_NO_BOOL(*ElemT, {
3627 APSInt MinIndex(ElemBitWidth, DestUnsigned);
3628 APSInt MinVal = Source.elem<T>(0).toAPSInt();
3629
3630 for (unsigned I = 1; I != SourceLen; ++I) {
3631 APSInt Val = Source.elem<T>(I).toAPSInt();
3632 if (MinVal.ugt(Val)) {
3633 MinVal = Val;
3634 MinIndex = I;
3635 }
3636 }
3637
3638 Dest.elem<T>(0) = static_cast<T>(MinVal);
3639 Dest.elem<T>(1) = static_cast<T>(MinIndex);
3640 for (unsigned I = 2; I != SourceLen; ++I) {
3641 Dest.elem<T>(I) = static_cast<T>(APSInt(ElemBitWidth, DestUnsigned));
3642 }
3643 });
3644 Dest.initializeAllElements();
3645 return true;
3646}
3647
3648static bool interp__builtin_ia32_pternlog(InterpState &S, CodePtr OpPC,
3649 const CallExpr *Call, bool MaskZ) {
3650 assert(Call->getNumArgs() == 5);
3651
3652 APSInt UVal;
3653 if (!popToAPSInt(S, E: Call->getArg(Arg: 4), Out&: UVal))
3654 return false;
3655 APInt U = UVal; // Lane mask
3656 APSInt ImmVal;
3657 if (!popToAPSInt(S, E: Call->getArg(Arg: 3), Out&: ImmVal))
3658 return false;
3659 APInt Imm = ImmVal; // Ternary truth table
3660 const Pointer &C = S.Stk.pop<Pointer>();
3661 const Pointer &B = S.Stk.pop<Pointer>();
3662 const Pointer &A = S.Stk.pop<Pointer>();
3663 const Pointer &Dst = S.Stk.peek<Pointer>();
3664
3665 unsigned DstLen = A.getNumElems();
3666 QualType ElemQT = getElemType(P: A);
3667 OptPrimType ElemT = S.getContext().classify(T: ElemQT);
3668 unsigned LaneWidth = S.getASTContext().getTypeSize(T: ElemQT);
3669 bool DstUnsigned = ElemQT->isUnsignedIntegerOrEnumerationType();
3670
3671 INT_TYPE_SWITCH_NO_BOOL(*ElemT, {
3672 for (unsigned I = 0; I != DstLen; ++I) {
3673 APInt ALane = A.elem<T>(I).toAPSInt();
3674 APInt BLane = B.elem<T>(I).toAPSInt();
3675 APInt CLane = C.elem<T>(I).toAPSInt();
3676 APInt RLane(LaneWidth, 0);
3677 if (U[I]) { // If lane not masked, compute ternary logic.
3678 for (unsigned Bit = 0; Bit != LaneWidth; ++Bit) {
3679 unsigned ABit = ALane[Bit];
3680 unsigned BBit = BLane[Bit];
3681 unsigned CBit = CLane[Bit];
3682 unsigned Idx = (ABit << 2) | (BBit << 1) | (CBit);
3683 RLane.setBitVal(Bit, Imm[Idx]);
3684 }
3685 Dst.elem<T>(I) = static_cast<T>(APSInt(RLane, DstUnsigned));
3686 } else if (MaskZ) { // If zero masked, zero the lane.
3687 Dst.elem<T>(I) = static_cast<T>(APSInt(RLane, DstUnsigned));
3688 } else { // Just masked, put in A lane.
3689 Dst.elem<T>(I) = static_cast<T>(APSInt(ALane, DstUnsigned));
3690 }
3691 }
3692 });
3693 Dst.initializeAllElements();
3694 return true;
3695}
3696
3697static bool interp__builtin_ia32_vec_ext(InterpState &S, CodePtr OpPC,
3698 const CallExpr *Call, unsigned ID) {
3699 assert(Call->getNumArgs() == 2);
3700
3701 APSInt ImmAPS;
3702 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: ImmAPS))
3703 return false;
3704 const Pointer &Vec = S.Stk.pop<Pointer>();
3705 if (!Vec.getFieldDesc()->isPrimitiveArray())
3706 return false;
3707
3708 unsigned NumElems = Vec.getNumElems();
3709 unsigned Index =
3710 static_cast<unsigned>(ImmAPS.getZExtValue() & (NumElems - 1));
3711
3712 PrimType ElemT = Vec.getFieldDesc()->getPrimType();
3713 // FIXME(#161685): Replace float+int split with a numeric-only type switch
3714 if (ElemT == PT_Float) {
3715 S.Stk.push<Floating>(Args&: Vec.elem<Floating>(I: Index));
3716 return true;
3717 }
3718 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3719 APSInt V = Vec.elem<T>(Index).toAPSInt();
3720 pushInteger(S, V, Call->getType());
3721 });
3722
3723 return true;
3724}
3725
3726static bool interp__builtin_ia32_vec_set(InterpState &S, CodePtr OpPC,
3727 const CallExpr *Call, unsigned ID) {
3728 assert(Call->getNumArgs() == 3);
3729
3730 APSInt ImmAPS;
3731 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: ImmAPS))
3732 return false;
3733 APSInt ValAPS;
3734 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: ValAPS))
3735 return false;
3736
3737 const Pointer &Base = S.Stk.pop<Pointer>();
3738 if (!Base.getFieldDesc()->isPrimitiveArray())
3739 return false;
3740
3741 const Pointer &Dst = S.Stk.peek<Pointer>();
3742
3743 unsigned NumElems = Base.getNumElems();
3744 unsigned Index =
3745 static_cast<unsigned>(ImmAPS.getZExtValue() & (NumElems - 1));
3746
3747 PrimType ElemT = Base.getFieldDesc()->getPrimType();
3748 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3749 for (unsigned I = 0; I != NumElems; ++I)
3750 Dst.elem<T>(I) = Base.elem<T>(I);
3751 Dst.elem<T>(Index) = static_cast<T>(ValAPS);
3752 });
3753
3754 Dst.initializeAllElements();
3755 return true;
3756}
3757
3758static bool evalICmpImm(uint8_t Imm, const APSInt &A, const APSInt &B,
3759 bool IsUnsigned) {
3760 switch (Imm & 0x7) {
3761 case 0x00: // _MM_CMPINT_EQ
3762 return (A == B);
3763 case 0x01: // _MM_CMPINT_LT
3764 return IsUnsigned ? A.ult(RHS: B) : A.slt(RHS: B);
3765 case 0x02: // _MM_CMPINT_LE
3766 return IsUnsigned ? A.ule(RHS: B) : A.sle(RHS: B);
3767 case 0x03: // _MM_CMPINT_FALSE
3768 return false;
3769 case 0x04: // _MM_CMPINT_NE
3770 return (A != B);
3771 case 0x05: // _MM_CMPINT_NLT
3772 return IsUnsigned ? A.uge(RHS: B) : A.sge(RHS: B);
3773 case 0x06: // _MM_CMPINT_NLE
3774 return IsUnsigned ? A.ugt(RHS: B) : A.sgt(RHS: B);
3775 case 0x07: // _MM_CMPINT_TRUE
3776 return true;
3777 default:
3778 llvm_unreachable("Invalid Op");
3779 }
3780}
3781
3782static bool interp__builtin_ia32_cmp_mask(InterpState &S, CodePtr OpPC,
3783 const CallExpr *Call, unsigned ID,
3784 bool IsUnsigned) {
3785 assert(Call->getNumArgs() == 4);
3786
3787 APSInt Mask;
3788 if (!popToAPSInt(S, E: Call->getArg(Arg: 3), Out&: Mask))
3789 return false;
3790 APSInt Opcode;
3791 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: Opcode))
3792 return false;
3793 unsigned CmpOp = static_cast<unsigned>(Opcode.getZExtValue());
3794 const Pointer &RHS = S.Stk.pop<Pointer>();
3795 const Pointer &LHS = S.Stk.pop<Pointer>();
3796
3797 assert(LHS.getNumElems() == RHS.getNumElems());
3798
3799 APInt RetMask = APInt::getZero(numBits: LHS.getNumElems());
3800 unsigned VectorLen = LHS.getNumElems();
3801 PrimType ElemT = LHS.getFieldDesc()->getPrimType();
3802
3803 for (unsigned ElemNum = 0; ElemNum < VectorLen; ++ElemNum) {
3804 APSInt A, B;
3805 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3806 A = LHS.elem<T>(ElemNum).toAPSInt();
3807 B = RHS.elem<T>(ElemNum).toAPSInt();
3808 });
3809 RetMask.setBitVal(BitPosition: ElemNum,
3810 BitValue: Mask[ElemNum] && evalICmpImm(Imm: CmpOp, A, B, IsUnsigned));
3811 }
3812 pushInteger(S, Val: RetMask, QT: Call->getType());
3813 return true;
3814}
3815
3816static bool interp__builtin_ia32_vpconflict(InterpState &S, CodePtr OpPC,
3817 const CallExpr *Call) {
3818 assert(Call->getNumArgs() == 1);
3819
3820 QualType Arg0Type = Call->getArg(Arg: 0)->getType();
3821 const auto *VecT = Arg0Type->castAs<VectorType>();
3822 PrimType ElemT = *S.getContext().classify(T: VecT->getElementType());
3823 unsigned NumElems = VecT->getNumElements();
3824 bool DestUnsigned = Call->getType()->isUnsignedIntegerOrEnumerationType();
3825 const Pointer &Src = S.Stk.pop<Pointer>();
3826 const Pointer &Dst = S.Stk.peek<Pointer>();
3827
3828 for (unsigned I = 0; I != NumElems; ++I) {
3829 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
3830 APSInt ElemI = Src.elem<T>(I).toAPSInt();
3831 APInt ConflictMask(ElemI.getBitWidth(), 0);
3832 for (unsigned J = 0; J != I; ++J) {
3833 APSInt ElemJ = Src.elem<T>(J).toAPSInt();
3834 ConflictMask.setBitVal(J, ElemI == ElemJ);
3835 }
3836 Dst.elem<T>(I) = static_cast<T>(APSInt(ConflictMask, DestUnsigned));
3837 });
3838 }
3839 Dst.initializeAllElements();
3840 return true;
3841}
3842
3843static bool interp__builtin_ia32_cvt_vec2mask(InterpState &S, CodePtr OpPC,
3844 const CallExpr *Call,
3845 unsigned ID) {
3846 assert(Call->getNumArgs() == 1);
3847
3848 const Pointer &Vec = S.Stk.pop<Pointer>();
3849 unsigned RetWidth = S.getASTContext().getIntWidth(T: Call->getType());
3850 APInt RetMask(RetWidth, 0);
3851
3852 unsigned VectorLen = Vec.getNumElems();
3853 PrimType ElemT = Vec.getFieldDesc()->getPrimType();
3854
3855 for (unsigned ElemNum = 0; ElemNum != VectorLen; ++ElemNum) {
3856 APSInt A;
3857 INT_TYPE_SWITCH_NO_BOOL(ElemT, { A = Vec.elem<T>(ElemNum).toAPSInt(); });
3858 unsigned MSB = A[A.getBitWidth() - 1];
3859 RetMask.setBitVal(BitPosition: ElemNum, BitValue: MSB);
3860 }
3861 pushInteger(S, Val: RetMask, QT: Call->getType());
3862 return true;
3863}
3864
3865static bool interp__builtin_ia32_cvt_mask2vec(InterpState &S, CodePtr OpPC,
3866 const CallExpr *Call,
3867 unsigned ID) {
3868 assert(Call->getNumArgs() == 1);
3869
3870 APSInt Mask;
3871 if (!popToAPSInt(S, E: Call->getArg(Arg: 0), Out&: Mask))
3872 return false;
3873
3874 const Pointer &Vec = S.Stk.peek<Pointer>();
3875 unsigned NumElems = Vec.getNumElems();
3876 PrimType ElemT = Vec.getFieldDesc()->getPrimType();
3877
3878 for (unsigned I = 0; I != NumElems; ++I) {
3879 bool BitSet = Mask[I];
3880
3881 INT_TYPE_SWITCH_NO_BOOL(
3882 ElemT, { Vec.elem<T>(I) = BitSet ? T::from(-1) : T::from(0); });
3883 }
3884
3885 Vec.initializeAllElements();
3886
3887 return true;
3888}
3889
3890static bool interp__builtin_ia32_cvtsd2ss(InterpState &S, CodePtr OpPC,
3891 const CallExpr *Call,
3892 bool HasRoundingMask) {
3893 APSInt Rounding, MaskInt;
3894 Pointer Src, B, A;
3895
3896 if (HasRoundingMask) {
3897 assert(Call->getNumArgs() == 5);
3898 if (!popToAPSInt(S, E: Call->getArg(Arg: 4), Out&: Rounding))
3899 return false;
3900 if (!popToAPSInt(S, E: Call->getArg(Arg: 3), Out&: MaskInt))
3901 return false;
3902 Src = S.Stk.pop<Pointer>();
3903 B = S.Stk.pop<Pointer>();
3904 A = S.Stk.pop<Pointer>();
3905 if (!CheckLoad(S, OpPC, Ptr: A) || !CheckLoad(S, OpPC, Ptr: B) ||
3906 !CheckLoad(S, OpPC, Ptr: Src))
3907 return false;
3908 } else {
3909 assert(Call->getNumArgs() == 2);
3910 B = S.Stk.pop<Pointer>();
3911 A = S.Stk.pop<Pointer>();
3912 if (!CheckLoad(S, OpPC, Ptr: A) || !CheckLoad(S, OpPC, Ptr: B))
3913 return false;
3914 }
3915
3916 const auto *DstVTy = Call->getType()->castAs<VectorType>();
3917 unsigned NumElems = DstVTy->getNumElements();
3918 const Pointer &Dst = S.Stk.peek<Pointer>();
3919
3920 // Copy all elements except lane 0 (overwritten below) from A to Dst.
3921 for (unsigned I = 1; I != NumElems; ++I)
3922 Dst.elem<Floating>(I) = A.elem<Floating>(I);
3923
3924 // Convert element 0 from double to float, or use Src if masked off.
3925 if (!HasRoundingMask || (MaskInt.getZExtValue() & 0x1)) {
3926 assert(S.getASTContext().FloatTy == DstVTy->getElementType() &&
3927 "cvtsd2ss requires float element type in destination vector");
3928
3929 Floating Conv = S.allocFloat(
3930 Sem: S.getASTContext().getFloatTypeSemantics(T: DstVTy->getElementType()));
3931 APFloat SrcVal = B.elem<Floating>(I: 0).getAPFloat();
3932 if (!convertDoubleToFloatStrict(Src: SrcVal, Dst&: Conv, S, DiagExpr: Call))
3933 return false;
3934 Dst.elem<Floating>(I: 0) = Conv;
3935 } else {
3936 Dst.elem<Floating>(I: 0) = Src.elem<Floating>(I: 0);
3937 }
3938
3939 Dst.initializeAllElements();
3940 return true;
3941}
3942
3943static bool interp__builtin_ia32_cvtpd2ps(InterpState &S, CodePtr OpPC,
3944 const CallExpr *Call, bool IsMasked,
3945 bool HasRounding) {
3946 APSInt MaskVal;
3947 Pointer PassThrough;
3948 Pointer Src;
3949 APSInt Rounding;
3950
3951 if (IsMasked) {
3952 // Pop in reverse order.
3953 if (HasRounding) {
3954 if (!popToAPSInt(S, E: Call->getArg(Arg: 3), Out&: Rounding))
3955 return false;
3956 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: MaskVal))
3957 return false;
3958 PassThrough = S.Stk.pop<Pointer>();
3959 Src = S.Stk.pop<Pointer>();
3960 } else {
3961 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: MaskVal))
3962 return false;
3963 PassThrough = S.Stk.pop<Pointer>();
3964 Src = S.Stk.pop<Pointer>();
3965 }
3966
3967 if (!CheckLoad(S, OpPC, Ptr: PassThrough))
3968 return false;
3969 } else {
3970 // Pop source only.
3971 Src = S.Stk.pop<Pointer>();
3972 }
3973
3974 if (!CheckLoad(S, OpPC, Ptr: Src))
3975 return false;
3976
3977 const auto *RetVTy = Call->getType()->castAs<VectorType>();
3978 unsigned RetElems = RetVTy->getNumElements();
3979 unsigned SrcElems = Src.getNumElems();
3980 const Pointer &Dst = S.Stk.peek<Pointer>();
3981
3982 // Initialize destination with passthrough or zeros.
3983 for (unsigned I = 0; I != RetElems; ++I)
3984 if (IsMasked)
3985 Dst.elem<Floating>(I) = PassThrough.elem<Floating>(I);
3986 else
3987 Dst.elem<Floating>(I) = Floating(APFloat(0.0f));
3988
3989 assert(S.getASTContext().FloatTy == RetVTy->getElementType() &&
3990 "cvtpd2ps requires float element type in return vector");
3991
3992 // Convert double to float for enabled elements (only process source elements
3993 // that exist).
3994 for (unsigned I = 0; I != SrcElems; ++I) {
3995 if (IsMasked && !MaskVal[I])
3996 continue;
3997
3998 APFloat SrcVal = Src.elem<Floating>(I).getAPFloat();
3999
4000 Floating Conv = S.allocFloat(
4001 Sem: S.getASTContext().getFloatTypeSemantics(T: RetVTy->getElementType()));
4002 if (!convertDoubleToFloatStrict(Src: SrcVal, Dst&: Conv, S, DiagExpr: Call))
4003 return false;
4004 Dst.elem<Floating>(I) = Conv;
4005 }
4006
4007 Dst.initializeAllElements();
4008 return true;
4009}
4010
4011static bool interp__builtin_ia32_shuffle_generic(
4012 InterpState &S, CodePtr OpPC, const CallExpr *Call,
4013 llvm::function_ref<std::pair<unsigned, int>(unsigned, const APInt &)>
4014 GetSourceIndex) {
4015
4016 assert(Call->getNumArgs() == 2 || Call->getNumArgs() == 3);
4017
4018 APInt ShuffleMask;
4019 Pointer A, MaskVector, B;
4020 bool IsVectorMask = false;
4021 bool IsSingleOperand = (Call->getNumArgs() == 2);
4022
4023 if (IsSingleOperand) {
4024 QualType MaskType = Call->getArg(Arg: 1)->getType();
4025 if (MaskType->isVectorType()) {
4026 IsVectorMask = true;
4027 MaskVector = S.Stk.pop<Pointer>();
4028 A = S.Stk.pop<Pointer>();
4029 B = A;
4030 } else if (MaskType->isIntegerType()) {
4031 APSInt MaskVal;
4032 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: MaskVal))
4033 return false;
4034 ShuffleMask = MaskVal;
4035 A = S.Stk.pop<Pointer>();
4036 B = A;
4037 } else {
4038 return false;
4039 }
4040 } else {
4041 QualType Arg2Type = Call->getArg(Arg: 2)->getType();
4042 if (Arg2Type->isVectorType()) {
4043 IsVectorMask = true;
4044 B = S.Stk.pop<Pointer>();
4045 MaskVector = S.Stk.pop<Pointer>();
4046 A = S.Stk.pop<Pointer>();
4047 } else if (Arg2Type->isIntegerType()) {
4048 APSInt MaskVal;
4049 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: MaskVal))
4050 return false;
4051 ShuffleMask = MaskVal;
4052 B = S.Stk.pop<Pointer>();
4053 A = S.Stk.pop<Pointer>();
4054 } else {
4055 return false;
4056 }
4057 }
4058
4059 QualType Arg0Type = Call->getArg(Arg: 0)->getType();
4060 const auto *VecT = Arg0Type->castAs<VectorType>();
4061 PrimType ElemT = *S.getContext().classify(T: VecT->getElementType());
4062 unsigned NumElems = VecT->getNumElements();
4063
4064 const Pointer &Dst = S.Stk.peek<Pointer>();
4065
4066 PrimType MaskElemT = PT_Uint32;
4067 if (IsVectorMask) {
4068 QualType Arg1Type = Call->getArg(Arg: 1)->getType();
4069 const auto *MaskVecT = Arg1Type->castAs<VectorType>();
4070 QualType MaskElemType = MaskVecT->getElementType();
4071 MaskElemT = *S.getContext().classify(T: MaskElemType);
4072 }
4073
4074 for (unsigned DstIdx = 0; DstIdx != NumElems; ++DstIdx) {
4075 if (IsVectorMask) {
4076 INT_TYPE_SWITCH(MaskElemT,
4077 { ShuffleMask = MaskVector.elem<T>(DstIdx).toAPSInt(); });
4078 }
4079
4080 auto [SrcVecIdx, SrcIdx] = GetSourceIndex(DstIdx, ShuffleMask);
4081
4082 if (SrcIdx < 0) {
4083 // Zero out this element
4084 if (ElemT == PT_Float) {
4085 Dst.elem<Floating>(I: DstIdx) = Floating(
4086 S.getASTContext().getFloatTypeSemantics(T: VecT->getElementType()));
4087 } else {
4088 INT_TYPE_SWITCH_NO_BOOL(ElemT, { Dst.elem<T>(DstIdx) = T::from(0); });
4089 }
4090 } else {
4091 const Pointer &Src = (SrcVecIdx == 0) ? A : B;
4092 TYPE_SWITCH(ElemT, { Dst.elem<T>(DstIdx) = Src.elem<T>(SrcIdx); });
4093 }
4094 }
4095 Dst.initializeAllElements();
4096
4097 return true;
4098}
4099
4100static bool interp__builtin_ia32_shuffle_generic(
4101 InterpState &S, CodePtr OpPC, const CallExpr *Call,
4102 llvm::function_ref<std::pair<unsigned, int>(unsigned, unsigned)>
4103 GetSourceIndex) {
4104 return interp__builtin_ia32_shuffle_generic(
4105 S, OpPC, Call,
4106 GetSourceIndex: [&GetSourceIndex](unsigned DstIdx,
4107 const APInt &Mask) -> std::pair<unsigned, int> {
4108 return GetSourceIndex(DstIdx, Mask.getZExtValue());
4109 });
4110}
4111
4112static bool interp__builtin_ia32_shift_with_count(
4113 InterpState &S, CodePtr OpPC, const CallExpr *Call,
4114 llvm::function_ref<APInt(const APInt &, uint64_t)> ShiftOp,
4115 llvm::function_ref<APInt(const APInt &, unsigned)> OverflowOp) {
4116
4117 assert(Call->getNumArgs() == 2);
4118
4119 const Pointer &Count = S.Stk.pop<Pointer>();
4120 const Pointer &Source = S.Stk.pop<Pointer>();
4121
4122 QualType SourceType = Call->getArg(Arg: 0)->getType();
4123 QualType CountType = Call->getArg(Arg: 1)->getType();
4124 assert(SourceType->isVectorType() && CountType->isVectorType());
4125
4126 const auto *SourceVecT = SourceType->castAs<VectorType>();
4127 const auto *CountVecT = CountType->castAs<VectorType>();
4128 PrimType SourceElemT = *S.getContext().classify(T: SourceVecT->getElementType());
4129 PrimType CountElemT = *S.getContext().classify(T: CountVecT->getElementType());
4130
4131 const Pointer &Dst = S.Stk.peek<Pointer>();
4132
4133 unsigned DestEltWidth =
4134 S.getASTContext().getTypeSize(T: SourceVecT->getElementType());
4135 bool IsDestUnsigned = SourceVecT->getElementType()->isUnsignedIntegerType();
4136 unsigned DestLen = SourceVecT->getNumElements();
4137 unsigned CountEltWidth =
4138 S.getASTContext().getTypeSize(T: CountVecT->getElementType());
4139 unsigned NumBitsInQWord = 64;
4140 unsigned NumCountElts = NumBitsInQWord / CountEltWidth;
4141
4142 uint64_t CountLQWord = 0;
4143 for (unsigned EltIdx = 0; EltIdx != NumCountElts; ++EltIdx) {
4144 uint64_t Elt = 0;
4145 INT_TYPE_SWITCH(CountElemT,
4146 { Elt = static_cast<uint64_t>(Count.elem<T>(EltIdx)); });
4147 CountLQWord |= (Elt << (EltIdx * CountEltWidth));
4148 }
4149
4150 for (unsigned EltIdx = 0; EltIdx != DestLen; ++EltIdx) {
4151 APSInt Elt;
4152 INT_TYPE_SWITCH(SourceElemT, { Elt = Source.elem<T>(EltIdx).toAPSInt(); });
4153
4154 APInt Result;
4155 if (CountLQWord < DestEltWidth) {
4156 Result = ShiftOp(Elt, CountLQWord);
4157 } else {
4158 Result = OverflowOp(Elt, DestEltWidth);
4159 }
4160 if (IsDestUnsigned) {
4161 INT_TYPE_SWITCH(SourceElemT, {
4162 Dst.elem<T>(EltIdx) = T::from(Result.getZExtValue());
4163 });
4164 } else {
4165 INT_TYPE_SWITCH(SourceElemT, {
4166 Dst.elem<T>(EltIdx) = T::from(Result.getSExtValue());
4167 });
4168 }
4169 }
4170
4171 Dst.initializeAllElements();
4172 return true;
4173}
4174
4175static bool interp__builtin_ia32_shufbitqmb_mask(InterpState &S, CodePtr OpPC,
4176 const CallExpr *Call) {
4177
4178 assert(Call->getNumArgs() == 3);
4179
4180 QualType SourceType = Call->getArg(Arg: 0)->getType();
4181 QualType ShuffleMaskType = Call->getArg(Arg: 1)->getType();
4182 QualType ZeroMaskType = Call->getArg(Arg: 2)->getType();
4183 if (!SourceType->isVectorType() || !ShuffleMaskType->isVectorType() ||
4184 !ZeroMaskType->isIntegerType()) {
4185 return false;
4186 }
4187
4188 Pointer Source, ShuffleMask;
4189 APSInt ZeroMask;
4190 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: ZeroMask))
4191 return false;
4192 ShuffleMask = S.Stk.pop<Pointer>();
4193 Source = S.Stk.pop<Pointer>();
4194
4195 const auto *SourceVecT = SourceType->castAs<VectorType>();
4196 const auto *ShuffleMaskVecT = ShuffleMaskType->castAs<VectorType>();
4197 assert(SourceVecT->getNumElements() == ShuffleMaskVecT->getNumElements());
4198 assert(ZeroMask.getBitWidth() == SourceVecT->getNumElements());
4199
4200 PrimType SourceElemT = *S.getContext().classify(T: SourceVecT->getElementType());
4201 PrimType ShuffleMaskElemT =
4202 *S.getContext().classify(T: ShuffleMaskVecT->getElementType());
4203
4204 unsigned NumBytesInQWord = 8;
4205 unsigned NumBitsInByte = 8;
4206 unsigned NumBytes = SourceVecT->getNumElements();
4207 unsigned NumQWords = NumBytes / NumBytesInQWord;
4208 unsigned RetWidth = ZeroMask.getBitWidth();
4209 APSInt RetMask(llvm::APInt(RetWidth, 0), /*isUnsigned=*/true);
4210
4211 for (unsigned QWordId = 0; QWordId != NumQWords; ++QWordId) {
4212 APInt SourceQWord(64, 0);
4213 for (unsigned ByteIdx = 0; ByteIdx != NumBytesInQWord; ++ByteIdx) {
4214 uint64_t Byte = 0;
4215 INT_TYPE_SWITCH(SourceElemT, {
4216 Byte = static_cast<uint64_t>(
4217 Source.elem<T>(QWordId * NumBytesInQWord + ByteIdx));
4218 });
4219 SourceQWord.insertBits(SubBits: APInt(8, Byte & 0xFF), bitPosition: ByteIdx * NumBitsInByte);
4220 }
4221
4222 for (unsigned ByteIdx = 0; ByteIdx != NumBytesInQWord; ++ByteIdx) {
4223 unsigned SelIdx = QWordId * NumBytesInQWord + ByteIdx;
4224 unsigned M = 0;
4225 INT_TYPE_SWITCH(ShuffleMaskElemT, {
4226 M = static_cast<unsigned>(ShuffleMask.elem<T>(SelIdx)) & 0x3F;
4227 });
4228
4229 if (ZeroMask[SelIdx]) {
4230 RetMask.setBitVal(BitPosition: SelIdx, BitValue: SourceQWord[M]);
4231 }
4232 }
4233 }
4234
4235 pushInteger(S, Val: RetMask, QT: Call->getType());
4236 return true;
4237}
4238
4239static bool interp__builtin_ia32_vcvtps2ph(InterpState &S, CodePtr OpPC,
4240 const CallExpr *Call) {
4241 // Arguments are: vector of floats, rounding immediate
4242 assert(Call->getNumArgs() == 2);
4243
4244 APSInt Imm;
4245 if (!popToAPSInt(S, E: Call->getArg(Arg: 1), Out&: Imm))
4246 return false;
4247 const Pointer &Src = S.Stk.pop<Pointer>();
4248 const Pointer &Dst = S.Stk.peek<Pointer>();
4249
4250 assert(Src.getFieldDesc()->isPrimitiveArray());
4251 assert(Dst.getFieldDesc()->isPrimitiveArray());
4252
4253 const auto *SrcVTy = Call->getArg(Arg: 0)->getType()->castAs<VectorType>();
4254 unsigned SrcNumElems = SrcVTy->getNumElements();
4255 const auto *DstVTy = Call->getType()->castAs<VectorType>();
4256 unsigned DstNumElems = DstVTy->getNumElements();
4257
4258 const llvm::fltSemantics &HalfSem =
4259 S.getASTContext().getFloatTypeSemantics(T: S.getASTContext().HalfTy);
4260
4261 // imm[2] == 1 means use MXCSR rounding mode.
4262 // In that case, we can only evaluate if the conversion is exact.
4263 int ImmVal = Imm.getZExtValue();
4264 bool UseMXCSR = (ImmVal & 4) != 0;
4265 bool IsFPConstrained =
4266 Call->getFPFeaturesInEffect(LO: S.getASTContext().getLangOpts())
4267 .isFPConstrained();
4268
4269 llvm::RoundingMode RM;
4270 if (!UseMXCSR) {
4271 switch (ImmVal & 3) {
4272 case 0:
4273 RM = llvm::RoundingMode::NearestTiesToEven;
4274 break;
4275 case 1:
4276 RM = llvm::RoundingMode::TowardNegative;
4277 break;
4278 case 2:
4279 RM = llvm::RoundingMode::TowardPositive;
4280 break;
4281 case 3:
4282 RM = llvm::RoundingMode::TowardZero;
4283 break;
4284 default:
4285 llvm_unreachable("Invalid immediate rounding mode");
4286 }
4287 } else {
4288 // For MXCSR, we must check for exactness. We can use any rounding mode
4289 // for the trial conversion since the result is the same if it's exact.
4290 RM = llvm::RoundingMode::NearestTiesToEven;
4291 }
4292
4293 QualType DstElemQT = Dst.getFieldDesc()->getElemQualType();
4294 PrimType DstElemT = *S.getContext().classify(T: DstElemQT);
4295
4296 for (unsigned I = 0; I != SrcNumElems; ++I) {
4297 Floating SrcVal = Src.elem<Floating>(I);
4298 APFloat DstVal = SrcVal.getAPFloat();
4299
4300 bool LostInfo;
4301 APFloat::opStatus St = DstVal.convert(ToSemantics: HalfSem, RM, losesInfo: &LostInfo);
4302
4303 if (UseMXCSR && IsFPConstrained && St != APFloat::opOK) {
4304 S.FFDiag(SI: S.Current->getSource(PC: OpPC),
4305 DiagId: diag::note_constexpr_dynamic_rounding);
4306 return false;
4307 }
4308
4309 INT_TYPE_SWITCH_NO_BOOL(DstElemT, {
4310 // Convert the destination value's bit pattern to an unsigned integer,
4311 // then reconstruct the element using the target type's 'from' method.
4312 uint64_t RawBits = DstVal.bitcastToAPInt().getZExtValue();
4313 Dst.elem<T>(I) = T::from(RawBits);
4314 });
4315 }
4316
4317 // Zero out remaining elements if the destination has more elements
4318 // (e.g., vcvtps2ph converting 4 floats to 8 shorts).
4319 if (DstNumElems > SrcNumElems) {
4320 for (unsigned I = SrcNumElems; I != DstNumElems; ++I) {
4321 INT_TYPE_SWITCH_NO_BOOL(DstElemT, { Dst.elem<T>(I) = T::from(0); });
4322 }
4323 }
4324
4325 Dst.initializeAllElements();
4326 return true;
4327}
4328
4329static bool interp__builtin_ia32_multishiftqb(InterpState &S, CodePtr OpPC,
4330 const CallExpr *Call) {
4331 assert(Call->getNumArgs() == 2);
4332
4333 QualType ATy = Call->getArg(Arg: 0)->getType();
4334 QualType BTy = Call->getArg(Arg: 1)->getType();
4335 if (!ATy->isVectorType() || !BTy->isVectorType()) {
4336 return false;
4337 }
4338
4339 const Pointer &BPtr = S.Stk.pop<Pointer>();
4340 const Pointer &APtr = S.Stk.pop<Pointer>();
4341 const auto *AVecT = ATy->castAs<VectorType>();
4342 assert(AVecT->getNumElements() ==
4343 BTy->castAs<VectorType>()->getNumElements());
4344
4345 PrimType ElemT = *S.getContext().classify(T: AVecT->getElementType());
4346
4347 unsigned NumBytesInQWord = 8;
4348 unsigned NumBitsInByte = 8;
4349 unsigned NumBytes = AVecT->getNumElements();
4350 unsigned NumQWords = NumBytes / NumBytesInQWord;
4351 const Pointer &Dst = S.Stk.peek<Pointer>();
4352
4353 for (unsigned QWordId = 0; QWordId != NumQWords; ++QWordId) {
4354 APInt BQWord(64, 0);
4355 for (unsigned ByteIdx = 0; ByteIdx != NumBytesInQWord; ++ByteIdx) {
4356 unsigned Idx = QWordId * NumBytesInQWord + ByteIdx;
4357 INT_TYPE_SWITCH(ElemT, {
4358 uint64_t Byte = static_cast<uint64_t>(BPtr.elem<T>(Idx));
4359 BQWord.insertBits(APInt(8, Byte & 0xFF), ByteIdx * NumBitsInByte);
4360 });
4361 }
4362
4363 for (unsigned ByteIdx = 0; ByteIdx != NumBytesInQWord; ++ByteIdx) {
4364 unsigned Idx = QWordId * NumBytesInQWord + ByteIdx;
4365 uint64_t Ctrl = 0;
4366 INT_TYPE_SWITCH(
4367 ElemT, { Ctrl = static_cast<uint64_t>(APtr.elem<T>(Idx)) & 0x3F; });
4368
4369 APInt Byte(8, 0);
4370 for (unsigned BitIdx = 0; BitIdx != NumBitsInByte; ++BitIdx) {
4371 Byte.setBitVal(BitPosition: BitIdx, BitValue: BQWord[(Ctrl + BitIdx) & 0x3F]);
4372 }
4373 INT_TYPE_SWITCH(ElemT,
4374 { Dst.elem<T>(Idx) = T::from(Byte.getZExtValue()); });
4375 }
4376 }
4377
4378 Dst.initializeAllElements();
4379
4380 return true;
4381}
4382
4383static bool interp__builtin_ia32_gfni_affine(InterpState &S, CodePtr OpPC,
4384 const CallExpr *Call,
4385 bool Inverse) {
4386 assert(Call->getNumArgs() == 3);
4387 QualType XType = Call->getArg(Arg: 0)->getType();
4388 QualType AType = Call->getArg(Arg: 1)->getType();
4389 QualType ImmType = Call->getArg(Arg: 2)->getType();
4390 if (!XType->isVectorType() || !AType->isVectorType() ||
4391 !ImmType->isIntegerType()) {
4392 return false;
4393 }
4394
4395 Pointer X, A;
4396 APSInt Imm;
4397 if (!popToAPSInt(S, E: Call->getArg(Arg: 2), Out&: Imm))
4398 return false;
4399 A = S.Stk.pop<Pointer>();
4400 X = S.Stk.pop<Pointer>();
4401
4402 const Pointer &Dst = S.Stk.peek<Pointer>();
4403 const auto *AVecT = AType->castAs<VectorType>();
4404 assert(XType->castAs<VectorType>()->getNumElements() ==
4405 AVecT->getNumElements());
4406 unsigned NumBytesInQWord = 8;
4407 unsigned NumBytes = AVecT->getNumElements();
4408 unsigned NumBitsInQWord = 64;
4409 unsigned NumQWords = NumBytes / NumBytesInQWord;
4410 unsigned NumBitsInByte = 8;
4411 PrimType AElemT = *S.getContext().classify(T: AVecT->getElementType());
4412
4413 // computing A*X + Imm
4414 for (unsigned QWordIdx = 0; QWordIdx != NumQWords; ++QWordIdx) {
4415 // Extract the QWords from X, A
4416 APInt XQWord(NumBitsInQWord, 0);
4417 APInt AQWord(NumBitsInQWord, 0);
4418 for (unsigned ByteIdx = 0; ByteIdx != NumBytesInQWord; ++ByteIdx) {
4419 unsigned Idx = QWordIdx * NumBytesInQWord + ByteIdx;
4420 uint8_t XByte;
4421 uint8_t AByte;
4422 INT_TYPE_SWITCH(AElemT, {
4423 XByte = static_cast<uint8_t>(X.elem<T>(Idx));
4424 AByte = static_cast<uint8_t>(A.elem<T>(Idx));
4425 });
4426
4427 XQWord.insertBits(SubBits: APInt(NumBitsInByte, XByte), bitPosition: ByteIdx * NumBitsInByte);
4428 AQWord.insertBits(SubBits: APInt(NumBitsInByte, AByte), bitPosition: ByteIdx * NumBitsInByte);
4429 }
4430
4431 for (unsigned ByteIdx = 0; ByteIdx != NumBytesInQWord; ++ByteIdx) {
4432 unsigned Idx = QWordIdx * NumBytesInQWord + ByteIdx;
4433 uint8_t XByte =
4434 XQWord.lshr(shiftAmt: ByteIdx * NumBitsInByte).getLoBits(numBits: 8).getZExtValue();
4435 INT_TYPE_SWITCH(AElemT, {
4436 Dst.elem<T>(Idx) = T::from(GFNIAffine(XByte, AQWord, Imm, Inverse));
4437 });
4438 }
4439 }
4440 Dst.initializeAllElements();
4441 return true;
4442}
4443
4444static bool interp__builtin_ia32_gfni_mul(InterpState &S, CodePtr OpPC,
4445 const CallExpr *Call) {
4446 assert(Call->getNumArgs() == 2);
4447
4448 QualType AType = Call->getArg(Arg: 0)->getType();
4449 QualType BType = Call->getArg(Arg: 1)->getType();
4450 if (!AType->isVectorType() || !BType->isVectorType()) {
4451 return false;
4452 }
4453
4454 Pointer A, B;
4455 B = S.Stk.pop<Pointer>();
4456 A = S.Stk.pop<Pointer>();
4457
4458 const Pointer &Dst = S.Stk.peek<Pointer>();
4459 const auto *AVecT = AType->castAs<VectorType>();
4460 assert(AVecT->getNumElements() ==
4461 BType->castAs<VectorType>()->getNumElements());
4462
4463 PrimType AElemT = *S.getContext().classify(T: AVecT->getElementType());
4464 unsigned NumBytes = A.getNumElems();
4465
4466 for (unsigned ByteIdx = 0; ByteIdx != NumBytes; ++ByteIdx) {
4467 uint8_t AByte, BByte;
4468 INT_TYPE_SWITCH(AElemT, {
4469 AByte = static_cast<uint8_t>(A.elem<T>(ByteIdx));
4470 BByte = static_cast<uint8_t>(B.elem<T>(ByteIdx));
4471 Dst.elem<T>(ByteIdx) = T::from(GFNIMul(AByte, BByte));
4472 });
4473 }
4474
4475 Dst.initializeAllElements();
4476 return true;
4477}
4478
4479static bool interp__builtin_ia32_vpdp(InterpState &S, CodePtr OpPC,
4480 const CallExpr *Call, bool IsSaturating) {
4481 assert(Call->getNumArgs() == 3);
4482
4483 QualType SrcT = Call->getArg(Arg: 0)->getType();
4484 QualType OpAT = Call->getArg(Arg: 1)->getType();
4485 QualType OpBT = Call->getArg(Arg: 2)->getType();
4486 QualType DstT = Call->getType();
4487 if (!SrcT->isVectorType() || !OpAT->isVectorType() || !OpBT->isVectorType() ||
4488 !DstT->isVectorType())
4489 return false;
4490
4491 const auto *SrcVecT = SrcT->castAs<VectorType>();
4492 const auto *OpAVecT = OpAT->castAs<VectorType>();
4493 const auto *OpBVecT = OpBT->castAs<VectorType>();
4494 const auto *DstVecT = DstT->castAs<VectorType>();
4495
4496 assert(OpAVecT->getNumElements() == OpBVecT->getNumElements());
4497
4498 unsigned NumSrcElems = SrcVecT->getNumElements();
4499 unsigned NumOperandElems = OpAVecT->getNumElements();
4500 unsigned ElemsPerLane = NumOperandElems / NumSrcElems;
4501
4502 PrimType SrcElemT = *S.getContext().classify(T: SrcVecT->getElementType());
4503 PrimType OpAElemT = *S.getContext().classify(T: OpAVecT->getElementType());
4504 PrimType OpBElemT = *S.getContext().classify(T: OpBVecT->getElementType());
4505 PrimType DstElemT = *S.getContext().classify(T: DstVecT->getElementType());
4506
4507 assert(SrcElemT == DstElemT);
4508
4509 const Pointer &OpBPtr = S.Stk.pop<Pointer>();
4510 const Pointer &OpAPtr = S.Stk.pop<Pointer>();
4511 const Pointer &SrcPtr = S.Stk.pop<Pointer>();
4512 const Pointer &Dst = S.Stk.peek<Pointer>();
4513
4514 for (unsigned I = 0; I != NumSrcElems; ++I) {
4515 APSInt Acc;
4516 INT_TYPE_SWITCH_NO_BOOL(SrcElemT, { Acc = SrcPtr.elem<T>(I).toAPSInt(); });
4517 Acc = Acc.sext(width: 64);
4518 for (unsigned J = 0; J != ElemsPerLane; ++J) {
4519 APSInt OpA, OpB;
4520 INT_TYPE_SWITCH_NO_BOOL(
4521 OpAElemT, { OpA = OpAPtr.elem<T>(ElemsPerLane * I + J).toAPSInt(); });
4522 INT_TYPE_SWITCH_NO_BOOL(
4523 OpBElemT, { OpB = OpBPtr.elem<T>(ElemsPerLane * I + J).toAPSInt(); });
4524 OpA = APSInt(OpA.extend(width: 64), false);
4525 OpB = APSInt(OpB.extend(width: 64), false);
4526 Acc += OpA * OpB;
4527 }
4528 if (IsSaturating)
4529 Acc = APSInt(Acc.truncSSat(width: 32), false);
4530 else
4531 Acc = APSInt(Acc.trunc(width: 32), false);
4532 INT_TYPE_SWITCH_NO_BOOL(DstElemT,
4533 { Dst.elem<T>(I) = static_cast<T>(Acc); });
4534 }
4535 Dst.initializeAllElements();
4536 return true;
4537}
4538
4539// Bit Matrix Multiply and Accumulate (AVX512BMM). Each 256-bit lane holds a
4540// 16x16 bit matrix as 16 x i16 elements; element i is row i and bit j of that
4541// element is entry [i][j]. The accumulator (third argument, src1 in the AMD
4542// ISA) provides the initial value of each result bit, into which the bit-matrix
4543// product of the first two arguments (src2 * src3) is reduced with OR (vbmacor)
4544// or XOR (vbmacxor):
4545// for i in 0..15, j in 0..15:
4546// bit = C[16*i+j]
4547// for k in 0..15: bit OP= A[16*i+k] & B[16*k+j]
4548// dest[16*i+j] = bit
4549static bool interp__builtin_ia32_bmac(InterpState &S, CodePtr OpPC,
4550 const CallExpr *Call, bool IsXor) {
4551 assert(Call->getNumArgs() == 3);
4552
4553 // AST-based type checks before popping the stack.
4554 QualType AType = Call->getArg(Arg: 0)->getType();
4555 QualType BType = Call->getArg(Arg: 1)->getType();
4556 QualType CType = Call->getArg(Arg: 2)->getType();
4557 if (!AType->isVectorType() || !BType->isVectorType() ||
4558 !CType->isVectorType())
4559 return false;
4560
4561 const Pointer &C = S.Stk.pop<Pointer>();
4562 const Pointer &B = S.Stk.pop<Pointer>();
4563 const Pointer &A = S.Stk.pop<Pointer>();
4564 const Pointer &Dst = S.Stk.peek<Pointer>();
4565
4566 // check if all three primitive arrays are with 16-bit elements.
4567 auto isValid16BitArray = [](const Pointer &P) {
4568 const Descriptor *D = P.getFieldDesc();
4569 if (!D->isPrimitiveArray())
4570 return false;
4571 PrimType PT = D->getPrimType();
4572 return ((PT == PT_Sint16) || (PT == PT_Uint16));
4573 };
4574
4575 if (!isValid16BitArray(A) || !isValid16BitArray(B) || !isValid16BitArray(C))
4576 return false;
4577
4578 PrimType ElemT = A.getFieldDesc()->getPrimType();
4579 unsigned NumElems = A.getNumElems();
4580 assert(NumElems % 16 == 0 && "BMM operates on 256-bit lanes of 16 x i16");
4581 bool DstUnsigned = ElemT == PT_Uint16;
4582
4583 // Lanes are always 16-bit; gather them so the reduction below is untyped.
4584 SmallVector<uint16_t> AVals(NumElems), BVals(NumElems), Acc(NumElems);
4585 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
4586 for (unsigned I = 0; I != NumElems; ++I) {
4587 AVals[I] = (uint16_t)A.elem<T>(I).toAPSInt().getZExtValue();
4588 BVals[I] = (uint16_t)B.elem<T>(I).toAPSInt().getZExtValue();
4589 Acc[I] = (uint16_t)C.elem<T>(I).toAPSInt().getZExtValue();
4590 }
4591 });
4592
4593 for (unsigned Lane = 0; Lane != NumElems; Lane += 16) {
4594 for (unsigned I = 0; I != 16; ++I) {
4595 uint16_t AVal = AVals[Lane + I], DVal = Acc[Lane + I];
4596 for (unsigned J = 0; J != 16; ++J) {
4597 // Seed the reduction with the accumulator bit, then fold in each
4598 // product term with the same operator (OR for vbmacor, XOR for
4599 // vbmacxor).
4600 unsigned Bit = (DVal >> J) & 1u;
4601 for (unsigned K = 0; K != 16; ++K) {
4602 unsigned Product = ((AVal >> K) & 1u) & ((BVals[Lane + K] >> J) & 1u);
4603 Bit = IsXor ? (Bit ^ Product) : (Bit | Product);
4604 }
4605 DVal = (DVal & ~(uint16_t(1) << J)) | (uint16_t(Bit) << J);
4606 }
4607 Acc[Lane + I] = DVal;
4608 }
4609 }
4610
4611 INT_TYPE_SWITCH_NO_BOOL(ElemT, {
4612 for (unsigned I = 0; I != NumElems; ++I)
4613 Dst.elem<T>(I) = static_cast<T>(APSInt(APInt(16, Acc[I]), DstUnsigned));
4614 });
4615 Dst.initializeAllElements();
4616 return true;
4617}
4618
4619static bool interp_builtin_ia32_cvt_scalar_to_int(InterpState &S, CodePtr OpPC,
4620 const CallExpr *E) {
4621 Pointer SrcVecPtr = S.Stk.pop<Pointer>();
4622 const Floating &FloatElem = SrcVecPtr.elem<Floating>(I: 0);
4623
4624 unsigned BitWidth = S.getASTContext().getIntWidth(T: E->getType());
4625 bool IsUnsigned = E->getType()->isUnsignedIntegerType();
4626
4627 llvm::APSInt IntResult(BitWidth, IsUnsigned);
4628 bool IsExact = false;
4629 // We only allow exact conversions so rounding mode does not matter for cvt*
4630 // and cvtt* builtins
4631 FloatElem.getAPFloat().convertToInteger(
4632 Result&: IntResult, RM: llvm::APFloat::rmTowardZero, IsExact: &IsExact);
4633 if (!IsExact)
4634 return false;
4635
4636 pushInteger(S, Val: IntResult, QT: E->getType());
4637 return true;
4638}
4639
4640static bool interp_builtin_ia32_cvt_vector_to_int(InterpState &S, CodePtr OpPC,
4641 const CallExpr *E) {
4642 Pointer SrcVecPtr = S.Stk.pop<Pointer>();
4643 const Pointer &Dst = S.Stk.peek<Pointer>();
4644
4645 unsigned NumSrcElems = SrcVecPtr.getNumElems();
4646 unsigned NumDstElems = Dst.getNumElems();
4647
4648 if (NumSrcElems > NumDstElems)
4649 return false;
4650
4651 QualType ElemType = Dst.getFieldDesc()->getElemQualType();
4652 unsigned BitWidth = S.getASTContext().getIntWidth(T: ElemType);
4653 bool IsUnsigned = ElemType->isUnsignedIntegerType();
4654
4655 PrimType ElemT = *S.getContext().classify(T: ElemType);
4656 for (unsigned I = 0; I != NumSrcElems; ++I) {
4657 const Floating &FloatElem = SrcVecPtr.elem<Floating>(I);
4658 llvm::APSInt IntResult(BitWidth, IsUnsigned);
4659
4660 bool IsExact = false;
4661 // We only allow exact conversions so rounding mode does not matter for
4662 // cvt* and cvtt* builtins
4663 FloatElem.getAPFloat().convertToInteger(
4664 Result&: IntResult, RM: llvm::APFloat::rmTowardZero, IsExact: &IsExact);
4665 if (!IsExact)
4666 return false;
4667 INT_TYPE_SWITCH_NO_BOOL(
4668 ElemT, { Dst.elem<T>(I) = T::from(IntResult.getZExtValue()); });
4669 }
4670
4671 // Zero out remaining elements if the destination has more elements
4672 // (e.g., cvtpd2dq converting 2 doubles(_m128d) to 2 ints stored in _m128i).
4673 for (unsigned I = NumSrcElems; I != NumDstElems; ++I)
4674 INT_TYPE_SWITCH_NO_BOOL(ElemT, { Dst.elem<T>(I) = T::from(0); });
4675
4676 Dst.initializeAllElements();
4677 return true;
4678}
4679
4680bool InterpretBuiltin(InterpState &S, CodePtr OpPC, const CallExpr *Call,
4681 uint32_t BuiltinID) {
4682 const InterpFrame *Frame = S.Current;
4683 switch (BuiltinID) {
4684 case Builtin::BI__builtin_is_constant_evaluated:
4685 return interp__builtin_is_constant_evaluated(S, OpPC, Frame, Call);
4686
4687 case Builtin::BI__builtin_assume:
4688 case Builtin::BI__assume:
4689 return interp__builtin_assume(S, OpPC, Frame, Call);
4690
4691 case Builtin::BI__builtin_strcmp:
4692 case Builtin::BIstrcmp:
4693 case Builtin::BI__builtin_strncmp:
4694 case Builtin::BIstrncmp:
4695 case Builtin::BI__builtin_wcsncmp:
4696 case Builtin::BIwcsncmp:
4697 case Builtin::BI__builtin_wcscmp:
4698 case Builtin::BIwcscmp:
4699 return interp__builtin_strcmp(S, OpPC, Frame, Call, ID: BuiltinID);
4700
4701 case Builtin::BI__builtin_strlen:
4702 case Builtin::BIstrlen:
4703 case Builtin::BI__builtin_wcslen:
4704 case Builtin::BIwcslen:
4705 return interp__builtin_strlen(S, OpPC, Frame, Call, ID: BuiltinID);
4706
4707 case Builtin::BI__builtin_nan:
4708 case Builtin::BI__builtin_nanf:
4709 case Builtin::BI__builtin_nanl:
4710 case Builtin::BI__builtin_nanf16:
4711 case Builtin::BI__builtin_nanf128:
4712 return interp__builtin_nan(S, OpPC, Frame, Call, /*Signaling=*/false);
4713
4714 case Builtin::BI__builtin_nans:
4715 case Builtin::BI__builtin_nansf:
4716 case Builtin::BI__builtin_nansl:
4717 case Builtin::BI__builtin_nansf16:
4718 case Builtin::BI__builtin_nansf128:
4719 return interp__builtin_nan(S, OpPC, Frame, Call, /*Signaling=*/true);
4720
4721 case Builtin::BI__builtin_huge_val:
4722 case Builtin::BI__builtin_huge_valf:
4723 case Builtin::BI__builtin_huge_vall:
4724 case Builtin::BI__builtin_huge_valf16:
4725 case Builtin::BI__builtin_huge_valf128:
4726 case Builtin::BI__builtin_inf:
4727 case Builtin::BI__builtin_inff:
4728 case Builtin::BI__builtin_infl:
4729 case Builtin::BI__builtin_inff16:
4730 case Builtin::BI__builtin_inff128:
4731 return interp__builtin_inf(S, OpPC, Frame, Call);
4732
4733 case Builtin::BI__builtin_copysign:
4734 case Builtin::BI__builtin_copysignf:
4735 case Builtin::BI__builtin_copysignl:
4736 case Builtin::BI__builtin_copysignf128:
4737 return interp__builtin_copysign(S, OpPC, Frame);
4738
4739 case Builtin::BI__builtin_fmin:
4740 case Builtin::BI__builtin_fminf:
4741 case Builtin::BI__builtin_fminl:
4742 case Builtin::BI__builtin_fminf16:
4743 case Builtin::BI__builtin_fminf128:
4744 return interp__builtin_fmin(S, OpPC, Frame, /*IsNumBuiltin=*/false);
4745
4746 case Builtin::BI__builtin_fminimum_num:
4747 case Builtin::BI__builtin_fminimum_numf:
4748 case Builtin::BI__builtin_fminimum_numl:
4749 case Builtin::BI__builtin_fminimum_numf16:
4750 case Builtin::BI__builtin_fminimum_numf128:
4751 return interp__builtin_fmin(S, OpPC, Frame, /*IsNumBuiltin=*/true);
4752
4753 case Builtin::BI__builtin_fmax:
4754 case Builtin::BI__builtin_fmaxf:
4755 case Builtin::BI__builtin_fmaxl:
4756 case Builtin::BI__builtin_fmaxf16:
4757 case Builtin::BI__builtin_fmaxf128:
4758 return interp__builtin_fmax(S, OpPC, Frame, /*IsNumBuiltin=*/false);
4759
4760 case Builtin::BI__builtin_fmaximum_num:
4761 case Builtin::BI__builtin_fmaximum_numf:
4762 case Builtin::BI__builtin_fmaximum_numl:
4763 case Builtin::BI__builtin_fmaximum_numf16:
4764 case Builtin::BI__builtin_fmaximum_numf128:
4765 return interp__builtin_fmax(S, OpPC, Frame, /*IsNumBuiltin=*/true);
4766
4767 case Builtin::BI__builtin_isnan:
4768 return interp__builtin_isnan(S, OpPC, Frame, Call);
4769
4770 case Builtin::BI__builtin_issignaling:
4771 return interp__builtin_issignaling(S, OpPC, Frame, Call);
4772
4773 case Builtin::BI__builtin_isinf:
4774 return interp__builtin_isinf(S, OpPC, Frame, /*Sign=*/CheckSign: false, Call);
4775
4776 case Builtin::BI__builtin_isinf_sign:
4777 return interp__builtin_isinf(S, OpPC, Frame, /*Sign=*/CheckSign: true, Call);
4778
4779 case Builtin::BI__builtin_isfinite:
4780 return interp__builtin_isfinite(S, OpPC, Frame, Call);
4781
4782 case Builtin::BI__builtin_isnormal:
4783 return interp__builtin_isnormal(S, OpPC, Frame, Call);
4784
4785 case Builtin::BI__builtin_issubnormal:
4786 return interp__builtin_issubnormal(S, OpPC, Frame, Call);
4787
4788 case Builtin::BI__builtin_iszero:
4789 return interp__builtin_iszero(S, OpPC, Frame, Call);
4790
4791 case Builtin::BI__builtin_signbit:
4792 case Builtin::BI__builtin_signbitf:
4793 case Builtin::BI__builtin_signbitl:
4794 return interp__builtin_signbit(S, OpPC, Frame, Call);
4795
4796 case Builtin::BI__builtin_isgreater:
4797 case Builtin::BI__builtin_isgreaterequal:
4798 case Builtin::BI__builtin_isless:
4799 case Builtin::BI__builtin_islessequal:
4800 case Builtin::BI__builtin_islessgreater:
4801 case Builtin::BI__builtin_isunordered:
4802 return interp_floating_comparison(S, OpPC, Call, ID: BuiltinID);
4803
4804 case Builtin::BI__builtin_isfpclass:
4805 return interp__builtin_isfpclass(S, OpPC, Frame, Call);
4806
4807 case Builtin::BI__builtin_fpclassify:
4808 return interp__builtin_fpclassify(S, OpPC, Frame, Call);
4809
4810 case Builtin::BI__builtin_fabs:
4811 case Builtin::BI__builtin_fabsf:
4812 case Builtin::BI__builtin_fabsl:
4813 case Builtin::BI__builtin_fabsf128:
4814 return interp__builtin_fabs(S, OpPC, Frame);
4815
4816 case Builtin::BI__builtin_abs:
4817 case Builtin::BI__builtin_labs:
4818 case Builtin::BI__builtin_llabs:
4819 return interp__builtin_abs(S, OpPC, Frame, Call);
4820
4821 case Builtin::BI__builtin_popcount:
4822 case Builtin::BI__builtin_popcountl:
4823 case Builtin::BI__builtin_popcountll:
4824 case Builtin::BI__builtin_popcountg:
4825 case Builtin::BI__popcnt16: // Microsoft variants of popcount
4826 case Builtin::BI__popcnt:
4827 case Builtin::BI__popcnt64:
4828 return interp__builtin_popcount(S, OpPC, Frame, Call);
4829
4830 case Builtin::BI__builtin_parity:
4831 case Builtin::BI__builtin_parityl:
4832 case Builtin::BI__builtin_parityll:
4833 return interp__builtin_elementwise_int_unaryop(
4834 S, OpPC, Call, Fn: [](const APSInt &Val) {
4835 return APInt(Val.getBitWidth(), Val.popcount() % 2);
4836 });
4837 case Builtin::BI__builtin_clrsb:
4838 case Builtin::BI__builtin_clrsbl:
4839 case Builtin::BI__builtin_clrsbll:
4840 return interp__builtin_elementwise_int_unaryop(
4841 S, OpPC, Call, Fn: [](const APSInt &Val) {
4842 return APInt(Val.getBitWidth(),
4843 Val.getBitWidth() - Val.getSignificantBits());
4844 });
4845 case Builtin::BI__builtin_bitreverseg:
4846 case Builtin::BI__builtin_bitreverse8:
4847 case Builtin::BI__builtin_bitreverse16:
4848 case Builtin::BI__builtin_bitreverse32:
4849 case Builtin::BI__builtin_bitreverse64:
4850 return interp__builtin_elementwise_int_unaryop(
4851 S, OpPC, Call, Fn: [](const APSInt &Val) { return Val.reverseBits(); });
4852
4853 case Builtin::BI__builtin_classify_type:
4854 return interp__builtin_classify_type(S, OpPC, Frame, Call);
4855
4856 case Builtin::BI__builtin_expect:
4857 case Builtin::BI__builtin_expect_with_probability:
4858 return interp__builtin_expect(S, OpPC, Frame, Call);
4859
4860 case Builtin::BI__builtin_rotateleft8:
4861 case Builtin::BI__builtin_rotateleft16:
4862 case Builtin::BI__builtin_rotateleft32:
4863 case Builtin::BI__builtin_rotateleft64:
4864 case Builtin::BI__builtin_stdc_rotate_left:
4865 case Builtin::BIstdc_rotate_left_uc:
4866 case Builtin::BIstdc_rotate_left_us:
4867 case Builtin::BIstdc_rotate_left_ui:
4868 case Builtin::BIstdc_rotate_left_ul:
4869 case Builtin::BIstdc_rotate_left_ull:
4870 case Builtin::BI_rotl8: // Microsoft variants of rotate left
4871 case Builtin::BI_rotl16:
4872 case Builtin::BI_rotl:
4873 case Builtin::BI_lrotl:
4874 case Builtin::BI_rotl64:
4875 case Builtin::BI__builtin_rotateright8:
4876 case Builtin::BI__builtin_rotateright16:
4877 case Builtin::BI__builtin_rotateright32:
4878 case Builtin::BI__builtin_rotateright64:
4879 case Builtin::BI__builtin_stdc_rotate_right:
4880 case Builtin::BIstdc_rotate_right_uc:
4881 case Builtin::BIstdc_rotate_right_us:
4882 case Builtin::BIstdc_rotate_right_ui:
4883 case Builtin::BIstdc_rotate_right_ul:
4884 case Builtin::BIstdc_rotate_right_ull:
4885 case Builtin::BI_rotr8: // Microsoft variants of rotate right
4886 case Builtin::BI_rotr16:
4887 case Builtin::BI_rotr:
4888 case Builtin::BI_lrotr:
4889 case Builtin::BI_rotr64: {
4890 // Determine if this is a rotate right operation
4891 bool IsRotateRight;
4892 switch (BuiltinID) {
4893 case Builtin::BI__builtin_rotateright8:
4894 case Builtin::BI__builtin_rotateright16:
4895 case Builtin::BI__builtin_rotateright32:
4896 case Builtin::BI__builtin_rotateright64:
4897 case Builtin::BI__builtin_stdc_rotate_right:
4898 case Builtin::BIstdc_rotate_right_uc:
4899 case Builtin::BIstdc_rotate_right_us:
4900 case Builtin::BIstdc_rotate_right_ui:
4901 case Builtin::BIstdc_rotate_right_ul:
4902 case Builtin::BIstdc_rotate_right_ull:
4903 case Builtin::BI_rotr8:
4904 case Builtin::BI_rotr16:
4905 case Builtin::BI_rotr:
4906 case Builtin::BI_lrotr:
4907 case Builtin::BI_rotr64:
4908 IsRotateRight = true;
4909 break;
4910 default:
4911 IsRotateRight = false;
4912 break;
4913 }
4914
4915 return interp__builtin_elementwise_int_binop(
4916 S, OpPC, Call, Fn: [IsRotateRight](const APSInt &Value, APSInt Amount) {
4917 Amount = NormalizeRotateAmount(Value, Amount);
4918 return IsRotateRight ? Value.rotr(rotateAmt: Amount.getZExtValue())
4919 : Value.rotl(rotateAmt: Amount.getZExtValue());
4920 });
4921 }
4922
4923 case Builtin::BIstdc_leading_zeros_uc:
4924 case Builtin::BIstdc_leading_zeros_us:
4925 case Builtin::BIstdc_leading_zeros_ui:
4926 case Builtin::BIstdc_leading_zeros_ul:
4927 case Builtin::BIstdc_leading_zeros_ull:
4928 case Builtin::BI__builtin_stdc_leading_zeros: {
4929 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
4930 return interp__builtin_elementwise_int_unaryop(
4931 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
4932 return APInt(ResWidth, Val.countl_zero());
4933 });
4934 }
4935
4936 case Builtin::BIstdc_leading_ones_uc:
4937 case Builtin::BIstdc_leading_ones_us:
4938 case Builtin::BIstdc_leading_ones_ui:
4939 case Builtin::BIstdc_leading_ones_ul:
4940 case Builtin::BIstdc_leading_ones_ull:
4941 case Builtin::BI__builtin_stdc_leading_ones: {
4942 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
4943 return interp__builtin_elementwise_int_unaryop(
4944 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
4945 return APInt(ResWidth, Val.countl_one());
4946 });
4947 }
4948
4949 case Builtin::BIstdc_trailing_zeros_uc:
4950 case Builtin::BIstdc_trailing_zeros_us:
4951 case Builtin::BIstdc_trailing_zeros_ui:
4952 case Builtin::BIstdc_trailing_zeros_ul:
4953 case Builtin::BIstdc_trailing_zeros_ull:
4954 case Builtin::BI__builtin_stdc_trailing_zeros: {
4955 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
4956 return interp__builtin_elementwise_int_unaryop(
4957 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
4958 return APInt(ResWidth, Val.countr_zero());
4959 });
4960 }
4961
4962 case Builtin::BIstdc_trailing_ones_uc:
4963 case Builtin::BIstdc_trailing_ones_us:
4964 case Builtin::BIstdc_trailing_ones_ui:
4965 case Builtin::BIstdc_trailing_ones_ul:
4966 case Builtin::BIstdc_trailing_ones_ull:
4967 case Builtin::BI__builtin_stdc_trailing_ones: {
4968 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
4969 return interp__builtin_elementwise_int_unaryop(
4970 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
4971 return APInt(ResWidth, Val.countr_one());
4972 });
4973 }
4974
4975 case Builtin::BIstdc_first_leading_zero_uc:
4976 case Builtin::BIstdc_first_leading_zero_us:
4977 case Builtin::BIstdc_first_leading_zero_ui:
4978 case Builtin::BIstdc_first_leading_zero_ul:
4979 case Builtin::BIstdc_first_leading_zero_ull:
4980 case Builtin::BI__builtin_stdc_first_leading_zero: {
4981 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
4982 return interp__builtin_elementwise_int_unaryop(
4983 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
4984 return APInt(ResWidth, Val.isAllOnes() ? 0 : Val.countl_one() + 1);
4985 });
4986 }
4987
4988 case Builtin::BIstdc_first_leading_one_uc:
4989 case Builtin::BIstdc_first_leading_one_us:
4990 case Builtin::BIstdc_first_leading_one_ui:
4991 case Builtin::BIstdc_first_leading_one_ul:
4992 case Builtin::BIstdc_first_leading_one_ull:
4993 case Builtin::BI__builtin_stdc_first_leading_one: {
4994 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
4995 return interp__builtin_elementwise_int_unaryop(
4996 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
4997 return APInt(ResWidth, Val.isZero() ? 0 : Val.countl_zero() + 1);
4998 });
4999 }
5000
5001 case Builtin::BIstdc_first_trailing_zero_uc:
5002 case Builtin::BIstdc_first_trailing_zero_us:
5003 case Builtin::BIstdc_first_trailing_zero_ui:
5004 case Builtin::BIstdc_first_trailing_zero_ul:
5005 case Builtin::BIstdc_first_trailing_zero_ull:
5006 case Builtin::BI__builtin_stdc_first_trailing_zero: {
5007 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
5008 return interp__builtin_elementwise_int_unaryop(
5009 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
5010 return APInt(ResWidth, Val.isAllOnes() ? 0 : Val.countr_one() + 1);
5011 });
5012 }
5013
5014 case Builtin::BIstdc_first_trailing_one_uc:
5015 case Builtin::BIstdc_first_trailing_one_us:
5016 case Builtin::BIstdc_first_trailing_one_ui:
5017 case Builtin::BIstdc_first_trailing_one_ul:
5018 case Builtin::BIstdc_first_trailing_one_ull:
5019 case Builtin::BI__builtin_stdc_first_trailing_one: {
5020 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
5021 return interp__builtin_elementwise_int_unaryop(
5022 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
5023 return APInt(ResWidth, Val.isZero() ? 0 : Val.countr_zero() + 1);
5024 });
5025 }
5026
5027 case Builtin::BIstdc_count_zeros_uc:
5028 case Builtin::BIstdc_count_zeros_us:
5029 case Builtin::BIstdc_count_zeros_ui:
5030 case Builtin::BIstdc_count_zeros_ul:
5031 case Builtin::BIstdc_count_zeros_ull:
5032 case Builtin::BI__builtin_stdc_count_zeros: {
5033 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
5034 return interp__builtin_elementwise_int_unaryop(
5035 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
5036 unsigned BitWidth = Val.getBitWidth();
5037 return APInt(ResWidth, BitWidth - Val.popcount());
5038 });
5039 }
5040
5041 case Builtin::BIstdc_count_ones_uc:
5042 case Builtin::BIstdc_count_ones_us:
5043 case Builtin::BIstdc_count_ones_ui:
5044 case Builtin::BIstdc_count_ones_ul:
5045 case Builtin::BIstdc_count_ones_ull:
5046 case Builtin::BI__builtin_stdc_count_ones: {
5047 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
5048 return interp__builtin_elementwise_int_unaryop(
5049 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
5050 return APInt(ResWidth, Val.popcount());
5051 });
5052 }
5053
5054 case Builtin::BIstdc_has_single_bit_uc:
5055 case Builtin::BIstdc_has_single_bit_us:
5056 case Builtin::BIstdc_has_single_bit_ui:
5057 case Builtin::BIstdc_has_single_bit_ul:
5058 case Builtin::BIstdc_has_single_bit_ull:
5059 case Builtin::BI__builtin_stdc_has_single_bit: {
5060 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
5061 return interp__builtin_elementwise_int_unaryop(
5062 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
5063 return APInt(ResWidth, Val.popcount() == 1 ? 1 : 0);
5064 });
5065 }
5066
5067 case Builtin::BIstdc_bit_width_uc:
5068 case Builtin::BIstdc_bit_width_us:
5069 case Builtin::BIstdc_bit_width_ui:
5070 case Builtin::BIstdc_bit_width_ul:
5071 case Builtin::BIstdc_bit_width_ull:
5072 case Builtin::BI__builtin_stdc_bit_width: {
5073 unsigned ResWidth = S.getASTContext().getIntWidth(T: Call->getType());
5074 return interp__builtin_elementwise_int_unaryop(
5075 S, OpPC, Call, Fn: [ResWidth](const APSInt &Val) {
5076 unsigned BitWidth = Val.getBitWidth();
5077 return APInt(ResWidth, BitWidth - Val.countl_zero());
5078 });
5079 }
5080
5081 case Builtin::BIstdc_bit_floor_uc:
5082 case Builtin::BIstdc_bit_floor_us:
5083 case Builtin::BIstdc_bit_floor_ui:
5084 case Builtin::BIstdc_bit_floor_ul:
5085 case Builtin::BIstdc_bit_floor_ull:
5086 case Builtin::BI__builtin_stdc_bit_floor:
5087 return interp__builtin_elementwise_int_unaryop(
5088 S, OpPC, Call, Fn: [](const APSInt &Val) {
5089 unsigned BitWidth = Val.getBitWidth();
5090 if (Val.isZero())
5091 return APInt::getZero(numBits: BitWidth);
5092 return APInt::getOneBitSet(numBits: BitWidth,
5093 BitNo: BitWidth - Val.countl_zero() - 1);
5094 });
5095
5096 case Builtin::BIstdc_bit_ceil_uc:
5097 case Builtin::BIstdc_bit_ceil_us:
5098 case Builtin::BIstdc_bit_ceil_ui:
5099 case Builtin::BIstdc_bit_ceil_ul:
5100 case Builtin::BIstdc_bit_ceil_ull:
5101 case Builtin::BI__builtin_stdc_bit_ceil:
5102 return interp__builtin_elementwise_int_unaryop(
5103 S, OpPC, Call, Fn: [](const APSInt &Val) {
5104 unsigned BitWidth = Val.getBitWidth();
5105 if (Val.ule(RHS: 1))
5106 return APInt(BitWidth, 1);
5107 APInt V = Val;
5108 APInt ValMinusOne = V - 1;
5109 unsigned LeadingZeros = ValMinusOne.countl_zero();
5110 if (LeadingZeros == 0)
5111 return APInt(BitWidth, 0); // overflows; wrap to 0
5112 return APInt::getOneBitSet(numBits: BitWidth, BitNo: BitWidth - LeadingZeros);
5113 });
5114
5115 case Builtin::BI__builtin_ffs:
5116 case Builtin::BI__builtin_ffsl:
5117 case Builtin::BI__builtin_ffsll:
5118 return interp__builtin_elementwise_int_unaryop(
5119 S, OpPC, Call, Fn: [](const APSInt &Val) {
5120 return APInt(Val.getBitWidth(),
5121 Val.isZero() ? 0u : Val.countTrailingZeros() + 1u);
5122 });
5123
5124 case Builtin::BIaddressof:
5125 case Builtin::BI__addressof:
5126 case Builtin::BI__builtin_addressof:
5127 assert(isNoopBuiltin(BuiltinID));
5128 return interp__builtin_addressof(S, OpPC, Frame, Call);
5129
5130 case Builtin::BIas_const:
5131 case Builtin::BIforward:
5132 case Builtin::BIforward_like:
5133 case Builtin::BImove:
5134 case Builtin::BImove_if_noexcept:
5135 assert(isNoopBuiltin(BuiltinID));
5136 return interp__builtin_move(S, OpPC, Frame, Call);
5137
5138 case Builtin::BI__builtin_eh_return_data_regno:
5139 return interp__builtin_eh_return_data_regno(S, OpPC, Frame, Call);
5140
5141 case Builtin::BI__builtin_launder:
5142 assert(isNoopBuiltin(BuiltinID));
5143 return true;
5144
5145 case Builtin::BI__builtin_add_overflow:
5146 case Builtin::BI__builtin_sub_overflow:
5147 case Builtin::BI__builtin_mul_overflow:
5148 case Builtin::BI__builtin_sadd_overflow:
5149 case Builtin::BI__builtin_uadd_overflow:
5150 case Builtin::BI__builtin_uaddl_overflow:
5151 case Builtin::BI__builtin_uaddll_overflow:
5152 case Builtin::BI__builtin_usub_overflow:
5153 case Builtin::BI__builtin_usubl_overflow:
5154 case Builtin::BI__builtin_usubll_overflow:
5155 case Builtin::BI__builtin_umul_overflow:
5156 case Builtin::BI__builtin_umull_overflow:
5157 case Builtin::BI__builtin_umulll_overflow:
5158 case Builtin::BI__builtin_saddl_overflow:
5159 case Builtin::BI__builtin_saddll_overflow:
5160 case Builtin::BI__builtin_ssub_overflow:
5161 case Builtin::BI__builtin_ssubl_overflow:
5162 case Builtin::BI__builtin_ssubll_overflow:
5163 case Builtin::BI__builtin_smul_overflow:
5164 case Builtin::BI__builtin_smull_overflow:
5165 case Builtin::BI__builtin_smulll_overflow:
5166 return interp__builtin_overflowop(S, OpPC, Call, BuiltinOp: BuiltinID);
5167
5168 case Builtin::BI__builtin_addcb:
5169 case Builtin::BI__builtin_addcs:
5170 case Builtin::BI__builtin_addc:
5171 case Builtin::BI__builtin_addcl:
5172 case Builtin::BI__builtin_addcll:
5173 case Builtin::BI__builtin_subcb:
5174 case Builtin::BI__builtin_subcs:
5175 case Builtin::BI__builtin_subc:
5176 case Builtin::BI__builtin_subcl:
5177 case Builtin::BI__builtin_subcll:
5178 return interp__builtin_carryop(S, OpPC, Frame, Call, BuiltinOp: BuiltinID);
5179
5180 case Builtin::BI__builtin_clz:
5181 case Builtin::BI__builtin_clzl:
5182 case Builtin::BI__builtin_clzll:
5183 case Builtin::BI__builtin_clzs:
5184 case Builtin::BI__builtin_clzg:
5185 case Builtin::BI__lzcnt16: // Microsoft variants of count leading-zeroes
5186 case Builtin::BI__lzcnt:
5187 case Builtin::BI__lzcnt64:
5188 return interp__builtin_clz(S, OpPC, Frame, Call, BuiltinOp: BuiltinID);
5189
5190 case Builtin::BI__builtin_ctz:
5191 case Builtin::BI__builtin_ctzl:
5192 case Builtin::BI__builtin_ctzll:
5193 case Builtin::BI__builtin_ctzs:
5194 case Builtin::BI__builtin_ctzg:
5195 return interp__builtin_ctz(S, OpPC, Frame, Call, BuiltinID);
5196
5197 case Builtin::BI__builtin_elementwise_clzg:
5198 case Builtin::BI__builtin_elementwise_ctzg:
5199 return interp__builtin_elementwise_countzeroes(S, OpPC, Frame, Call,
5200 BuiltinID);
5201 case Builtin::BI__builtin_bswapg:
5202 case Builtin::BI__builtin_bswap16:
5203 case Builtin::BI__builtin_bswap32:
5204 case Builtin::BI__builtin_bswap64:
5205 case Builtin::BIstdc_memreverse8u8:
5206 case Builtin::BIstdc_memreverse8u16:
5207 case Builtin::BIstdc_memreverse8u32:
5208 case Builtin::BIstdc_memreverse8u64:
5209 return interp__builtin_bswap(S, OpPC, Frame, Call);
5210
5211 case Builtin::BIstdc_memreverse8:
5212 case Builtin::BI__builtin_stdc_memreverse8:
5213 return interp__builtin_stdc_memreverse8(S, OpPC, Frame, Call);
5214
5215 case Builtin::BI__atomic_always_lock_free:
5216 case Builtin::BI__atomic_is_lock_free:
5217 return interp__builtin_atomic_lock_free(S, OpPC, Frame, Call, BuiltinOp: BuiltinID);
5218
5219 case Builtin::BI__c11_atomic_is_lock_free:
5220 return interp__builtin_c11_atomic_is_lock_free(S, OpPC, Frame, Call);
5221
5222 case Builtin::BI__builtin_complex:
5223 return interp__builtin_complex(S, OpPC, Frame, Call);
5224
5225 case Builtin::BI__builtin_is_aligned:
5226 case Builtin::BI__builtin_align_up:
5227 case Builtin::BI__builtin_align_down:
5228 return interp__builtin_is_aligned_up_down(S, OpPC, Frame, Call, BuiltinOp: BuiltinID);
5229
5230 case Builtin::BI__builtin_assume_aligned:
5231 return interp__builtin_assume_aligned(S, OpPC, Frame, Call);
5232
5233 case clang::X86::BI__builtin_ia32_crc32qi:
5234 return interp__builtin_ia32_crc32(S, OpPC, Frame, Call, DataBytes: 1);
5235 case clang::X86::BI__builtin_ia32_crc32hi:
5236 return interp__builtin_ia32_crc32(S, OpPC, Frame, Call, DataBytes: 2);
5237 case clang::X86::BI__builtin_ia32_crc32si:
5238 return interp__builtin_ia32_crc32(S, OpPC, Frame, Call, DataBytes: 4);
5239 case clang::X86::BI__builtin_ia32_crc32di:
5240 return interp__builtin_ia32_crc32(S, OpPC, Frame, Call, DataBytes: 8);
5241
5242 case clang::X86::BI__builtin_ia32_bextr_u32:
5243 case clang::X86::BI__builtin_ia32_bextr_u64:
5244 case clang::X86::BI__builtin_ia32_bextri_u32:
5245 case clang::X86::BI__builtin_ia32_bextri_u64:
5246 return interp__builtin_elementwise_int_binop(
5247 S, OpPC, Call, Fn: [](const APSInt &Val, const APSInt &Idx) {
5248 unsigned BitWidth = Val.getBitWidth();
5249 uint64_t Shift = Idx.extractBitsAsZExtValue(numBits: 8, bitPosition: 0);
5250 uint64_t Length = Idx.extractBitsAsZExtValue(numBits: 8, bitPosition: 8);
5251 if (Length > BitWidth) {
5252 Length = BitWidth;
5253 }
5254
5255 // Handle out of bounds cases.
5256 if (Length == 0 || Shift >= BitWidth)
5257 return APInt(BitWidth, 0);
5258
5259 uint64_t Result = Val.getZExtValue() >> Shift;
5260 Result &= llvm::maskTrailingOnes<uint64_t>(N: Length);
5261 return APInt(BitWidth, Result);
5262 });
5263
5264 case clang::X86::BI__builtin_ia32_bzhi_si:
5265 case clang::X86::BI__builtin_ia32_bzhi_di:
5266 return interp__builtin_elementwise_int_binop(
5267 S, OpPC, Call, Fn: [](const APSInt &Val, const APSInt &Idx) {
5268 unsigned BitWidth = Val.getBitWidth();
5269 uint64_t Index = Idx.extractBitsAsZExtValue(numBits: 8, bitPosition: 0);
5270 APSInt Result = Val;
5271
5272 if (Index < BitWidth)
5273 Result.clearHighBits(hiBits: BitWidth - Index);
5274
5275 return Result;
5276 });
5277
5278 case clang::X86::BI__builtin_ia32_ktestcqi:
5279 case clang::X86::BI__builtin_ia32_ktestchi:
5280 case clang::X86::BI__builtin_ia32_ktestcsi:
5281 case clang::X86::BI__builtin_ia32_ktestcdi:
5282 return interp__builtin_elementwise_int_binop(
5283 S, OpPC, Call, Fn: [](const APSInt &A, const APSInt &B) {
5284 return APInt(sizeof(unsigned char) * 8, (~A & B) == 0);
5285 });
5286
5287 case clang::X86::BI__builtin_ia32_ktestzqi:
5288 case clang::X86::BI__builtin_ia32_ktestzhi:
5289 case clang::X86::BI__builtin_ia32_ktestzsi:
5290 case clang::X86::BI__builtin_ia32_ktestzdi:
5291 return interp__builtin_elementwise_int_binop(
5292 S, OpPC, Call, Fn: [](const APSInt &A, const APSInt &B) {
5293 return APInt(sizeof(unsigned char) * 8, (A & B) == 0);
5294 });
5295
5296 case clang::X86::BI__builtin_ia32_kortestcqi:
5297 case clang::X86::BI__builtin_ia32_kortestchi:
5298 case clang::X86::BI__builtin_ia32_kortestcsi:
5299 case clang::X86::BI__builtin_ia32_kortestcdi:
5300 return interp__builtin_elementwise_int_binop(
5301 S, OpPC, Call, Fn: [](const APSInt &A, const APSInt &B) {
5302 return APInt(sizeof(unsigned char) * 8, ~(A | B) == 0);
5303 });
5304
5305 case clang::X86::BI__builtin_ia32_kortestzqi:
5306 case clang::X86::BI__builtin_ia32_kortestzhi:
5307 case clang::X86::BI__builtin_ia32_kortestzsi:
5308 case clang::X86::BI__builtin_ia32_kortestzdi:
5309 return interp__builtin_elementwise_int_binop(
5310 S, OpPC, Call, Fn: [](const APSInt &A, const APSInt &B) {
5311 return APInt(sizeof(unsigned char) * 8, (A | B) == 0);
5312 });
5313
5314 case clang::X86::BI__builtin_ia32_kshiftliqi:
5315 case clang::X86::BI__builtin_ia32_kshiftlihi:
5316 case clang::X86::BI__builtin_ia32_kshiftlisi:
5317 case clang::X86::BI__builtin_ia32_kshiftlidi:
5318 return interp__builtin_elementwise_int_binop(
5319 S, OpPC, Call, Fn: [](const APSInt &LHS, const APSInt &RHS) {
5320 unsigned Amt = RHS.getZExtValue() & 0xFF;
5321 if (Amt >= LHS.getBitWidth())
5322 return APInt::getZero(numBits: LHS.getBitWidth());
5323 return LHS.shl(shiftAmt: Amt);
5324 });
5325
5326 case clang::X86::BI__builtin_ia32_kshiftriqi:
5327 case clang::X86::BI__builtin_ia32_kshiftrihi:
5328 case clang::X86::BI__builtin_ia32_kshiftrisi:
5329 case clang::X86::BI__builtin_ia32_kshiftridi:
5330 return interp__builtin_elementwise_int_binop(
5331 S, OpPC, Call, Fn: [](const APSInt &LHS, const APSInt &RHS) {
5332 unsigned Amt = RHS.getZExtValue() & 0xFF;
5333 if (Amt >= LHS.getBitWidth())
5334 return APInt::getZero(numBits: LHS.getBitWidth());
5335 return LHS.lshr(shiftAmt: Amt);
5336 });
5337
5338 case clang::X86::BI__builtin_ia32_lzcnt_u16:
5339 case clang::X86::BI__builtin_ia32_lzcnt_u32:
5340 case clang::X86::BI__builtin_ia32_lzcnt_u64:
5341 return interp__builtin_elementwise_int_unaryop(
5342 S, OpPC, Call, Fn: [](const APSInt &Src) {
5343 return APInt(Src.getBitWidth(), Src.countLeadingZeros());
5344 });
5345
5346 case clang::X86::BI__builtin_ia32_tzcnt_u16:
5347 case clang::X86::BI__builtin_ia32_tzcnt_u32:
5348 case clang::X86::BI__builtin_ia32_tzcnt_u64:
5349 return interp__builtin_elementwise_int_unaryop(
5350 S, OpPC, Call, Fn: [](const APSInt &Src) {
5351 return APInt(Src.getBitWidth(), Src.countTrailingZeros());
5352 });
5353
5354 case clang::X86::BI__builtin_ia32_addcarryx_u32:
5355 case clang::X86::BI__builtin_ia32_addcarryx_u64:
5356 return interp__builtin_ia32_addcarry_subborrow(S, OpPC, Frame, Call,
5357 /*IsAdd=*/true);
5358
5359 case clang::X86::BI__builtin_ia32_subborrow_u32:
5360 case clang::X86::BI__builtin_ia32_subborrow_u64:
5361 return interp__builtin_ia32_addcarry_subborrow(S, OpPC, Frame, Call,
5362 /*IsAdd=*/false);
5363
5364 case Builtin::BI__builtin_os_log_format_buffer_size:
5365 return interp__builtin_os_log_format_buffer_size(S, OpPC, Frame, Call);
5366
5367 case Builtin::BI__builtin_ptrauth_string_discriminator:
5368 return interp__builtin_ptrauth_string_discriminator(S, OpPC, Frame, Call);
5369
5370 case Builtin::BI__builtin_infer_alloc_token:
5371 return interp__builtin_infer_alloc_token(S, OpPC, Frame, Call);
5372
5373 case Builtin::BI__noop:
5374 pushInteger(S, Val: 0, QT: Call->getType());
5375 return true;
5376
5377 case Builtin::BI__builtin_operator_new:
5378 return interp__builtin_operator_new(S, OpPC, Frame, Call);
5379
5380 case Builtin::BI__builtin_operator_delete:
5381 return interp__builtin_operator_delete(S, OpPC, Frame, Call);
5382
5383 case Builtin::BI__arithmetic_fence:
5384 return interp__builtin_arithmetic_fence(S, OpPC, Frame, Call);
5385
5386 case Builtin::BI__builtin_reduce_add:
5387 case Builtin::BI__builtin_reduce_mul:
5388 case Builtin::BI__builtin_reduce_and:
5389 case Builtin::BI__builtin_reduce_or:
5390 case Builtin::BI__builtin_reduce_xor:
5391 case Builtin::BI__builtin_reduce_min:
5392 case Builtin::BI__builtin_reduce_max:
5393 return interp__builtin_vector_reduce(S, OpPC, Call, ID: BuiltinID);
5394
5395 case Builtin::BI__builtin_elementwise_popcount:
5396 return interp__builtin_elementwise_int_unaryop(
5397 S, OpPC, Call, Fn: [](const APSInt &Src) {
5398 return APInt(Src.getBitWidth(), Src.popcount());
5399 });
5400 case Builtin::BI__builtin_elementwise_bitreverse:
5401 return interp__builtin_elementwise_int_unaryop(
5402 S, OpPC, Call, Fn: [](const APSInt &Src) { return Src.reverseBits(); });
5403
5404 case Builtin::BI__builtin_elementwise_abs:
5405 return interp__builtin_elementwise_abs(S, OpPC, Frame, Call, BuiltinID);
5406
5407 case Builtin::BI__builtin_memcpy:
5408 case Builtin::BImemcpy:
5409 case Builtin::BI__builtin_wmemcpy:
5410 case Builtin::BIwmemcpy:
5411 case Builtin::BI__builtin_memmove:
5412 case Builtin::BImemmove:
5413 case Builtin::BI__builtin_wmemmove:
5414 case Builtin::BIwmemmove:
5415 return interp__builtin_memcpy(S, OpPC, Frame, Call, ID: BuiltinID);
5416
5417 case Builtin::BI__builtin_memcmp:
5418 case Builtin::BImemcmp:
5419 case Builtin::BI__builtin_bcmp:
5420 case Builtin::BIbcmp:
5421 case Builtin::BI__builtin_wmemcmp:
5422 case Builtin::BIwmemcmp:
5423 return interp__builtin_memcmp(S, OpPC, Frame, Call, ID: BuiltinID);
5424
5425 case Builtin::BImemchr:
5426 case Builtin::BI__builtin_memchr:
5427 case Builtin::BIstrchr:
5428 case Builtin::BI__builtin_strchr:
5429 case Builtin::BIwmemchr:
5430 case Builtin::BI__builtin_wmemchr:
5431 case Builtin::BIwcschr:
5432 case Builtin::BI__builtin_wcschr:
5433 case Builtin::BI__builtin_char_memchr:
5434 return interp__builtin_memchr(S, OpPC, Call, ID: BuiltinID);
5435
5436 case Builtin::BI__builtin_object_size:
5437 return interp__builtin_object_size(S, OpPC, Frame, Call,
5438 /*IsDynamic=*/false);
5439 case Builtin::BI__builtin_dynamic_object_size:
5440 return interp__builtin_object_size(S, OpPC, Frame, Call,
5441 /*IsDynamic=*/true);
5442
5443 case Builtin::BI__builtin_is_within_lifetime:
5444 return interp__builtin_is_within_lifetime(S, OpPC, Call);
5445
5446 case Builtin::BI__builtin_elementwise_add_sat:
5447 return interp__builtin_elementwise_int_binop(
5448 S, OpPC, Call, Fn: [](const APSInt &LHS, const APSInt &RHS) {
5449 return LHS.isSigned() ? LHS.sadd_sat(RHS) : LHS.uadd_sat(RHS);
5450 });
5451
5452 case Builtin::BI__builtin_elementwise_sub_sat:
5453 return interp__builtin_elementwise_int_binop(
5454 S, OpPC, Call, Fn: [](const APSInt &LHS, const APSInt &RHS) {
5455 return LHS.isSigned() ? LHS.ssub_sat(RHS) : LHS.usub_sat(RHS);
5456 });
5457
5458 case Builtin::BI__builtin_elementwise_pdep:
5459 return interp__builtin_elementwise_int_binop(S, OpPC, Call,
5460 Fn: llvm::APIntOps::pdep);
5461
5462 case Builtin::BI__builtin_elementwise_pext:
5463 return interp__builtin_elementwise_int_binop(S, OpPC, Call,
5464 Fn: llvm::APIntOps::pext);
5465
5466 case X86::BI__builtin_ia32_extract128i256:
5467 case X86::BI__builtin_ia32_vextractf128_pd256:
5468 case X86::BI__builtin_ia32_vextractf128_ps256:
5469 case X86::BI__builtin_ia32_vextractf128_si256:
5470 return interp__builtin_ia32_extract_vector(S, OpPC, Call, ID: BuiltinID);
5471
5472 case X86::BI__builtin_ia32_extractf32x4_256_mask:
5473 case X86::BI__builtin_ia32_extractf32x4_mask:
5474 case X86::BI__builtin_ia32_extractf32x8_mask:
5475 case X86::BI__builtin_ia32_extractf64x2_256_mask:
5476 case X86::BI__builtin_ia32_extractf64x2_512_mask:
5477 case X86::BI__builtin_ia32_extractf64x4_mask:
5478 case X86::BI__builtin_ia32_extracti32x4_256_mask:
5479 case X86::BI__builtin_ia32_extracti32x4_mask:
5480 case X86::BI__builtin_ia32_extracti32x8_mask:
5481 case X86::BI__builtin_ia32_extracti64x2_256_mask:
5482 case X86::BI__builtin_ia32_extracti64x2_512_mask:
5483 case X86::BI__builtin_ia32_extracti64x4_mask:
5484 return interp__builtin_ia32_extract_vector_masked(S, OpPC, Call, ID: BuiltinID);
5485
5486 case clang::X86::BI__builtin_ia32_pmulhrsw128:
5487 case clang::X86::BI__builtin_ia32_pmulhrsw256:
5488 case clang::X86::BI__builtin_ia32_pmulhrsw512:
5489 return interp__builtin_elementwise_int_binop(
5490 S, OpPC, Call, Fn: [](const APSInt &LHS, const APSInt &RHS) {
5491 return (llvm::APIntOps::mulsExtended(C1: LHS, C2: RHS).ashr(ShiftAmt: 14) + 1)
5492 .extractBits(numBits: 16, bitPosition: 1);
5493 });
5494
5495 case clang::X86::BI__builtin_ia32_movmskps:
5496 case clang::X86::BI__builtin_ia32_movmskpd:
5497 case clang::X86::BI__builtin_ia32_pmovmskb128:
5498 case clang::X86::BI__builtin_ia32_pmovmskb256:
5499 case clang::X86::BI__builtin_ia32_movmskps256:
5500 case clang::X86::BI__builtin_ia32_movmskpd256: {
5501 return interp__builtin_ia32_movmsk_op(S, OpPC, Call);
5502 }
5503
5504 case X86::BI__builtin_ia32_psignb128:
5505 case X86::BI__builtin_ia32_psignb256:
5506 case X86::BI__builtin_ia32_psignw128:
5507 case X86::BI__builtin_ia32_psignw256:
5508 case X86::BI__builtin_ia32_psignd128:
5509 case X86::BI__builtin_ia32_psignd256:
5510 return interp__builtin_elementwise_int_binop(
5511 S, OpPC, Call, Fn: [](const APInt &AElem, const APInt &BElem) {
5512 if (BElem.isZero())
5513 return APInt::getZero(numBits: AElem.getBitWidth());
5514 if (BElem.isNegative())
5515 return -AElem;
5516 return AElem;
5517 });
5518
5519 case clang::X86::BI__builtin_ia32_pavgb128:
5520 case clang::X86::BI__builtin_ia32_pavgw128:
5521 case clang::X86::BI__builtin_ia32_pavgb256:
5522 case clang::X86::BI__builtin_ia32_pavgw256:
5523 case clang::X86::BI__builtin_ia32_pavgb512:
5524 case clang::X86::BI__builtin_ia32_pavgw512:
5525 return interp__builtin_elementwise_int_binop(S, OpPC, Call,
5526 Fn: llvm::APIntOps::avgCeilU);
5527
5528 case clang::X86::BI__builtin_ia32_pmaddubsw128:
5529 case clang::X86::BI__builtin_ia32_pmaddubsw256:
5530 case clang::X86::BI__builtin_ia32_pmaddubsw512:
5531 return interp__builtin_ia32_pmul(
5532 S, OpPC, Call,
5533 Fn: [](const APSInt &LoLHS, const APSInt &HiLHS, const APSInt &LoRHS,
5534 const APSInt &HiRHS) {
5535 unsigned BitWidth = 2 * LoLHS.getBitWidth();
5536 return (LoLHS.zext(width: BitWidth) * LoRHS.sext(width: BitWidth))
5537 .sadd_sat(RHS: (HiLHS.zext(width: BitWidth) * HiRHS.sext(width: BitWidth)));
5538 });
5539
5540 case clang::X86::BI__builtin_ia32_pmaddwd128:
5541 case clang::X86::BI__builtin_ia32_pmaddwd256:
5542 case clang::X86::BI__builtin_ia32_pmaddwd512:
5543 return interp__builtin_ia32_pmul(
5544 S, OpPC, Call,
5545 Fn: [](const APSInt &LoLHS, const APSInt &HiLHS, const APSInt &LoRHS,
5546 const APSInt &HiRHS) {
5547 unsigned BitWidth = 2 * LoLHS.getBitWidth();
5548 return (LoLHS.sext(width: BitWidth) * LoRHS.sext(width: BitWidth)) +
5549 (HiLHS.sext(width: BitWidth) * HiRHS.sext(width: BitWidth));
5550 });
5551
5552 case clang::X86::BI__builtin_ia32_psadbw128:
5553 case clang::X86::BI__builtin_ia32_psadbw256:
5554 case clang::X86::BI__builtin_ia32_psadbw512:
5555 return interp__builtin_ia32_psadbw(S, OpPC, Call);
5556
5557 case clang::X86::BI__builtin_ia32_dbpsadbw128:
5558 case clang::X86::BI__builtin_ia32_dbpsadbw256:
5559 case clang::X86::BI__builtin_ia32_dbpsadbw512:
5560 return interp__builtin_ia32_dbpsadbw(S, OpPC, Call);
5561
5562 case clang::X86::BI__builtin_ia32_mpsadbw128:
5563 case clang::X86::BI__builtin_ia32_mpsadbw256:
5564 return interp__builtin_ia32_mpsadbw(S, OpPC, Call);
5565
5566 case clang::X86::BI__builtin_ia32_pmulhuw128:
5567 case clang::X86::BI__builtin_ia32_pmulhuw256:
5568 case clang::X86::BI__builtin_ia32_pmulhuw512:
5569 return interp__builtin_elementwise_int_binop(S, OpPC, Call,
5570 Fn: llvm::APIntOps::mulhu);
5571
5572 case clang::X86::BI__builtin_ia32_pmulhw128:
5573 case clang::X86::BI__builtin_ia32_pmulhw256:
5574 case clang::X86::BI__builtin_ia32_pmulhw512:
5575 return interp__builtin_elementwise_int_binop(S, OpPC, Call,
5576 Fn: llvm::APIntOps::mulhs);
5577
5578 case clang::X86::BI__builtin_ia32_psllv2di:
5579 case clang::X86::BI__builtin_ia32_psllv4di:
5580 case clang::X86::BI__builtin_ia32_psllv4si:
5581 case clang::X86::BI__builtin_ia32_psllv8di:
5582 case clang::X86::BI__builtin_ia32_psllv8hi:
5583 case clang::X86::BI__builtin_ia32_psllv8si:
5584 case clang::X86::BI__builtin_ia32_psllv16hi:
5585 case clang::X86::BI__builtin_ia32_psllv16si:
5586 case clang::X86::BI__builtin_ia32_psllv32hi:
5587 case clang::X86::BI__builtin_ia32_psllwi128:
5588 case clang::X86::BI__builtin_ia32_psllwi256:
5589 case clang::X86::BI__builtin_ia32_psllwi512:
5590 case clang::X86::BI__builtin_ia32_pslldi128:
5591 case clang::X86::BI__builtin_ia32_pslldi256:
5592 case clang::X86::BI__builtin_ia32_pslldi512:
5593 case clang::X86::BI__builtin_ia32_psllqi128:
5594 case clang::X86::BI__builtin_ia32_psllqi256:
5595 case clang::X86::BI__builtin_ia32_psllqi512:
5596 return interp__builtin_elementwise_int_binop(
5597 S, OpPC, Call, Fn: [](const APSInt &LHS, const APSInt &RHS) {
5598 if (RHS.uge(RHS: LHS.getBitWidth())) {
5599 return APInt::getZero(numBits: LHS.getBitWidth());
5600 }
5601 return LHS.shl(shiftAmt: RHS.getZExtValue());
5602 });
5603
5604 case clang::X86::BI__builtin_ia32_psrav4si:
5605 case clang::X86::BI__builtin_ia32_psrav8di:
5606 case clang::X86::BI__builtin_ia32_psrav8hi:
5607 case clang::X86::BI__builtin_ia32_psrav8si:
5608 case clang::X86::BI__builtin_ia32_psrav16hi:
5609 case clang::X86::BI__builtin_ia32_psrav16si:
5610 case clang::X86::BI__builtin_ia32_psrav32hi:
5611 case clang::X86::BI__builtin_ia32_psravq128:
5612 case clang::X86::BI__builtin_ia32_psravq256:
5613 case clang::X86::BI__builtin_ia32_psrawi128:
5614 case clang::X86::BI__builtin_ia32_psrawi256:
5615 case clang::X86::BI__builtin_ia32_psrawi512:
5616 case clang::X86::BI__builtin_ia32_psradi128:
5617 case clang::X86::BI__builtin_ia32_psradi256:
5618 case clang::X86::BI__builtin_ia32_psradi512:
5619 case clang::X86::BI__builtin_ia32_psraqi128:
5620 case clang::X86::BI__builtin_ia32_psraqi256:
5621 case clang::X86::BI__builtin_ia32_psraqi512:
5622 return interp__builtin_elementwise_int_binop(
5623 S, OpPC, Call, Fn: [](const APSInt &LHS, const APSInt &RHS) {
5624 if (RHS.uge(RHS: LHS.getBitWidth())) {
5625 return LHS.ashr(ShiftAmt: LHS.getBitWidth() - 1);
5626 }
5627 return LHS.ashr(ShiftAmt: RHS.getZExtValue());
5628 });
5629
5630 case clang::X86::BI__builtin_ia32_psrlv2di:
5631 case clang::X86::BI__builtin_ia32_psrlv4di:
5632 case clang::X86::BI__builtin_ia32_psrlv4si:
5633 case clang::X86::BI__builtin_ia32_psrlv8di:
5634 case clang::X86::BI__builtin_ia32_psrlv8hi:
5635 case clang::X86::BI__builtin_ia32_psrlv8si:
5636 case clang::X86::BI__builtin_ia32_psrlv16hi:
5637 case clang::X86::BI__builtin_ia32_psrlv16si:
5638 case clang::X86::BI__builtin_ia32_psrlv32hi:
5639 case clang::X86::BI__builtin_ia32_psrlwi128:
5640 case clang::X86::BI__builtin_ia32_psrlwi256:
5641 case clang::X86::BI__builtin_ia32_psrlwi512:
5642 case clang::X86::BI__builtin_ia32_psrldi128:
5643 case clang::X86::BI__builtin_ia32_psrldi256:
5644 case clang::X86::BI__builtin_ia32_psrldi512:
5645 case clang::X86::BI__builtin_ia32_psrlqi128:
5646 case clang::X86::BI__builtin_ia32_psrlqi256:
5647 case clang::X86::BI__builtin_ia32_psrlqi512:
5648 return interp__builtin_elementwise_int_binop(
5649 S, OpPC, Call, Fn: [](const APSInt &LHS, const APSInt &RHS) {
5650 if (RHS.uge(RHS: LHS.getBitWidth())) {
5651 return APInt::getZero(numBits: LHS.getBitWidth());
5652 }
5653 return LHS.lshr(shiftAmt: RHS.getZExtValue());
5654 });
5655 case clang::X86::BI__builtin_ia32_packsswb128:
5656 case clang::X86::BI__builtin_ia32_packsswb256:
5657 case clang::X86::BI__builtin_ia32_packsswb512:
5658 case clang::X86::BI__builtin_ia32_packssdw128:
5659 case clang::X86::BI__builtin_ia32_packssdw256:
5660 case clang::X86::BI__builtin_ia32_packssdw512:
5661 return interp__builtin_ia32_pack(S, OpPC, E: Call, PackFn: [](const APSInt &Src) {
5662 return APInt(Src).truncSSat(width: Src.getBitWidth() / 2);
5663 });
5664 case clang::X86::BI__builtin_ia32_packusdw128:
5665 case clang::X86::BI__builtin_ia32_packusdw256:
5666 case clang::X86::BI__builtin_ia32_packusdw512:
5667 case clang::X86::BI__builtin_ia32_packuswb128:
5668 case clang::X86::BI__builtin_ia32_packuswb256:
5669 case clang::X86::BI__builtin_ia32_packuswb512:
5670 return interp__builtin_ia32_pack(S, OpPC, E: Call, PackFn: [](const APSInt &Src) {
5671 return APInt(Src).truncSSatU(width: Src.getBitWidth() / 2);
5672 });
5673
5674 case clang::X86::BI__builtin_ia32_selectss_128:
5675 case clang::X86::BI__builtin_ia32_selectsd_128:
5676 case clang::X86::BI__builtin_ia32_selectsh_128:
5677 case clang::X86::BI__builtin_ia32_selectsbf_128:
5678 return interp__builtin_ia32_select_scalar(S, Call);
5679 case clang::X86::BI__builtin_ia32_vprotbi:
5680 case clang::X86::BI__builtin_ia32_vprotdi:
5681 case clang::X86::BI__builtin_ia32_vprotqi:
5682 case clang::X86::BI__builtin_ia32_vprotwi:
5683 case clang::X86::BI__builtin_ia32_prold128:
5684 case clang::X86::BI__builtin_ia32_prold256:
5685 case clang::X86::BI__builtin_ia32_prold512:
5686 case clang::X86::BI__builtin_ia32_prolq128:
5687 case clang::X86::BI__builtin_ia32_prolq256:
5688 case clang::X86::BI__builtin_ia32_prolq512:
5689 return interp__builtin_elementwise_int_binop(
5690 S, OpPC, Call,
5691 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS.rotl(rotateAmt: RHS); });
5692
5693 case clang::X86::BI__builtin_ia32_prord128:
5694 case clang::X86::BI__builtin_ia32_prord256:
5695 case clang::X86::BI__builtin_ia32_prord512:
5696 case clang::X86::BI__builtin_ia32_prorq128:
5697 case clang::X86::BI__builtin_ia32_prorq256:
5698 case clang::X86::BI__builtin_ia32_prorq512:
5699 return interp__builtin_elementwise_int_binop(
5700 S, OpPC, Call,
5701 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS.rotr(rotateAmt: RHS); });
5702
5703 case Builtin::BI__builtin_elementwise_max:
5704 case Builtin::BI__builtin_elementwise_min:
5705 return interp__builtin_elementwise_maxmin(S, OpPC, Call, BuiltinID);
5706
5707 case clang::X86::BI__builtin_ia32_phaddw128:
5708 case clang::X86::BI__builtin_ia32_phaddw256:
5709 case clang::X86::BI__builtin_ia32_phaddd128:
5710 case clang::X86::BI__builtin_ia32_phaddd256:
5711 return interp_builtin_horizontal_int_binop(
5712 S, OpPC, Call,
5713 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS + RHS; });
5714 case clang::X86::BI__builtin_ia32_phaddsw128:
5715 case clang::X86::BI__builtin_ia32_phaddsw256:
5716 return interp_builtin_horizontal_int_binop(
5717 S, OpPC, Call,
5718 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS.sadd_sat(RHS); });
5719 case clang::X86::BI__builtin_ia32_phsubw128:
5720 case clang::X86::BI__builtin_ia32_phsubw256:
5721 case clang::X86::BI__builtin_ia32_phsubd128:
5722 case clang::X86::BI__builtin_ia32_phsubd256:
5723 return interp_builtin_horizontal_int_binop(
5724 S, OpPC, Call,
5725 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS - RHS; });
5726 case clang::X86::BI__builtin_ia32_phsubsw128:
5727 case clang::X86::BI__builtin_ia32_phsubsw256:
5728 return interp_builtin_horizontal_int_binop(
5729 S, OpPC, Call,
5730 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS.ssub_sat(RHS); });
5731 case clang::X86::BI__builtin_ia32_haddpd:
5732 case clang::X86::BI__builtin_ia32_haddps:
5733 case clang::X86::BI__builtin_ia32_haddpd256:
5734 case clang::X86::BI__builtin_ia32_haddps256:
5735 return interp_builtin_horizontal_fp_binop(
5736 S, OpPC, Call,
5737 Fn: [](const APFloat &LHS, const APFloat &RHS, llvm::RoundingMode RM) {
5738 APFloat F = LHS;
5739 F.add(RHS, RM);
5740 return F;
5741 });
5742 case clang::X86::BI__builtin_ia32_hsubpd:
5743 case clang::X86::BI__builtin_ia32_hsubps:
5744 case clang::X86::BI__builtin_ia32_hsubpd256:
5745 case clang::X86::BI__builtin_ia32_hsubps256:
5746 return interp_builtin_horizontal_fp_binop(
5747 S, OpPC, Call,
5748 Fn: [](const APFloat &LHS, const APFloat &RHS, llvm::RoundingMode RM) {
5749 APFloat F = LHS;
5750 F.subtract(RHS, RM);
5751 return F;
5752 });
5753 case clang::X86::BI__builtin_ia32_addsubpd:
5754 case clang::X86::BI__builtin_ia32_addsubps:
5755 case clang::X86::BI__builtin_ia32_addsubpd256:
5756 case clang::X86::BI__builtin_ia32_addsubps256:
5757 return interp__builtin_ia32_addsub(S, OpPC, Call);
5758
5759 case clang::X86::BI__builtin_ia32_pmuldq128:
5760 case clang::X86::BI__builtin_ia32_pmuldq256:
5761 case clang::X86::BI__builtin_ia32_pmuldq512:
5762 return interp__builtin_ia32_pmul(
5763 S, OpPC, Call,
5764 Fn: [](const APSInt &LoLHS, const APSInt &HiLHS, const APSInt &LoRHS,
5765 const APSInt &HiRHS) {
5766 return llvm::APIntOps::mulsExtended(C1: LoLHS, C2: LoRHS);
5767 });
5768
5769 case clang::X86::BI__builtin_ia32_pmuludq128:
5770 case clang::X86::BI__builtin_ia32_pmuludq256:
5771 case clang::X86::BI__builtin_ia32_pmuludq512:
5772 return interp__builtin_ia32_pmul(
5773 S, OpPC, Call,
5774 Fn: [](const APSInt &LoLHS, const APSInt &HiLHS, const APSInt &LoRHS,
5775 const APSInt &HiRHS) {
5776 return llvm::APIntOps::muluExtended(C1: LoLHS, C2: LoRHS);
5777 });
5778
5779 case clang::X86::BI__builtin_ia32_pclmulqdq128:
5780 case clang::X86::BI__builtin_ia32_pclmulqdq256:
5781 case clang::X86::BI__builtin_ia32_pclmulqdq512:
5782 return interp__builtin_ia32_pclmulqdq(S, OpPC, Call);
5783 case Builtin::BI__builtin_elementwise_clmul:
5784 return interp__builtin_elementwise_int_binop(S, OpPC, Call,
5785 Fn: llvm::APIntOps::clmul);
5786
5787 case Builtin::BI__builtin_elementwise_fma:
5788 return interp__builtin_elementwise_triop_fp(
5789 S, OpPC, Call,
5790 Fn: [](const APFloat &X, const APFloat &Y, const APFloat &Z,
5791 llvm::RoundingMode RM) {
5792 APFloat F = X;
5793 F.fusedMultiplyAdd(Multiplicand: Y, Addend: Z, RM);
5794 return F;
5795 });
5796
5797 case X86::BI__builtin_ia32_vpmadd52luq128:
5798 case X86::BI__builtin_ia32_vpmadd52luq256:
5799 case X86::BI__builtin_ia32_vpmadd52luq512:
5800 return interp__builtin_elementwise_triop(
5801 S, OpPC, Call, Fn: [](const APSInt &A, const APSInt &B, const APSInt &C) {
5802 return A + (B.trunc(width: 52) * C.trunc(width: 52)).zext(width: 64);
5803 });
5804 case X86::BI__builtin_ia32_vpmadd52huq128:
5805 case X86::BI__builtin_ia32_vpmadd52huq256:
5806 case X86::BI__builtin_ia32_vpmadd52huq512:
5807 return interp__builtin_elementwise_triop(
5808 S, OpPC, Call, Fn: [](const APSInt &A, const APSInt &B, const APSInt &C) {
5809 return A + llvm::APIntOps::mulhu(C1: B.trunc(width: 52), C2: C.trunc(width: 52)).zext(width: 64);
5810 });
5811
5812 case X86::BI__builtin_ia32_vpshldd128:
5813 case X86::BI__builtin_ia32_vpshldd256:
5814 case X86::BI__builtin_ia32_vpshldd512:
5815 case X86::BI__builtin_ia32_vpshldq128:
5816 case X86::BI__builtin_ia32_vpshldq256:
5817 case X86::BI__builtin_ia32_vpshldq512:
5818 case X86::BI__builtin_ia32_vpshldw128:
5819 case X86::BI__builtin_ia32_vpshldw256:
5820 case X86::BI__builtin_ia32_vpshldw512:
5821 return interp__builtin_elementwise_triop(
5822 S, OpPC, Call,
5823 Fn: [](const APSInt &Hi, const APSInt &Lo, const APSInt &Amt) {
5824 return llvm::APIntOps::fshl(Hi, Lo, Shift: Amt);
5825 });
5826
5827 case X86::BI__builtin_ia32_vpshrdd128:
5828 case X86::BI__builtin_ia32_vpshrdd256:
5829 case X86::BI__builtin_ia32_vpshrdd512:
5830 case X86::BI__builtin_ia32_vpshrdq128:
5831 case X86::BI__builtin_ia32_vpshrdq256:
5832 case X86::BI__builtin_ia32_vpshrdq512:
5833 case X86::BI__builtin_ia32_vpshrdw128:
5834 case X86::BI__builtin_ia32_vpshrdw256:
5835 case X86::BI__builtin_ia32_vpshrdw512:
5836 // NOTE: Reversed Hi/Lo operands.
5837 return interp__builtin_elementwise_triop(
5838 S, OpPC, Call,
5839 Fn: [](const APSInt &Lo, const APSInt &Hi, const APSInt &Amt) {
5840 return llvm::APIntOps::fshr(Hi, Lo, Shift: Amt);
5841 });
5842 case X86::BI__builtin_ia32_vpconflictsi_128:
5843 case X86::BI__builtin_ia32_vpconflictsi_256:
5844 case X86::BI__builtin_ia32_vpconflictsi_512:
5845 case X86::BI__builtin_ia32_vpconflictdi_128:
5846 case X86::BI__builtin_ia32_vpconflictdi_256:
5847 case X86::BI__builtin_ia32_vpconflictdi_512:
5848 return interp__builtin_ia32_vpconflict(S, OpPC, Call);
5849 case X86::BI__builtin_ia32_compressdf128_mask:
5850 case X86::BI__builtin_ia32_compressdf256_mask:
5851 case X86::BI__builtin_ia32_compressdf512_mask:
5852 case X86::BI__builtin_ia32_compressdi128_mask:
5853 case X86::BI__builtin_ia32_compressdi256_mask:
5854 case X86::BI__builtin_ia32_compressdi512_mask:
5855 case X86::BI__builtin_ia32_compresshi128_mask:
5856 case X86::BI__builtin_ia32_compresshi256_mask:
5857 case X86::BI__builtin_ia32_compresshi512_mask:
5858 case X86::BI__builtin_ia32_compressqi128_mask:
5859 case X86::BI__builtin_ia32_compressqi256_mask:
5860 case X86::BI__builtin_ia32_compressqi512_mask:
5861 case X86::BI__builtin_ia32_compresssf128_mask:
5862 case X86::BI__builtin_ia32_compresssf256_mask:
5863 case X86::BI__builtin_ia32_compresssf512_mask:
5864 case X86::BI__builtin_ia32_compresssi128_mask:
5865 case X86::BI__builtin_ia32_compresssi256_mask:
5866 case X86::BI__builtin_ia32_compresssi512_mask: {
5867 unsigned NumElems =
5868 Call->getArg(Arg: 0)->getType()->castAs<VectorType>()->getNumElements();
5869 return interp__builtin_ia32_shuffle_generic(
5870 S, OpPC, Call, GetSourceIndex: [NumElems](unsigned DstIdx, const APInt &ShuffleMask) {
5871 APInt CompressMask = ShuffleMask.trunc(width: NumElems);
5872 if (DstIdx < CompressMask.popcount()) {
5873 while (DstIdx != 0) {
5874 CompressMask = CompressMask & (CompressMask - 1);
5875 DstIdx--;
5876 }
5877 return std::pair<unsigned, int>{
5878 0, static_cast<int>(CompressMask.countr_zero())};
5879 }
5880 return std::pair<unsigned, int>{1, static_cast<int>(DstIdx)};
5881 });
5882 }
5883 case X86::BI__builtin_ia32_expanddf128_mask:
5884 case X86::BI__builtin_ia32_expanddf256_mask:
5885 case X86::BI__builtin_ia32_expanddf512_mask:
5886 case X86::BI__builtin_ia32_expanddi128_mask:
5887 case X86::BI__builtin_ia32_expanddi256_mask:
5888 case X86::BI__builtin_ia32_expanddi512_mask:
5889 case X86::BI__builtin_ia32_expandhi128_mask:
5890 case X86::BI__builtin_ia32_expandhi256_mask:
5891 case X86::BI__builtin_ia32_expandhi512_mask:
5892 case X86::BI__builtin_ia32_expandqi128_mask:
5893 case X86::BI__builtin_ia32_expandqi256_mask:
5894 case X86::BI__builtin_ia32_expandqi512_mask:
5895 case X86::BI__builtin_ia32_expandsf128_mask:
5896 case X86::BI__builtin_ia32_expandsf256_mask:
5897 case X86::BI__builtin_ia32_expandsf512_mask:
5898 case X86::BI__builtin_ia32_expandsi128_mask:
5899 case X86::BI__builtin_ia32_expandsi256_mask:
5900 case X86::BI__builtin_ia32_expandsi512_mask: {
5901 return interp__builtin_ia32_shuffle_generic(
5902 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, const APInt &ShuffleMask) {
5903 // Trunc to the sub-mask for the dst index and count the number of
5904 // src elements used prior to that.
5905 APInt ExpandMask = ShuffleMask.trunc(width: DstIdx + 1);
5906 if (ExpandMask[DstIdx]) {
5907 int SrcIdx = ExpandMask.popcount() - 1;
5908 return std::pair<unsigned, int>{0, SrcIdx};
5909 }
5910 return std::pair<unsigned, int>{1, static_cast<int>(DstIdx)};
5911 });
5912 }
5913 case clang::X86::BI__builtin_ia32_blendpd:
5914 case clang::X86::BI__builtin_ia32_blendpd256:
5915 case clang::X86::BI__builtin_ia32_blendps:
5916 case clang::X86::BI__builtin_ia32_blendps256:
5917 case clang::X86::BI__builtin_ia32_pblendw128:
5918 case clang::X86::BI__builtin_ia32_pblendw256:
5919 case clang::X86::BI__builtin_ia32_pblendd128:
5920 case clang::X86::BI__builtin_ia32_pblendd256:
5921 return interp__builtin_ia32_shuffle_generic(
5922 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
5923 // Bit index for mask.
5924 unsigned MaskBit = (ShuffleMask >> (DstIdx % 8)) & 0x1;
5925 unsigned SrcVecIdx = MaskBit ? 1 : 0; // 1 = TrueVec, 0 = FalseVec
5926 return std::pair<unsigned, int>{SrcVecIdx, static_cast<int>(DstIdx)};
5927 });
5928
5929
5930
5931 case clang::X86::BI__builtin_ia32_blendvpd:
5932 case clang::X86::BI__builtin_ia32_blendvpd256:
5933 case clang::X86::BI__builtin_ia32_blendvps:
5934 case clang::X86::BI__builtin_ia32_blendvps256:
5935 return interp__builtin_elementwise_triop_fp(
5936 S, OpPC, Call,
5937 Fn: [](const APFloat &F, const APFloat &T, const APFloat &C,
5938 llvm::RoundingMode) { return C.isNegative() ? T : F; });
5939
5940 case clang::X86::BI__builtin_ia32_pblendvb128:
5941 case clang::X86::BI__builtin_ia32_pblendvb256:
5942 return interp__builtin_elementwise_triop(
5943 S, OpPC, Call, Fn: [](const APSInt &F, const APSInt &T, const APSInt &C) {
5944 return ((APInt)C).isNegative() ? T : F;
5945 });
5946 case X86::BI__builtin_ia32_ptestz128:
5947 case X86::BI__builtin_ia32_ptestz256:
5948 case X86::BI__builtin_ia32_vtestzps:
5949 case X86::BI__builtin_ia32_vtestzps256:
5950 case X86::BI__builtin_ia32_vtestzpd:
5951 case X86::BI__builtin_ia32_vtestzpd256:
5952 return interp__builtin_ia32_test_op(
5953 S, OpPC, Call,
5954 Fn: [](const APInt &A, const APInt &B) { return (A & B) == 0; });
5955 case X86::BI__builtin_ia32_ptestc128:
5956 case X86::BI__builtin_ia32_ptestc256:
5957 case X86::BI__builtin_ia32_vtestcps:
5958 case X86::BI__builtin_ia32_vtestcps256:
5959 case X86::BI__builtin_ia32_vtestcpd:
5960 case X86::BI__builtin_ia32_vtestcpd256:
5961 return interp__builtin_ia32_test_op(
5962 S, OpPC, Call,
5963 Fn: [](const APInt &A, const APInt &B) { return (~A & B) == 0; });
5964 case X86::BI__builtin_ia32_ptestnzc128:
5965 case X86::BI__builtin_ia32_ptestnzc256:
5966 case X86::BI__builtin_ia32_vtestnzcps:
5967 case X86::BI__builtin_ia32_vtestnzcps256:
5968 case X86::BI__builtin_ia32_vtestnzcpd:
5969 case X86::BI__builtin_ia32_vtestnzcpd256:
5970 return interp__builtin_ia32_test_op(
5971 S, OpPC, Call, Fn: [](const APInt &A, const APInt &B) {
5972 return ((A & B) != 0) && ((~A & B) != 0);
5973 });
5974 case X86::BI__builtin_ia32_selectb_128:
5975 case X86::BI__builtin_ia32_selectb_256:
5976 case X86::BI__builtin_ia32_selectb_512:
5977 case X86::BI__builtin_ia32_selectw_128:
5978 case X86::BI__builtin_ia32_selectw_256:
5979 case X86::BI__builtin_ia32_selectw_512:
5980 case X86::BI__builtin_ia32_selectd_128:
5981 case X86::BI__builtin_ia32_selectd_256:
5982 case X86::BI__builtin_ia32_selectd_512:
5983 case X86::BI__builtin_ia32_selectq_128:
5984 case X86::BI__builtin_ia32_selectq_256:
5985 case X86::BI__builtin_ia32_selectq_512:
5986 case X86::BI__builtin_ia32_selectph_128:
5987 case X86::BI__builtin_ia32_selectph_256:
5988 case X86::BI__builtin_ia32_selectph_512:
5989 case X86::BI__builtin_ia32_selectpbf_128:
5990 case X86::BI__builtin_ia32_selectpbf_256:
5991 case X86::BI__builtin_ia32_selectpbf_512:
5992 case X86::BI__builtin_ia32_selectps_128:
5993 case X86::BI__builtin_ia32_selectps_256:
5994 case X86::BI__builtin_ia32_selectps_512:
5995 case X86::BI__builtin_ia32_selectpd_128:
5996 case X86::BI__builtin_ia32_selectpd_256:
5997 case X86::BI__builtin_ia32_selectpd_512:
5998 return interp__builtin_ia32_select(S, OpPC, Call);
5999
6000 case X86::BI__builtin_ia32_shufps:
6001 case X86::BI__builtin_ia32_shufps256:
6002 case X86::BI__builtin_ia32_shufps512:
6003 return interp__builtin_ia32_shuffle_generic(
6004 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6005 unsigned NumElemPerLane = 4;
6006 unsigned NumSelectableElems = NumElemPerLane / 2;
6007 unsigned BitsPerElem = 2;
6008 unsigned IndexMask = 0x3;
6009 unsigned MaskBits = 8;
6010 unsigned Lane = DstIdx / NumElemPerLane;
6011 unsigned ElemInLane = DstIdx % NumElemPerLane;
6012 unsigned LaneOffset = Lane * NumElemPerLane;
6013 unsigned SrcIdx = ElemInLane >= NumSelectableElems ? 1 : 0;
6014 unsigned BitIndex = (DstIdx * BitsPerElem) % MaskBits;
6015 unsigned Index = (ShuffleMask >> BitIndex) & IndexMask;
6016 return std::pair<unsigned, int>{SrcIdx,
6017 static_cast<int>(LaneOffset + Index)};
6018 });
6019 case X86::BI__builtin_ia32_shufpd:
6020 case X86::BI__builtin_ia32_shufpd256:
6021 case X86::BI__builtin_ia32_shufpd512:
6022 return interp__builtin_ia32_shuffle_generic(
6023 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6024 unsigned NumElemPerLane = 2;
6025 unsigned NumSelectableElems = NumElemPerLane / 2;
6026 unsigned BitsPerElem = 1;
6027 unsigned IndexMask = 0x1;
6028 unsigned MaskBits = 8;
6029 unsigned Lane = DstIdx / NumElemPerLane;
6030 unsigned ElemInLane = DstIdx % NumElemPerLane;
6031 unsigned LaneOffset = Lane * NumElemPerLane;
6032 unsigned SrcIdx = ElemInLane >= NumSelectableElems ? 1 : 0;
6033 unsigned BitIndex = (DstIdx * BitsPerElem) % MaskBits;
6034 unsigned Index = (ShuffleMask >> BitIndex) & IndexMask;
6035 return std::pair<unsigned, int>{SrcIdx,
6036 static_cast<int>(LaneOffset + Index)};
6037 });
6038
6039 case X86::BI__builtin_ia32_vgf2p8affineinvqb_v16qi:
6040 case X86::BI__builtin_ia32_vgf2p8affineinvqb_v32qi:
6041 case X86::BI__builtin_ia32_vgf2p8affineinvqb_v64qi:
6042 return interp__builtin_ia32_gfni_affine(S, OpPC, Call, Inverse: true);
6043 case X86::BI__builtin_ia32_vgf2p8affineqb_v16qi:
6044 case X86::BI__builtin_ia32_vgf2p8affineqb_v32qi:
6045 case X86::BI__builtin_ia32_vgf2p8affineqb_v64qi:
6046 return interp__builtin_ia32_gfni_affine(S, OpPC, Call, Inverse: false);
6047
6048 case X86::BI__builtin_ia32_vgf2p8mulb_v16qi:
6049 case X86::BI__builtin_ia32_vgf2p8mulb_v32qi:
6050 case X86::BI__builtin_ia32_vgf2p8mulb_v64qi:
6051 return interp__builtin_ia32_gfni_mul(S, OpPC, Call);
6052
6053 case X86::BI__builtin_ia32_bmacor16x16x16_v16hi:
6054 case X86::BI__builtin_ia32_bmacor16x16x16_v32hi:
6055 return interp__builtin_ia32_bmac(S, OpPC, Call, /*IsXor=*/false);
6056 case X86::BI__builtin_ia32_bmacxor16x16x16_v16hi:
6057 case X86::BI__builtin_ia32_bmacxor16x16x16_v32hi:
6058 return interp__builtin_ia32_bmac(S, OpPC, Call, /*IsXor=*/true);
6059
6060 case X86::BI__builtin_ia32_insertps128:
6061 return interp__builtin_ia32_shuffle_generic(
6062 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned Mask) {
6063 // Bits [3:0]: zero mask - if bit is set, zero this element
6064 if ((Mask & (1 << DstIdx)) != 0) {
6065 return std::pair<unsigned, int>{0, -1};
6066 }
6067 // Bits [7:6]: select element from source vector Y (0-3)
6068 // Bits [5:4]: select destination position (0-3)
6069 unsigned SrcElem = (Mask >> 6) & 0x3;
6070 unsigned DstElem = (Mask >> 4) & 0x3;
6071 if (DstIdx == DstElem) {
6072 // Insert element from source vector (B) at this position
6073 return std::pair<unsigned, int>{1, static_cast<int>(SrcElem)};
6074 } else {
6075 // Copy from destination vector (A)
6076 return std::pair<unsigned, int>{0, static_cast<int>(DstIdx)};
6077 }
6078 });
6079 case X86::BI__builtin_ia32_permvarsi256:
6080 case X86::BI__builtin_ia32_permvarsf256:
6081 case X86::BI__builtin_ia32_permvardf512:
6082 case X86::BI__builtin_ia32_permvardi512:
6083 case X86::BI__builtin_ia32_permvarhi128:
6084 return interp__builtin_ia32_shuffle_generic(
6085 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6086 int Offset = ShuffleMask & 0x7;
6087 return std::pair<unsigned, int>{0, Offset};
6088 });
6089 case X86::BI__builtin_ia32_permvarqi128:
6090 case X86::BI__builtin_ia32_permvarhi256:
6091 case X86::BI__builtin_ia32_permvarsi512:
6092 case X86::BI__builtin_ia32_permvarsf512:
6093 return interp__builtin_ia32_shuffle_generic(
6094 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6095 int Offset = ShuffleMask & 0xF;
6096 return std::pair<unsigned, int>{0, Offset};
6097 });
6098 case X86::BI__builtin_ia32_permvardi256:
6099 case X86::BI__builtin_ia32_permvardf256:
6100 return interp__builtin_ia32_shuffle_generic(
6101 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6102 int Offset = ShuffleMask & 0x3;
6103 return std::pair<unsigned, int>{0, Offset};
6104 });
6105 case X86::BI__builtin_ia32_permvarqi256:
6106 case X86::BI__builtin_ia32_permvarhi512:
6107 return interp__builtin_ia32_shuffle_generic(
6108 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6109 int Offset = ShuffleMask & 0x1F;
6110 return std::pair<unsigned, int>{0, Offset};
6111 });
6112 case X86::BI__builtin_ia32_permvarqi512:
6113 return interp__builtin_ia32_shuffle_generic(
6114 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6115 int Offset = ShuffleMask & 0x3F;
6116 return std::pair<unsigned, int>{0, Offset};
6117 });
6118 case X86::BI__builtin_ia32_vpermi2varq128:
6119 case X86::BI__builtin_ia32_vpermi2varpd128:
6120 return interp__builtin_ia32_shuffle_generic(
6121 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6122 int Offset = ShuffleMask & 0x1;
6123 unsigned SrcIdx = (ShuffleMask >> 1) & 0x1;
6124 return std::pair<unsigned, int>{SrcIdx, Offset};
6125 });
6126 case X86::BI__builtin_ia32_vpermi2vard128:
6127 case X86::BI__builtin_ia32_vpermi2varps128:
6128 case X86::BI__builtin_ia32_vpermi2varq256:
6129 case X86::BI__builtin_ia32_vpermi2varpd256:
6130 return interp__builtin_ia32_shuffle_generic(
6131 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6132 int Offset = ShuffleMask & 0x3;
6133 unsigned SrcIdx = (ShuffleMask >> 2) & 0x1;
6134 return std::pair<unsigned, int>{SrcIdx, Offset};
6135 });
6136 case X86::BI__builtin_ia32_vpermi2varhi128:
6137 case X86::BI__builtin_ia32_vpermi2vard256:
6138 case X86::BI__builtin_ia32_vpermi2varps256:
6139 case X86::BI__builtin_ia32_vpermi2varq512:
6140 case X86::BI__builtin_ia32_vpermi2varpd512:
6141 return interp__builtin_ia32_shuffle_generic(
6142 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6143 int Offset = ShuffleMask & 0x7;
6144 unsigned SrcIdx = (ShuffleMask >> 3) & 0x1;
6145 return std::pair<unsigned, int>{SrcIdx, Offset};
6146 });
6147 case X86::BI__builtin_ia32_vpermi2varqi128:
6148 case X86::BI__builtin_ia32_vpermi2varhi256:
6149 case X86::BI__builtin_ia32_vpermi2vard512:
6150 case X86::BI__builtin_ia32_vpermi2varps512:
6151 return interp__builtin_ia32_shuffle_generic(
6152 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6153 int Offset = ShuffleMask & 0xF;
6154 unsigned SrcIdx = (ShuffleMask >> 4) & 0x1;
6155 return std::pair<unsigned, int>{SrcIdx, Offset};
6156 });
6157 case X86::BI__builtin_ia32_vpermi2varqi256:
6158 case X86::BI__builtin_ia32_vpermi2varhi512:
6159 return interp__builtin_ia32_shuffle_generic(
6160 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6161 int Offset = ShuffleMask & 0x1F;
6162 unsigned SrcIdx = (ShuffleMask >> 5) & 0x1;
6163 return std::pair<unsigned, int>{SrcIdx, Offset};
6164 });
6165 case X86::BI__builtin_ia32_vpermi2varqi512:
6166 return interp__builtin_ia32_shuffle_generic(
6167 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6168 int Offset = ShuffleMask & 0x3F;
6169 unsigned SrcIdx = (ShuffleMask >> 6) & 0x1;
6170 return std::pair<unsigned, int>{SrcIdx, Offset};
6171 });
6172 case X86::BI__builtin_ia32_vperm2f128_pd256:
6173 case X86::BI__builtin_ia32_vperm2f128_ps256:
6174 case X86::BI__builtin_ia32_vperm2f128_si256:
6175 case X86::BI__builtin_ia32_permti256: {
6176 unsigned NumElements =
6177 Call->getArg(Arg: 0)->getType()->castAs<VectorType>()->getNumElements();
6178 unsigned PreservedBitsCnt = NumElements >> 2;
6179 return interp__builtin_ia32_shuffle_generic(
6180 S, OpPC, Call,
6181 GetSourceIndex: [PreservedBitsCnt](unsigned DstIdx, unsigned ShuffleMask) {
6182 unsigned ControlBitsCnt = DstIdx >> PreservedBitsCnt << 2;
6183 unsigned ControlBits = ShuffleMask >> ControlBitsCnt;
6184
6185 if (ControlBits & 0b1000)
6186 return std::make_pair(x: 0u, y: -1);
6187
6188 unsigned SrcVecIdx = (ControlBits & 0b10) >> 1;
6189 unsigned PreservedBitsMask = (1 << PreservedBitsCnt) - 1;
6190 int SrcIdx = ((ControlBits & 0b1) << PreservedBitsCnt) |
6191 (DstIdx & PreservedBitsMask);
6192 return std::make_pair(x&: SrcVecIdx, y&: SrcIdx);
6193 });
6194 }
6195 case X86::BI__builtin_ia32_pshufb128:
6196 case X86::BI__builtin_ia32_pshufb256:
6197 case X86::BI__builtin_ia32_pshufb512:
6198 return interp__builtin_ia32_shuffle_generic(
6199 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6200 uint8_t Ctlb = static_cast<uint8_t>(ShuffleMask);
6201 if (Ctlb & 0x80)
6202 return std::make_pair(x: 0, y: -1);
6203
6204 unsigned LaneBase = (DstIdx / 16) * 16;
6205 unsigned SrcOffset = Ctlb & 0x0F;
6206 unsigned SrcIdx = LaneBase + SrcOffset;
6207 return std::make_pair(x: 0, y: static_cast<int>(SrcIdx));
6208 });
6209
6210 case X86::BI__builtin_ia32_pshuflw:
6211 case X86::BI__builtin_ia32_pshuflw256:
6212 case X86::BI__builtin_ia32_pshuflw512:
6213 return interp__builtin_ia32_shuffle_generic(
6214 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6215 unsigned LaneBase = (DstIdx / 8) * 8;
6216 unsigned LaneIdx = DstIdx % 8;
6217 if (LaneIdx < 4) {
6218 unsigned Sel = (ShuffleMask >> (2 * LaneIdx)) & 0x3;
6219 return std::make_pair(x: 0, y: static_cast<int>(LaneBase + Sel));
6220 }
6221
6222 return std::make_pair(x: 0, y: static_cast<int>(DstIdx));
6223 });
6224
6225 case X86::BI__builtin_ia32_pshufhw:
6226 case X86::BI__builtin_ia32_pshufhw256:
6227 case X86::BI__builtin_ia32_pshufhw512:
6228 return interp__builtin_ia32_shuffle_generic(
6229 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6230 unsigned LaneBase = (DstIdx / 8) * 8;
6231 unsigned LaneIdx = DstIdx % 8;
6232 if (LaneIdx >= 4) {
6233 unsigned Sel = (ShuffleMask >> (2 * (LaneIdx - 4))) & 0x3;
6234 return std::make_pair(x: 0, y: static_cast<int>(LaneBase + 4 + Sel));
6235 }
6236
6237 return std::make_pair(x: 0, y: static_cast<int>(DstIdx));
6238 });
6239
6240 case X86::BI__builtin_ia32_pshufd:
6241 case X86::BI__builtin_ia32_pshufd256:
6242 case X86::BI__builtin_ia32_pshufd512:
6243 case X86::BI__builtin_ia32_vpermilps:
6244 case X86::BI__builtin_ia32_vpermilps256:
6245 case X86::BI__builtin_ia32_vpermilps512:
6246 return interp__builtin_ia32_shuffle_generic(
6247 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6248 unsigned LaneBase = (DstIdx / 4) * 4;
6249 unsigned LaneIdx = DstIdx % 4;
6250 unsigned Sel = (ShuffleMask >> (2 * LaneIdx)) & 0x3;
6251 return std::make_pair(x: 0, y: static_cast<int>(LaneBase + Sel));
6252 });
6253
6254 case X86::BI__builtin_ia32_vpermilvarpd:
6255 case X86::BI__builtin_ia32_vpermilvarpd256:
6256 case X86::BI__builtin_ia32_vpermilvarpd512:
6257 return interp__builtin_ia32_shuffle_generic(
6258 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6259 unsigned NumElemPerLane = 2;
6260 unsigned Lane = DstIdx / NumElemPerLane;
6261 unsigned Offset = ShuffleMask & 0b10 ? 1 : 0;
6262 return std::make_pair(
6263 x: 0, y: static_cast<int>(Lane * NumElemPerLane + Offset));
6264 });
6265
6266 case X86::BI__builtin_ia32_vpermilvarps:
6267 case X86::BI__builtin_ia32_vpermilvarps256:
6268 case X86::BI__builtin_ia32_vpermilvarps512:
6269 return interp__builtin_ia32_shuffle_generic(
6270 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned ShuffleMask) {
6271 unsigned NumElemPerLane = 4;
6272 unsigned Lane = DstIdx / NumElemPerLane;
6273 unsigned Offset = ShuffleMask & 0b11;
6274 return std::make_pair(
6275 x: 0, y: static_cast<int>(Lane * NumElemPerLane + Offset));
6276 });
6277
6278 case X86::BI__builtin_ia32_vpermilpd:
6279 case X86::BI__builtin_ia32_vpermilpd256:
6280 case X86::BI__builtin_ia32_vpermilpd512:
6281 return interp__builtin_ia32_shuffle_generic(
6282 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned Control) {
6283 unsigned NumElemPerLane = 2;
6284 unsigned BitsPerElem = 1;
6285 unsigned MaskBits = 8;
6286 unsigned IndexMask = 0x1;
6287 unsigned Lane = DstIdx / NumElemPerLane;
6288 unsigned LaneOffset = Lane * NumElemPerLane;
6289 unsigned BitIndex = (DstIdx * BitsPerElem) % MaskBits;
6290 unsigned Index = (Control >> BitIndex) & IndexMask;
6291 return std::make_pair(x: 0, y: static_cast<int>(LaneOffset + Index));
6292 });
6293
6294 case X86::BI__builtin_ia32_permdf256:
6295 case X86::BI__builtin_ia32_permdi256:
6296 return interp__builtin_ia32_shuffle_generic(
6297 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned Control) {
6298 // permute4x64 operates on 4 64-bit elements
6299 // For element i (0-3), extract bits [2*i+1:2*i] from Control
6300 unsigned Index = (Control >> (2 * DstIdx)) & 0x3;
6301 return std::make_pair(x: 0, y: static_cast<int>(Index));
6302 });
6303
6304 case X86::BI__builtin_ia32_vpmultishiftqb128:
6305 case X86::BI__builtin_ia32_vpmultishiftqb256:
6306 case X86::BI__builtin_ia32_vpmultishiftqb512:
6307 return interp__builtin_ia32_multishiftqb(S, OpPC, Call);
6308 case X86::BI__builtin_ia32_kandqi:
6309 case X86::BI__builtin_ia32_kandhi:
6310 case X86::BI__builtin_ia32_kandsi:
6311 case X86::BI__builtin_ia32_kanddi:
6312 return interp__builtin_elementwise_int_binop(
6313 S, OpPC, Call,
6314 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS & RHS; });
6315
6316 case X86::BI__builtin_ia32_kandnqi:
6317 case X86::BI__builtin_ia32_kandnhi:
6318 case X86::BI__builtin_ia32_kandnsi:
6319 case X86::BI__builtin_ia32_kandndi:
6320 return interp__builtin_elementwise_int_binop(
6321 S, OpPC, Call,
6322 Fn: [](const APSInt &LHS, const APSInt &RHS) { return ~LHS & RHS; });
6323
6324 case X86::BI__builtin_ia32_korqi:
6325 case X86::BI__builtin_ia32_korhi:
6326 case X86::BI__builtin_ia32_korsi:
6327 case X86::BI__builtin_ia32_kordi:
6328 return interp__builtin_elementwise_int_binop(
6329 S, OpPC, Call,
6330 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS | RHS; });
6331
6332 case X86::BI__builtin_ia32_kxnorqi:
6333 case X86::BI__builtin_ia32_kxnorhi:
6334 case X86::BI__builtin_ia32_kxnorsi:
6335 case X86::BI__builtin_ia32_kxnordi:
6336 return interp__builtin_elementwise_int_binop(
6337 S, OpPC, Call,
6338 Fn: [](const APSInt &LHS, const APSInt &RHS) { return ~(LHS ^ RHS); });
6339
6340 case X86::BI__builtin_ia32_kxorqi:
6341 case X86::BI__builtin_ia32_kxorhi:
6342 case X86::BI__builtin_ia32_kxorsi:
6343 case X86::BI__builtin_ia32_kxordi:
6344 return interp__builtin_elementwise_int_binop(
6345 S, OpPC, Call,
6346 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS ^ RHS; });
6347
6348 case X86::BI__builtin_ia32_knotqi:
6349 case X86::BI__builtin_ia32_knothi:
6350 case X86::BI__builtin_ia32_knotsi:
6351 case X86::BI__builtin_ia32_knotdi:
6352 return interp__builtin_elementwise_int_unaryop(
6353 S, OpPC, Call, Fn: [](const APSInt &Src) { return ~Src; });
6354
6355 case X86::BI__builtin_ia32_kaddqi:
6356 case X86::BI__builtin_ia32_kaddhi:
6357 case X86::BI__builtin_ia32_kaddsi:
6358 case X86::BI__builtin_ia32_kadddi:
6359 return interp__builtin_elementwise_int_binop(
6360 S, OpPC, Call,
6361 Fn: [](const APSInt &LHS, const APSInt &RHS) { return LHS + RHS; });
6362
6363 case X86::BI__builtin_ia32_kmovb:
6364 case X86::BI__builtin_ia32_kmovw:
6365 case X86::BI__builtin_ia32_kmovd:
6366 case X86::BI__builtin_ia32_kmovq:
6367 return interp__builtin_elementwise_int_unaryop(
6368 S, OpPC, Call, Fn: [](const APSInt &Src) { return Src; });
6369
6370 case X86::BI__builtin_ia32_kunpckhi:
6371 case X86::BI__builtin_ia32_kunpckdi:
6372 case X86::BI__builtin_ia32_kunpcksi:
6373 return interp__builtin_elementwise_int_binop(
6374 S, OpPC, Call, Fn: [](const APSInt &A, const APSInt &B) {
6375 // Generic kunpack: extract lower half of each operand and concatenate
6376 // Result = A[HalfWidth-1:0] concat B[HalfWidth-1:0]
6377 unsigned BW = A.getBitWidth();
6378 return APSInt(A.trunc(width: BW / 2).concat(NewLSB: B.trunc(width: BW / 2)),
6379 A.isUnsigned());
6380 });
6381
6382 case X86::BI__builtin_ia32_phminposuw128:
6383 return interp__builtin_ia32_phminposuw(S, OpPC, Call);
6384
6385 case X86::BI__builtin_ia32_psraq128:
6386 case X86::BI__builtin_ia32_psraq256:
6387 case X86::BI__builtin_ia32_psraq512:
6388 case X86::BI__builtin_ia32_psrad128:
6389 case X86::BI__builtin_ia32_psrad256:
6390 case X86::BI__builtin_ia32_psrad512:
6391 case X86::BI__builtin_ia32_psraw128:
6392 case X86::BI__builtin_ia32_psraw256:
6393 case X86::BI__builtin_ia32_psraw512:
6394 return interp__builtin_ia32_shift_with_count(
6395 S, OpPC, Call,
6396 ShiftOp: [](const APInt &Elt, uint64_t Count) { return Elt.ashr(ShiftAmt: Count); },
6397 OverflowOp: [](const APInt &Elt, unsigned Width) { return Elt.ashr(ShiftAmt: Width - 1); });
6398
6399 case X86::BI__builtin_ia32_psllq128:
6400 case X86::BI__builtin_ia32_psllq256:
6401 case X86::BI__builtin_ia32_psllq512:
6402 case X86::BI__builtin_ia32_pslld128:
6403 case X86::BI__builtin_ia32_pslld256:
6404 case X86::BI__builtin_ia32_pslld512:
6405 case X86::BI__builtin_ia32_psllw128:
6406 case X86::BI__builtin_ia32_psllw256:
6407 case X86::BI__builtin_ia32_psllw512:
6408 return interp__builtin_ia32_shift_with_count(
6409 S, OpPC, Call,
6410 ShiftOp: [](const APInt &Elt, uint64_t Count) { return Elt.shl(shiftAmt: Count); },
6411 OverflowOp: [](const APInt &Elt, unsigned Width) { return APInt::getZero(numBits: Width); });
6412
6413 case X86::BI__builtin_ia32_psrlq128:
6414 case X86::BI__builtin_ia32_psrlq256:
6415 case X86::BI__builtin_ia32_psrlq512:
6416 case X86::BI__builtin_ia32_psrld128:
6417 case X86::BI__builtin_ia32_psrld256:
6418 case X86::BI__builtin_ia32_psrld512:
6419 case X86::BI__builtin_ia32_psrlw128:
6420 case X86::BI__builtin_ia32_psrlw256:
6421 case X86::BI__builtin_ia32_psrlw512:
6422 return interp__builtin_ia32_shift_with_count(
6423 S, OpPC, Call,
6424 ShiftOp: [](const APInt &Elt, uint64_t Count) { return Elt.lshr(shiftAmt: Count); },
6425 OverflowOp: [](const APInt &Elt, unsigned Width) { return APInt::getZero(numBits: Width); });
6426
6427 case X86::BI__builtin_ia32_pternlogd128_mask:
6428 case X86::BI__builtin_ia32_pternlogd256_mask:
6429 case X86::BI__builtin_ia32_pternlogd512_mask:
6430 case X86::BI__builtin_ia32_pternlogq128_mask:
6431 case X86::BI__builtin_ia32_pternlogq256_mask:
6432 case X86::BI__builtin_ia32_pternlogq512_mask:
6433 return interp__builtin_ia32_pternlog(S, OpPC, Call, /*MaskZ=*/false);
6434 case X86::BI__builtin_ia32_pternlogd128_maskz:
6435 case X86::BI__builtin_ia32_pternlogd256_maskz:
6436 case X86::BI__builtin_ia32_pternlogd512_maskz:
6437 case X86::BI__builtin_ia32_pternlogq128_maskz:
6438 case X86::BI__builtin_ia32_pternlogq256_maskz:
6439 case X86::BI__builtin_ia32_pternlogq512_maskz:
6440 return interp__builtin_ia32_pternlog(S, OpPC, Call, /*MaskZ=*/true);
6441 case Builtin::BI__builtin_elementwise_fshl:
6442 return interp__builtin_elementwise_triop(S, OpPC, Call,
6443 Fn: llvm::APIntOps::fshl);
6444 case Builtin::BI__builtin_elementwise_fshr:
6445 return interp__builtin_elementwise_triop(S, OpPC, Call,
6446 Fn: llvm::APIntOps::fshr);
6447
6448 case X86::BI__builtin_ia32_shuf_f32x4_256:
6449 case X86::BI__builtin_ia32_shuf_i32x4_256:
6450 case X86::BI__builtin_ia32_shuf_f64x2_256:
6451 case X86::BI__builtin_ia32_shuf_i64x2_256:
6452 case X86::BI__builtin_ia32_shuf_f32x4:
6453 case X86::BI__builtin_ia32_shuf_i32x4:
6454 case X86::BI__builtin_ia32_shuf_f64x2:
6455 case X86::BI__builtin_ia32_shuf_i64x2: {
6456 // Destination and sources A, B all have the same type.
6457 QualType VecQT = Call->getArg(Arg: 0)->getType();
6458 const auto *VecT = VecQT->castAs<VectorType>();
6459 unsigned NumElems = VecT->getNumElements();
6460 unsigned ElemBits = S.getASTContext().getTypeSize(T: VecT->getElementType());
6461 unsigned LaneBits = 128u;
6462 unsigned NumLanes = (NumElems * ElemBits) / LaneBits;
6463 unsigned NumElemsPerLane = LaneBits / ElemBits;
6464
6465 return interp__builtin_ia32_shuffle_generic(
6466 S, OpPC, Call,
6467 GetSourceIndex: [NumLanes, NumElemsPerLane](unsigned DstIdx, unsigned ShuffleMask) {
6468 // DstIdx determines source. ShuffleMask selects lane in source.
6469 unsigned BitsPerElem = NumLanes / 2;
6470 unsigned IndexMask = (1u << BitsPerElem) - 1;
6471 unsigned Lane = DstIdx / NumElemsPerLane;
6472 unsigned SrcIdx = (Lane < NumLanes / 2) ? 0 : 1;
6473 unsigned BitIdx = BitsPerElem * Lane;
6474 unsigned SrcLaneIdx = (ShuffleMask >> BitIdx) & IndexMask;
6475 unsigned ElemInLane = DstIdx % NumElemsPerLane;
6476 unsigned IdxToPick = SrcLaneIdx * NumElemsPerLane + ElemInLane;
6477 return std::pair<unsigned, int>{SrcIdx, IdxToPick};
6478 });
6479 }
6480
6481 case X86::BI__builtin_ia32_insertf32x4_256:
6482 case X86::BI__builtin_ia32_inserti32x4_256:
6483 case X86::BI__builtin_ia32_insertf64x2_256:
6484 case X86::BI__builtin_ia32_inserti64x2_256:
6485 case X86::BI__builtin_ia32_insertf32x4:
6486 case X86::BI__builtin_ia32_inserti32x4:
6487 case X86::BI__builtin_ia32_insertf64x2_512:
6488 case X86::BI__builtin_ia32_inserti64x2_512:
6489 case X86::BI__builtin_ia32_insertf32x8:
6490 case X86::BI__builtin_ia32_inserti32x8:
6491 case X86::BI__builtin_ia32_insertf64x4:
6492 case X86::BI__builtin_ia32_inserti64x4:
6493 case X86::BI__builtin_ia32_vinsertf128_ps256:
6494 case X86::BI__builtin_ia32_vinsertf128_pd256:
6495 case X86::BI__builtin_ia32_vinsertf128_si256:
6496 case X86::BI__builtin_ia32_insert128i256:
6497 return interp__builtin_ia32_insert_subvector(S, OpPC, Call, ID: BuiltinID);
6498
6499 case clang::X86::BI__builtin_ia32_vcvtps2ph:
6500 case clang::X86::BI__builtin_ia32_vcvtps2ph256:
6501 return interp__builtin_ia32_vcvtps2ph(S, OpPC, Call);
6502
6503 case X86::BI__builtin_ia32_vec_ext_v4hi:
6504 case X86::BI__builtin_ia32_vec_ext_v16qi:
6505 case X86::BI__builtin_ia32_vec_ext_v8hi:
6506 case X86::BI__builtin_ia32_vec_ext_v4si:
6507 case X86::BI__builtin_ia32_vec_ext_v2di:
6508 case X86::BI__builtin_ia32_vec_ext_v32qi:
6509 case X86::BI__builtin_ia32_vec_ext_v16hi:
6510 case X86::BI__builtin_ia32_vec_ext_v8si:
6511 case X86::BI__builtin_ia32_vec_ext_v4di:
6512 case X86::BI__builtin_ia32_vec_ext_v4sf:
6513 return interp__builtin_ia32_vec_ext(S, OpPC, Call, ID: BuiltinID);
6514
6515 case X86::BI__builtin_ia32_vec_set_v4hi:
6516 case X86::BI__builtin_ia32_vec_set_v16qi:
6517 case X86::BI__builtin_ia32_vec_set_v8hi:
6518 case X86::BI__builtin_ia32_vec_set_v4si:
6519 case X86::BI__builtin_ia32_vec_set_v2di:
6520 case X86::BI__builtin_ia32_vec_set_v32qi:
6521 case X86::BI__builtin_ia32_vec_set_v16hi:
6522 case X86::BI__builtin_ia32_vec_set_v8si:
6523 case X86::BI__builtin_ia32_vec_set_v4di:
6524 return interp__builtin_ia32_vec_set(S, OpPC, Call, ID: BuiltinID);
6525
6526 case X86::BI__builtin_ia32_cvtb2mask128:
6527 case X86::BI__builtin_ia32_cvtb2mask256:
6528 case X86::BI__builtin_ia32_cvtb2mask512:
6529 case X86::BI__builtin_ia32_cvtw2mask128:
6530 case X86::BI__builtin_ia32_cvtw2mask256:
6531 case X86::BI__builtin_ia32_cvtw2mask512:
6532 case X86::BI__builtin_ia32_cvtd2mask128:
6533 case X86::BI__builtin_ia32_cvtd2mask256:
6534 case X86::BI__builtin_ia32_cvtd2mask512:
6535 case X86::BI__builtin_ia32_cvtq2mask128:
6536 case X86::BI__builtin_ia32_cvtq2mask256:
6537 case X86::BI__builtin_ia32_cvtq2mask512:
6538 return interp__builtin_ia32_cvt_vec2mask(S, OpPC, Call, ID: BuiltinID);
6539
6540 case X86::BI__builtin_ia32_cvtmask2b128:
6541 case X86::BI__builtin_ia32_cvtmask2b256:
6542 case X86::BI__builtin_ia32_cvtmask2b512:
6543 case X86::BI__builtin_ia32_cvtmask2w128:
6544 case X86::BI__builtin_ia32_cvtmask2w256:
6545 case X86::BI__builtin_ia32_cvtmask2w512:
6546 case X86::BI__builtin_ia32_cvtmask2d128:
6547 case X86::BI__builtin_ia32_cvtmask2d256:
6548 case X86::BI__builtin_ia32_cvtmask2d512:
6549 case X86::BI__builtin_ia32_cvtmask2q128:
6550 case X86::BI__builtin_ia32_cvtmask2q256:
6551 case X86::BI__builtin_ia32_cvtmask2q512:
6552 return interp__builtin_ia32_cvt_mask2vec(S, OpPC, Call, ID: BuiltinID);
6553
6554 case X86::BI__builtin_ia32_cvtsd2ss:
6555 return interp__builtin_ia32_cvtsd2ss(S, OpPC, Call, HasRoundingMask: false);
6556
6557 case X86::BI__builtin_ia32_cvtsd2ss_round_mask:
6558 return interp__builtin_ia32_cvtsd2ss(S, OpPC, Call, HasRoundingMask: true);
6559
6560 case X86::BI__builtin_ia32_cvtpd2ps:
6561 case X86::BI__builtin_ia32_cvtpd2ps256:
6562 return interp__builtin_ia32_cvtpd2ps(S, OpPC, Call, IsMasked: false, HasRounding: false);
6563 case X86::BI__builtin_ia32_cvtpd2ps_mask:
6564 return interp__builtin_ia32_cvtpd2ps(S, OpPC, Call, IsMasked: true, HasRounding: false);
6565 case X86::BI__builtin_ia32_cvtpd2ps512_mask:
6566 return interp__builtin_ia32_cvtpd2ps(S, OpPC, Call, IsMasked: true, HasRounding: true);
6567
6568 case X86::BI__builtin_ia32_cmpb128_mask:
6569 case X86::BI__builtin_ia32_cmpw128_mask:
6570 case X86::BI__builtin_ia32_cmpd128_mask:
6571 case X86::BI__builtin_ia32_cmpq128_mask:
6572 case X86::BI__builtin_ia32_cmpb256_mask:
6573 case X86::BI__builtin_ia32_cmpw256_mask:
6574 case X86::BI__builtin_ia32_cmpd256_mask:
6575 case X86::BI__builtin_ia32_cmpq256_mask:
6576 case X86::BI__builtin_ia32_cmpb512_mask:
6577 case X86::BI__builtin_ia32_cmpw512_mask:
6578 case X86::BI__builtin_ia32_cmpd512_mask:
6579 case X86::BI__builtin_ia32_cmpq512_mask:
6580 return interp__builtin_ia32_cmp_mask(S, OpPC, Call, ID: BuiltinID,
6581 /*IsUnsigned=*/false);
6582
6583 case X86::BI__builtin_ia32_ucmpb128_mask:
6584 case X86::BI__builtin_ia32_ucmpw128_mask:
6585 case X86::BI__builtin_ia32_ucmpd128_mask:
6586 case X86::BI__builtin_ia32_ucmpq128_mask:
6587 case X86::BI__builtin_ia32_ucmpb256_mask:
6588 case X86::BI__builtin_ia32_ucmpw256_mask:
6589 case X86::BI__builtin_ia32_ucmpd256_mask:
6590 case X86::BI__builtin_ia32_ucmpq256_mask:
6591 case X86::BI__builtin_ia32_ucmpb512_mask:
6592 case X86::BI__builtin_ia32_ucmpw512_mask:
6593 case X86::BI__builtin_ia32_ucmpd512_mask:
6594 case X86::BI__builtin_ia32_ucmpq512_mask:
6595 return interp__builtin_ia32_cmp_mask(S, OpPC, Call, ID: BuiltinID,
6596 /*IsUnsigned=*/true);
6597
6598 case X86::BI__builtin_ia32_vpshufbitqmb128_mask:
6599 case X86::BI__builtin_ia32_vpshufbitqmb256_mask:
6600 case X86::BI__builtin_ia32_vpshufbitqmb512_mask:
6601 return interp__builtin_ia32_shufbitqmb_mask(S, OpPC, Call);
6602
6603 case X86::BI__builtin_ia32_pslldqi128_byteshift:
6604 case X86::BI__builtin_ia32_pslldqi256_byteshift:
6605 case X86::BI__builtin_ia32_pslldqi512_byteshift:
6606 // These SLLDQ intrinsics always operate on byte elements (8 bits).
6607 // The lane width is hardcoded to 16 to match the SIMD register size,
6608 // but the algorithm processes one byte per iteration,
6609 // so APInt(8, ...) is correct and intentional.
6610 return interp__builtin_ia32_shuffle_generic(
6611 S, OpPC, Call,
6612 GetSourceIndex: [](unsigned DstIdx, unsigned Shift) -> std::pair<unsigned, int> {
6613 unsigned LaneBase = (DstIdx / 16) * 16;
6614 unsigned LaneIdx = DstIdx % 16;
6615 if (LaneIdx < Shift)
6616 return std::make_pair(x: 0, y: -1);
6617
6618 return std::make_pair(x: 0,
6619 y: static_cast<int>(LaneBase + LaneIdx - Shift));
6620 });
6621
6622 case X86::BI__builtin_ia32_psrldqi128_byteshift:
6623 case X86::BI__builtin_ia32_psrldqi256_byteshift:
6624 case X86::BI__builtin_ia32_psrldqi512_byteshift:
6625 // These SRLDQ intrinsics always operate on byte elements (8 bits).
6626 // The lane width is hardcoded to 16 to match the SIMD register size,
6627 // but the algorithm processes one byte per iteration,
6628 // so APInt(8, ...) is correct and intentional.
6629 return interp__builtin_ia32_shuffle_generic(
6630 S, OpPC, Call,
6631 GetSourceIndex: [](unsigned DstIdx, unsigned Shift) -> std::pair<unsigned, int> {
6632 unsigned LaneBase = (DstIdx / 16) * 16;
6633 unsigned LaneIdx = DstIdx % 16;
6634 if (LaneIdx + Shift < 16)
6635 return std::make_pair(x: 0,
6636 y: static_cast<int>(LaneBase + LaneIdx + Shift));
6637
6638 return std::make_pair(x: 0, y: -1);
6639 });
6640
6641 case X86::BI__builtin_ia32_palignr128:
6642 case X86::BI__builtin_ia32_palignr256:
6643 case X86::BI__builtin_ia32_palignr512:
6644 return interp__builtin_ia32_shuffle_generic(
6645 S, OpPC, Call, GetSourceIndex: [](unsigned DstIdx, unsigned Shift) {
6646 // Default to -1 → zero-fill this destination element
6647 unsigned VecIdx = 1;
6648 int ElemIdx = -1;
6649
6650 int Lane = DstIdx / 16;
6651 int Offset = DstIdx % 16;
6652
6653 // Elements come from VecB first, then VecA after the shift boundary
6654 unsigned ShiftedIdx = Offset + (Shift & 0xFF);
6655 if (ShiftedIdx < 16) { // from VecB
6656 ElemIdx = ShiftedIdx + (Lane * 16);
6657 } else if (ShiftedIdx < 32) { // from VecA
6658 VecIdx = 0;
6659 ElemIdx = (ShiftedIdx - 16) + (Lane * 16);
6660 }
6661
6662 return std::pair<unsigned, int>{VecIdx, ElemIdx};
6663 });
6664
6665 case X86::BI__builtin_ia32_alignd128:
6666 case X86::BI__builtin_ia32_alignd256:
6667 case X86::BI__builtin_ia32_alignd512:
6668 case X86::BI__builtin_ia32_alignq128:
6669 case X86::BI__builtin_ia32_alignq256:
6670 case X86::BI__builtin_ia32_alignq512: {
6671 unsigned NumElems = Call->getType()->castAs<VectorType>()->getNumElements();
6672 return interp__builtin_ia32_shuffle_generic(
6673 S, OpPC, Call, GetSourceIndex: [NumElems](unsigned DstIdx, unsigned Shift) {
6674 unsigned Imm = Shift & 0xFF;
6675 unsigned EffectiveShift = Imm & (NumElems - 1);
6676 unsigned SourcePos = DstIdx + EffectiveShift;
6677 unsigned VecIdx = SourcePos < NumElems ? 1u : 0u;
6678 unsigned ElemIdx = SourcePos & (NumElems - 1);
6679 return std::pair<unsigned, int>{VecIdx, static_cast<int>(ElemIdx)};
6680 });
6681 }
6682
6683 case clang::X86::BI__builtin_ia32_minps:
6684 case clang::X86::BI__builtin_ia32_minpd:
6685 case clang::X86::BI__builtin_ia32_minph128:
6686 case clang::X86::BI__builtin_ia32_minph256:
6687 case clang::X86::BI__builtin_ia32_minps256:
6688 case clang::X86::BI__builtin_ia32_minpd256:
6689 case clang::X86::BI__builtin_ia32_minps512:
6690 case clang::X86::BI__builtin_ia32_minpd512:
6691 case clang::X86::BI__builtin_ia32_minph512:
6692 return interp__builtin_elementwise_fp_binop(
6693 S, OpPC, Call,
6694 Fn: [](const APFloat &A, const APFloat &B,
6695 std::optional<APSInt>) -> std::optional<APFloat> {
6696 if (A.isNaN() || A.isInfinity() || A.isDenormal() || B.isNaN() ||
6697 B.isInfinity() || B.isDenormal())
6698 return std::nullopt;
6699 if (A.isZero() && B.isZero())
6700 return B;
6701 return llvm::minimum(A, B);
6702 });
6703
6704 case clang::X86::BI__builtin_ia32_minss:
6705 case clang::X86::BI__builtin_ia32_minsd:
6706 return interp__builtin_elementwise_fp_binop(
6707 S, OpPC, Call,
6708 Fn: [](const APFloat &A, const APFloat &B,
6709 std::optional<APSInt> RoundingMode) -> std::optional<APFloat> {
6710 return EvalScalarMinMaxFp(A, B, RoundingMode, /*IsMin=*/true);
6711 },
6712 /*IsScalar=*/true);
6713
6714 case clang::X86::BI__builtin_ia32_minsd_round_mask:
6715 case clang::X86::BI__builtin_ia32_minss_round_mask:
6716 case clang::X86::BI__builtin_ia32_minsh_round_mask:
6717 case clang::X86::BI__builtin_ia32_maxsd_round_mask:
6718 case clang::X86::BI__builtin_ia32_maxss_round_mask:
6719 case clang::X86::BI__builtin_ia32_maxsh_round_mask: {
6720 bool IsMin = BuiltinID == clang::X86::BI__builtin_ia32_minsd_round_mask ||
6721 BuiltinID == clang::X86::BI__builtin_ia32_minss_round_mask ||
6722 BuiltinID == clang::X86::BI__builtin_ia32_minsh_round_mask;
6723 return interp__builtin_scalar_fp_round_mask_binop(
6724 S, OpPC, Call,
6725 Fn: [IsMin](const APFloat &A, const APFloat &B,
6726 std::optional<APSInt> RoundingMode) -> std::optional<APFloat> {
6727 return EvalScalarMinMaxFp(A, B, RoundingMode, IsMin);
6728 });
6729 }
6730
6731 case clang::X86::BI__builtin_ia32_maxps:
6732 case clang::X86::BI__builtin_ia32_maxpd:
6733 case clang::X86::BI__builtin_ia32_maxph128:
6734 case clang::X86::BI__builtin_ia32_maxph256:
6735 case clang::X86::BI__builtin_ia32_maxps256:
6736 case clang::X86::BI__builtin_ia32_maxpd256:
6737 case clang::X86::BI__builtin_ia32_maxps512:
6738 case clang::X86::BI__builtin_ia32_maxpd512:
6739 case clang::X86::BI__builtin_ia32_maxph512:
6740 return interp__builtin_elementwise_fp_binop(
6741 S, OpPC, Call,
6742 Fn: [](const APFloat &A, const APFloat &B,
6743 std::optional<APSInt>) -> std::optional<APFloat> {
6744 if (A.isNaN() || A.isInfinity() || A.isDenormal() || B.isNaN() ||
6745 B.isInfinity() || B.isDenormal())
6746 return std::nullopt;
6747 if (A.isZero() && B.isZero())
6748 return B;
6749 return llvm::maximum(A, B);
6750 });
6751
6752 case clang::X86::BI__builtin_ia32_maxss:
6753 case clang::X86::BI__builtin_ia32_maxsd:
6754 return interp__builtin_elementwise_fp_binop(
6755 S, OpPC, Call,
6756 Fn: [](const APFloat &A, const APFloat &B,
6757 std::optional<APSInt> RoundingMode) -> std::optional<APFloat> {
6758 return EvalScalarMinMaxFp(A, B, RoundingMode, /*IsMin=*/false);
6759 },
6760 /*IsScalar=*/true);
6761 case X86::BI__builtin_ia32_vpdpwssd128:
6762 case X86::BI__builtin_ia32_vpdpwssd256:
6763 case X86::BI__builtin_ia32_vpdpwssd512:
6764 case X86::BI__builtin_ia32_vpdpbusd128:
6765 case X86::BI__builtin_ia32_vpdpbusd256:
6766 case X86::BI__builtin_ia32_vpdpbusd512:
6767 return interp__builtin_ia32_vpdp(S, OpPC, Call, IsSaturating: false);
6768 case X86::BI__builtin_ia32_vpdpwssds128:
6769 case X86::BI__builtin_ia32_vpdpwssds256:
6770 case X86::BI__builtin_ia32_vpdpwssds512:
6771 case X86::BI__builtin_ia32_vpdpbusds128:
6772 case X86::BI__builtin_ia32_vpdpbusds256:
6773 case X86::BI__builtin_ia32_vpdpbusds512:
6774 return interp__builtin_ia32_vpdp(S, OpPC, Call, IsSaturating: true);
6775 case X86::BI__builtin_ia32_cvtss2si:
6776 case X86::BI__builtin_ia32_cvtsd2si:
6777 case X86::BI__builtin_ia32_cvttss2si:
6778 case X86::BI__builtin_ia32_cvttsd2si:
6779 case X86::BI__builtin_ia32_cvtss2si64:
6780 case X86::BI__builtin_ia32_cvtsd2si64:
6781 case X86::BI__builtin_ia32_cvttss2si64:
6782 case X86::BI__builtin_ia32_cvttsd2si64:
6783 return interp_builtin_ia32_cvt_scalar_to_int(S, OpPC, E: Call);
6784 case X86::BI__builtin_ia32_cvtpd2dq:
6785 case X86::BI__builtin_ia32_cvttpd2dq:
6786 case X86::BI__builtin_ia32_cvtps2dq:
6787 case X86::BI__builtin_ia32_cvtpd2dq256:
6788 case X86::BI__builtin_ia32_cvtps2dq256:
6789 case X86::BI__builtin_ia32_cvttps2dq:
6790 case X86::BI__builtin_ia32_cvttpd2dq256:
6791 case X86::BI__builtin_ia32_cvttps2dq256:
6792 return interp_builtin_ia32_cvt_vector_to_int(S, OpPC, E: Call);
6793 default:
6794 return Invalid(S, OpPC);
6795 }
6796
6797 llvm_unreachable("Unhandled builtin ID");
6798}
6799
6800bool InterpretOffsetOf(InterpState &S, CodePtr OpPC, const OffsetOfExpr *E,
6801 ArrayRef<int64_t> ArrayIndices, int64_t &IntResult) {
6802 S.getASTContext().recordOffsetOfEvaluation(E);
6803 CharUnits Result;
6804 unsigned N = E->getNumComponents();
6805 assert(N > 0);
6806
6807 unsigned ArrayIndex = 0;
6808 QualType CurrentType = E->getTypeSourceInfo()->getType();
6809 for (unsigned I = 0; I != N; ++I) {
6810 const OffsetOfNode &Node = E->getComponent(Idx: I);
6811 switch (Node.getKind()) {
6812 case OffsetOfNode::Field: {
6813 const FieldDecl *MemberDecl = Node.getField();
6814 const auto *RD = CurrentType->getAsRecordDecl();
6815 if (!RD || RD->isInvalidDecl())
6816 return false;
6817 const ASTRecordLayout &RL = S.getASTContext().getASTRecordLayout(D: RD);
6818 unsigned FieldIndex = MemberDecl->getFieldIndex();
6819 assert(FieldIndex < RL.getFieldCount() && "offsetof field in wrong type");
6820 Result +=
6821 S.getASTContext().toCharUnitsFromBits(BitSize: RL.getFieldOffset(FieldNo: FieldIndex));
6822 CurrentType = MemberDecl->getType().getNonReferenceType();
6823 break;
6824 }
6825 case OffsetOfNode::Array: {
6826 // When generating bytecode, we put all the index expressions as Sint64 on
6827 // the stack.
6828 int64_t Index = ArrayIndices[ArrayIndex];
6829 if (Index < 0)
6830 return Invalid(S, OpPC);
6831 const ArrayType *AT = S.getASTContext().getAsArrayType(T: CurrentType);
6832 if (!AT)
6833 return false;
6834 CurrentType = AT->getElementType();
6835 CharUnits ElementSize = S.getASTContext().getTypeSizeInChars(T: CurrentType);
6836 int64_t ElemSize = ElementSize.getQuantity();
6837 if (Index != 0 && ElemSize > (llvm::maxIntN(N: 64) / Index)) {
6838 S.FFDiag(Loc: S.Current->getLocation(PC: OpPC),
6839 DiagId: diag::note_constexpr_offsetof_overflow)
6840 << S.Current->getRange(PC: OpPC);
6841 return false;
6842 }
6843 int64_t Offset = Index * ElemSize;
6844 if (Result.getQuantity() > llvm::maxIntN(N: 64) - Offset) {
6845 S.FFDiag(Loc: S.Current->getLocation(PC: OpPC),
6846 DiagId: diag::note_constexpr_offsetof_overflow)
6847 << S.Current->getRange(PC: OpPC);
6848 return false;
6849 }
6850 Result += CharUnits::fromQuantity(Quantity: Offset);
6851 ++ArrayIndex;
6852 break;
6853 }
6854 case OffsetOfNode::Base: {
6855 const CXXBaseSpecifier *BaseSpec = Node.getBase();
6856 if (BaseSpec->isVirtual())
6857 return false;
6858
6859 // Find the layout of the class whose base we are looking into.
6860 const auto *RD = CurrentType->getAsCXXRecordDecl();
6861 if (!RD || RD->isInvalidDecl())
6862 return false;
6863 const ASTRecordLayout &RL = S.getASTContext().getASTRecordLayout(D: RD);
6864
6865 // Find the base class itself.
6866 CurrentType = BaseSpec->getType();
6867 const auto *BaseRD = CurrentType->getAsCXXRecordDecl();
6868 if (!BaseRD)
6869 return false;
6870
6871 // Add the offset to the base.
6872 Result += RL.getBaseClassOffset(Base: BaseRD);
6873 break;
6874 }
6875 case OffsetOfNode::Identifier:
6876 llvm_unreachable("Dependent OffsetOfExpr?");
6877 }
6878 }
6879
6880 IntResult = Result.getQuantity();
6881
6882 return true;
6883}
6884
6885bool SetThreeWayComparisonField(InterpState &S, CodePtr OpPC,
6886 const Pointer &Ptr, const APSInt &IntValue) {
6887
6888 const Record *R = Ptr.getRecord();
6889 assert(R);
6890 assert(R->getNumFields() == 1);
6891
6892 unsigned FieldOffset = R->getField(I: 0u)->Offset;
6893 PtrView FieldPtr = Ptr.view().atField(Offset: FieldOffset);
6894 PrimType FieldT = FieldPtr.getFieldDesc()->getPrimType();
6895
6896 INT_TYPE_SWITCH(FieldT,
6897 FieldPtr.deref<T>() = T::from(IntValue.getSExtValue()));
6898 FieldPtr.initialize();
6899 return true;
6900}
6901
6902static void zeroAll(PtrView Dest) {
6903 const Descriptor *Desc = Dest.getFieldDesc();
6904
6905 if (Desc->isPrimitive()) {
6906 TYPE_SWITCH(Desc->getPrimType(), {
6907 Dest.deref<T>().~T();
6908 new (&Dest.deref<T>()) T();
6909 });
6910 return;
6911 }
6912
6913 if (Desc->isRecord()) {
6914 const Record *R = Desc->ElemRecord;
6915 for (const Record::Field &F : R->fields()) {
6916 PtrView FieldPtr = Dest.atField(Offset: F.Offset);
6917 zeroAll(Dest: FieldPtr);
6918 }
6919 return;
6920 }
6921
6922 if (Desc->isPrimitiveArray()) {
6923 for (unsigned I = 0, N = Desc->getNumElems(); I != N; ++I) {
6924 TYPE_SWITCH(Desc->getPrimType(), {
6925 Dest.deref<T>().~T();
6926 new (&Dest.deref<T>()) T();
6927 });
6928 }
6929 return;
6930 }
6931
6932 if (Desc->isCompositeArray()) {
6933 for (unsigned I = 0, N = Desc->getNumElems(); I != N; ++I) {
6934 PtrView ElemPtr = Dest.atIndex(Idx: I).narrow();
6935 zeroAll(Dest: ElemPtr);
6936 }
6937 return;
6938 }
6939}
6940
6941static bool copyComposite(InterpState &S, CodePtr OpPC, PtrView Src,
6942 PtrView Dest, bool Activate, bool Diagnose);
6943static bool copyRecord(InterpState &S, CodePtr OpPC, PtrView Src, PtrView Dest,
6944 bool Activate = false, bool Diagnose = true) {
6945 [[maybe_unused]] const Descriptor *SrcDesc = Src.getFieldDesc();
6946 const Descriptor *DestDesc = Dest.getFieldDesc();
6947
6948 auto copyField = [&](const Record::Field &F, bool Activate) -> bool {
6949 PtrView DestField = Dest.atField(Offset: F.Offset);
6950 PtrView SrcField = Src.atField(Offset: F.Offset);
6951
6952 if (OptPrimType FT = F.T) {
6953 if (!SrcField.isInitialized()) {
6954 if (Diagnose)
6955 return diagnoseUninitialized(S, OpPC, Extern: false, B: SrcField.block(),
6956 LT: SrcField.getLifetime(), AK: AK_Read);
6957 // Just skip.
6958 return true;
6959 }
6960
6961 TYPE_SWITCH(*FT, DestField.deref<T>() = SrcField.deref<T>(););
6962 if (DestField.canBeInitialized())
6963 DestField.initialize();
6964 if (Activate)
6965 DestField.activate();
6966 return true;
6967 }
6968
6969 return copyComposite(S, OpPC, Src: SrcField, Dest: DestField, Activate, Diagnose);
6970 };
6971
6972 assert(SrcDesc->isRecord());
6973 assert(SrcDesc->ElemRecord == DestDesc->ElemRecord);
6974 const Record *R = DestDesc->ElemRecord;
6975 for (const Record::Field &F : R->fields()) {
6976 PtrView FP = Src.atField(Offset: F.Offset);
6977
6978 if (!CheckMutable(S, OpPC, Ptr: FP))
6979 return false;
6980
6981 if (R->isUnion()) {
6982 // For unions, only copy the active field. Zero all others.
6983 if (FP.isActive()) {
6984 if (!copyField(F, /*Activate=*/true))
6985 return false;
6986 } else {
6987 PtrView DestField = Dest.atField(Offset: F.Offset);
6988 zeroAll(Dest: DestField);
6989 }
6990 } else {
6991 if (!copyField(F, Activate))
6992 return false;
6993 }
6994 }
6995
6996 for (const Record::Base &B : R->bases()) {
6997 PtrView DestBase = Dest.atField(Offset: B.Offset);
6998 if (!copyRecord(S, OpPC, Src: Src.atField(Offset: B.Offset), Dest: DestBase, Activate,
6999 Diagnose))
7000 return false;
7001 }
7002
7003 Dest.initialize();
7004 if (Activate)
7005 Dest.activate();
7006 return true;
7007}
7008
7009static bool copyComposite(InterpState &S, CodePtr OpPC, PtrView Src,
7010 PtrView Dest, bool Activate = false,
7011 bool Diagnose = false) {
7012 assert(Src.isLive() && Dest.isLive());
7013
7014 [[maybe_unused]] const Descriptor *SrcDesc = Src.getFieldDesc();
7015 const Descriptor *DestDesc = Dest.getFieldDesc();
7016
7017 assert(!DestDesc->isPrimitive() && !SrcDesc->isPrimitive());
7018
7019 if (DestDesc->isPrimitiveArray()) {
7020 if (!SrcDesc->isPrimitiveArray())
7021 return false;
7022 // For floating types, check the actual QualType so we don't accidentally
7023 // mix up semantics.
7024 if (SrcDesc->getPrimType() == PT_Float) {
7025 if (!S.getASTContext().hasSimilarType(T1: SrcDesc->getElemQualType(),
7026 T2: DestDesc->getElemQualType()))
7027 return false;
7028 }
7029
7030 assert(SrcDesc->isPrimitiveArray());
7031 assert(SrcDesc->getNumElems() == DestDesc->getNumElems());
7032 assert(SrcDesc->getPrimType() == DestDesc->getPrimType());
7033 PrimType ET = DestDesc->getPrimType();
7034 for (unsigned I = 0, N = DestDesc->getNumElems(); I != N; ++I) {
7035 PtrView DestElem = Dest.atIndex(Idx: I);
7036 TYPE_SWITCH(ET, { DestElem.deref<T>() = Src.elem<T>(I); });
7037 DestElem.initializeElement(Index: I);
7038 }
7039 return true;
7040 }
7041
7042 if (DestDesc->isCompositeArray()) {
7043 if (!SrcDesc->isCompositeArray())
7044 return false;
7045 assert(SrcDesc->isCompositeArray());
7046 assert(SrcDesc->getNumElems() == DestDesc->getNumElems());
7047 for (unsigned I = 0, N = DestDesc->getNumElems(); I != N; ++I) {
7048 PtrView SrcElem = Src.atIndex(Idx: I).narrow();
7049 PtrView DestElem = Dest.atIndex(Idx: I).narrow();
7050 if (!copyComposite(S, OpPC, Src: SrcElem, Dest: DestElem, Activate))
7051 return false;
7052 }
7053 return true;
7054 }
7055
7056 if (DestDesc->isRecord()) {
7057 if (!SrcDesc->isRecord())
7058 return false;
7059 return copyRecord(S, OpPC, Src, Dest, Activate, Diagnose);
7060 }
7061 return Invalid(S, OpPC);
7062}
7063
7064bool DoMemcpy(InterpState &S, CodePtr OpPC, const Pointer &Src, Pointer &Dest,
7065 bool Activate, bool Diagnose) {
7066 if (!Src.isBlockPointer() || Src.getFieldDesc()->isPrimitive())
7067 return false;
7068 if (!Dest.isBlockPointer() || Dest.getFieldDesc()->isPrimitive())
7069 return false;
7070
7071 return copyComposite(S, OpPC, Src: Src.view(), Dest: Dest.view(), Activate, Diagnose);
7072}
7073
7074} // namespace interp
7075} // namespace clang
7076