1//===-- CodeGenTBAA.cpp - TBAA information for LLVM CodeGen ---------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This is the code that manages TBAA information and defines the TBAA policy
10// for the optimizer to use. Relevant standards text includes:
11//
12// C99 6.5p7
13// C++ [basic.lval] (p10 in n3126, p15 in some earlier versions)
14//
15//===----------------------------------------------------------------------===//
16
17#include "CodeGenTBAA.h"
18#include "ABIInfoImpl.h"
19#include "CGCXXABI.h"
20#include "CGRecordLayout.h"
21#include "CodeGenTypes.h"
22#include "clang/AST/ASTContext.h"
23#include "clang/AST/Attr.h"
24#include "clang/AST/Mangle.h"
25#include "clang/AST/RecordLayout.h"
26#include "clang/Basic/CodeGenOptions.h"
27#include "clang/Basic/TargetInfo.h"
28#include "clang/CodeGenUtils/RecordLayoutUtils.h"
29#include "llvm/IR/LLVMContext.h"
30#include "llvm/IR/Metadata.h"
31#include "llvm/IR/Module.h"
32#include "llvm/IR/Type.h"
33#include "llvm/Support/Debug.h"
34using namespace clang;
35using namespace CodeGen;
36
37CodeGenTBAA::CodeGenTBAA(ASTContext &Ctx, CodeGenTypes &CGTypes,
38 llvm::Module &M, const CodeGenOptions &CGO,
39 const LangOptions &Features)
40 : Context(Ctx), CGTypes(CGTypes), Module(M), CodeGenOpts(CGO),
41 Features(Features),
42 MangleCtx(ItaniumMangleContext::create(Context&: Ctx, Diags&: Ctx.getDiagnostics())),
43 MDHelper(M.getContext()), Root(nullptr), Char(nullptr) {}
44
45CodeGenTBAA::~CodeGenTBAA() {
46}
47
48llvm::MDNode *CodeGenTBAA::getRoot() {
49 // Define the root of the tree. This identifies the tree, so that
50 // if our LLVM IR is linked with LLVM IR from a different front-end
51 // (or a different version of this front-end), their TBAA trees will
52 // remain distinct, and the optimizer will treat them conservatively.
53 if (!Root) {
54 if (Features.CPlusPlus)
55 Root = MDHelper.createTBAARoot(Name: "Simple C++ TBAA");
56 else
57 Root = MDHelper.createTBAARoot(Name: "Simple C/C++ TBAA");
58 }
59
60 return Root;
61}
62
63llvm::MDNode *CodeGenTBAA::createScalarTypeNode(StringRef Name,
64 llvm::MDNode *Parent,
65 uint64_t Size) {
66 if (CodeGenOpts.NewStructPathTBAA) {
67 llvm::Metadata *Id = MDHelper.createString(Str: Name);
68 return MDHelper.createTBAATypeNode(Parent, Size, Id);
69 }
70 return MDHelper.createTBAAScalarTypeNode(Name, Parent);
71}
72
73llvm::MDNode *CodeGenTBAA::getChar() {
74 // Define the root of the tree for user-accessible memory. C and C++
75 // give special powers to char and certain similar types. However,
76 // these special powers only cover user-accessible memory, and doesn't
77 // include things like vtables.
78 if (!Char)
79 Char = createScalarTypeNode(Name: "omnipotent char", Parent: getRoot(), /* Size= */ 1);
80
81 return Char;
82}
83
84llvm::MDNode *CodeGenTBAA::getAnyPtr(unsigned PtrDepth) {
85 assert(PtrDepth >= 1 && "Pointer must have some depth");
86
87 // Populate at least PtrDepth elements in AnyPtrs. These are the type nodes
88 // for "any" pointers of increasing pointer depth, and are organized in the
89 // hierarchy: any pointer <- any p2 pointer <- any p3 pointer <- ...
90 //
91 // Note that AnyPtrs[Idx] is actually the node for pointer depth (Idx+1),
92 // since there is no node for pointer depth 0.
93 //
94 // These "any" pointer type nodes are used in pointer TBAA. The type node of
95 // a concrete pointer type has the "any" pointer type node of appropriate
96 // pointer depth as its parent. The "any" pointer type nodes are also used
97 // directly for accesses to void pointers, or to specific pointers that we
98 // conservatively do not distinguish in pointer TBAA (e.g. pointers to
99 // members). Essentially, this establishes that e.g. void** can alias with
100 // any type that can unify with T**, ignoring things like qualifiers. Here, T
101 // is a variable that represents an arbitrary type, including pointer types.
102 // As such, each depth is naturally a subtype of the previous depth, and thus
103 // transitively of all previous depths.
104 if (AnyPtrs.size() < PtrDepth) {
105 AnyPtrs.reserve(N: PtrDepth);
106 auto Size = Module.getDataLayout().getPointerSize();
107 // Populate first element.
108 if (AnyPtrs.empty())
109 AnyPtrs.push_back(Elt: createScalarTypeNode(Name: "any pointer", Parent: getChar(), Size));
110 // Populate further elements.
111 for (size_t Idx = AnyPtrs.size(); Idx < PtrDepth; ++Idx) {
112 auto Name = ("any p" + llvm::Twine(Idx + 1) + " pointer").str();
113 AnyPtrs.push_back(Elt: createScalarTypeNode(Name, Parent: AnyPtrs[Idx - 1], Size));
114 }
115 }
116
117 return AnyPtrs[PtrDepth - 1];
118}
119
120static bool TypeHasMayAlias(QualType QTy) {
121 // Tagged types have declarations, and therefore may have attributes.
122 if (auto *TD = QTy->getAsTagDecl())
123 if (TD->hasAttr<MayAliasAttr>())
124 return true;
125
126 // Also look for may_alias as a declaration attribute on a typedef.
127 // FIXME: We should follow GCC and model may_alias as a type attribute
128 // rather than as a declaration attribute.
129 while (auto *TT = QTy->getAs<TypedefType>()) {
130 if (TT->getDecl()->hasAttr<MayAliasAttr>())
131 return true;
132 QTy = TT->desugar();
133 }
134
135 // Also consider an array type as may_alias when its element type (at
136 // any level) is marked as such.
137 if (auto *ArrayTy = QTy->getAsArrayTypeUnsafe())
138 if (TypeHasMayAlias(QTy: ArrayTy->getElementType()))
139 return true;
140
141 return false;
142}
143
144/// Check if the given type is a valid base type to be used in access tags.
145static bool isValidBaseType(QualType QTy) {
146 if (const auto *RD = QTy->getAsRecordDecl()) {
147 // Incomplete types are not valid base access types.
148 if (!RD->isCompleteDefinition())
149 return false;
150 if (RD->hasFlexibleArrayMember())
151 return false;
152 // RD can be struct, union, class, interface or enum.
153 // For now, we only handle struct and class.
154 if (RD->isStruct() || RD->isClass())
155 return true;
156 }
157 return false;
158}
159
160llvm::MDNode *CodeGenTBAA::getTypeInfoHelper(const Type *Ty) {
161 uint64_t Size = Context.getTypeSizeInChars(T: Ty).getQuantity();
162
163 // Handle builtin types.
164 if (const BuiltinType *BTy = dyn_cast<BuiltinType>(Val: Ty)) {
165 switch (BTy->getKind()) {
166 // Character types are special and can alias anything.
167 // In C++, this technically only includes "char" and "unsigned char",
168 // and not "signed char". In C, it includes all three. For now,
169 // the risk of exploiting this detail in C++ seems likely to outweigh
170 // the benefit.
171 case BuiltinType::Char_U:
172 case BuiltinType::Char_S:
173 case BuiltinType::UChar:
174 case BuiltinType::SChar:
175 return getChar();
176
177 // Unsigned types can alias their corresponding signed types.
178 case BuiltinType::UShort:
179 return getTypeInfo(QTy: Context.ShortTy);
180 case BuiltinType::UInt:
181 return getTypeInfo(QTy: Context.IntTy);
182 case BuiltinType::ULong:
183 return getTypeInfo(QTy: Context.LongTy);
184 case BuiltinType::ULongLong:
185 return getTypeInfo(QTy: Context.LongLongTy);
186 case BuiltinType::UInt128:
187 return getTypeInfo(QTy: Context.Int128Ty);
188
189 case BuiltinType::UShortFract:
190 return getTypeInfo(QTy: Context.ShortFractTy);
191 case BuiltinType::UFract:
192 return getTypeInfo(QTy: Context.FractTy);
193 case BuiltinType::ULongFract:
194 return getTypeInfo(QTy: Context.LongFractTy);
195
196 case BuiltinType::SatUShortFract:
197 return getTypeInfo(QTy: Context.SatShortFractTy);
198 case BuiltinType::SatUFract:
199 return getTypeInfo(QTy: Context.SatFractTy);
200 case BuiltinType::SatULongFract:
201 return getTypeInfo(QTy: Context.SatLongFractTy);
202
203 case BuiltinType::UShortAccum:
204 return getTypeInfo(QTy: Context.ShortAccumTy);
205 case BuiltinType::UAccum:
206 return getTypeInfo(QTy: Context.AccumTy);
207 case BuiltinType::ULongAccum:
208 return getTypeInfo(QTy: Context.LongAccumTy);
209
210 case BuiltinType::SatUShortAccum:
211 return getTypeInfo(QTy: Context.SatShortAccumTy);
212 case BuiltinType::SatUAccum:
213 return getTypeInfo(QTy: Context.SatAccumTy);
214 case BuiltinType::SatULongAccum:
215 return getTypeInfo(QTy: Context.SatLongAccumTy);
216
217 // Treat all other builtin types as distinct types. This includes
218 // treating wchar_t, char16_t, and char32_t as distinct from their
219 // "underlying types".
220 default:
221 return createScalarTypeNode(Name: BTy->getName(Policy: Features), Parent: getChar(), Size);
222 }
223 }
224
225 // C++1z [basic.lval]p10: "If a program attempts to access the stored value of
226 // an object through a glvalue of other than one of the following types the
227 // behavior is undefined: [...] a char, unsigned char, or std::byte type."
228 if (Ty->isStdByteType())
229 return getChar();
230
231 // Handle pointers and references.
232 //
233 // C has a very strict rule for pointer aliasing. C23 6.7.6.1p2:
234 // For two pointer types to be compatible, both shall be identically
235 // qualified and both shall be pointers to compatible types.
236 //
237 // This rule is impractically strict; we want to at least ignore CVR
238 // qualifiers. Distinguishing by CVR qualifiers would make it UB to
239 // e.g. cast a `char **` to `const char * const *` and dereference it,
240 // which is too common and useful to invalidate. C++'s similar types
241 // rule permits qualifier differences in these nested positions; in fact,
242 // C++ even allows that cast as an implicit conversion.
243 //
244 // Other qualifiers could theoretically be distinguished, especially if
245 // they involve a significant representation difference. We don't
246 // currently do so, however.
247 if (Ty->isPointerType() || Ty->isReferenceType()) {
248 if (!CodeGenOpts.PointerTBAA)
249 return getAnyPtr();
250 // C++ [basic.lval]p11 permits objects to accessed through an l-value of
251 // similar type. Two types are similar under C++ [conv.qual]p2 if the
252 // decomposition of the types into pointers, member pointers, and arrays has
253 // the same structure when ignoring cv-qualifiers at each level of the
254 // decomposition. Meanwhile, C makes T(*)[] and T(*)[N] compatible, which
255 // would really complicate any attempt to distinguish pointers to arrays by
256 // their bounds. It's simpler, and much easier to explain to users, to
257 // simply treat all pointers to arrays as pointers to their element type for
258 // aliasing purposes. So when creating a TBAA tag for a pointer type, we
259 // recursively ignore both qualifiers and array types when decomposing the
260 // pointee type. The only meaningful remaining structure is the number of
261 // pointer types we encountered along the way, so we just produce the tag
262 // "p<depth> <base type tag>". If we do find a member pointer type, for now
263 // we just conservatively bail out with AnyPtr (below) rather than trying to
264 // create a tag that honors the similar-type rules while still
265 // distinguishing different kinds of member pointer.
266 unsigned PtrDepth = 0;
267 do {
268 PtrDepth++;
269 Ty = Ty->getPointeeType()->getBaseElementTypeUnsafe();
270 } while (Ty->isPointerType());
271
272 // While there are no special rules in the standards regarding void pointers
273 // and strict aliasing, emitting distinct tags for void pointers break some
274 // common idioms and there is no good alternative to re-write the code
275 // without strict-aliasing violations.
276 if (Ty->isVoidType())
277 return getAnyPtr(PtrDepth);
278
279 assert(!isa<VariableArrayType>(Ty));
280 // When the underlying type is a builtin type, we compute the pointee type
281 // string recursively, which is implicitly more forgiving than the standards
282 // require. Effectively, we are turning the question "are these types
283 // compatible/similar" into "are accesses to these types allowed to alias".
284 // In both C and C++, the latter question has special carve-outs for
285 // signedness mismatches that only apply at the top level. As a result, we
286 // are allowing e.g. `int *` l-values to access `unsigned *` objects.
287 SmallString<256> TyName;
288 if (isa<BuiltinType>(Val: Ty)) {
289 llvm::MDNode *ScalarMD = getTypeInfoHelper(Ty);
290 StringRef Name =
291 cast<llvm::MDString>(
292 Val: ScalarMD->getOperand(I: CodeGenOpts.NewStructPathTBAA ? 2 : 0))
293 ->getString();
294 TyName = Name;
295 } else {
296 // Be conservative if the type isn't a RecordType. We are specifically
297 // required to do this for member pointers until we implement the
298 // similar-types rule.
299 const auto *RT = Ty->getAsCanonical<RecordType>();
300 if (!RT)
301 return getAnyPtr(PtrDepth);
302
303 // For unnamed structs or unions C's compatible types rule applies. Two
304 // compatible types in different compilation units can have different
305 // mangled names, meaning the metadata emitted below would incorrectly
306 // mark them as no-alias. Use AnyPtr for such types in both C and C++, as
307 // C and C++ types may be visible when doing LTO.
308 //
309 // Note that using AnyPtr is overly conservative. We could summarize the
310 // members of the type, as per the C compatibility rule in the future.
311 // This also covers anonymous structs and unions, which have a different
312 // compatibility rule, but it doesn't matter because you can never have a
313 // pointer to an anonymous struct or union.
314 if (!RT->getDecl()->getDeclName())
315 return getAnyPtr(PtrDepth);
316
317 // For non-builtin types use the mangled name of the canonical type.
318 llvm::raw_svector_ostream TyOut(TyName);
319 MangleCtx->mangleCanonicalTypeName(T: QualType(Ty, 0), TyOut);
320 }
321
322 SmallString<256> OutName("p");
323 OutName += std::to_string(val: PtrDepth);
324 OutName += " ";
325 OutName += TyName;
326 return createScalarTypeNode(Name: OutName, Parent: getAnyPtr(PtrDepth), Size);
327 }
328
329 // Accesses to arrays are accesses to objects of their element types.
330 if (CodeGenOpts.NewStructPathTBAA && Ty->isArrayType())
331 return getTypeInfo(QTy: cast<ArrayType>(Val: Ty)->getElementType());
332
333 // Accesses to matrix types are accesses to objects of their element types.
334 if (const auto *MTy = dyn_cast<MatrixType>(Val: Ty)) {
335 assert(isa<ConstantMatrixType>(Ty) &&
336 "only ConstantMatrixType should reach CodeGen");
337 return getTypeInfo(QTy: MTy->getElementType());
338 }
339
340 // Enum types are distinct types. In C++ they have "underlying types",
341 // however they aren't related for TBAA.
342 if (const EnumType *ETy = dyn_cast<EnumType>(Val: Ty)) {
343 const EnumDecl *ED = ETy->getDecl()->getDefinitionOrSelf();
344 if (!Features.CPlusPlus)
345 return getTypeInfo(QTy: ED->getIntegerType());
346
347 // In C++ mode, types have linkage, so we can rely on the ODR and
348 // on their mangled names, if they're external.
349 // TODO: Is there a way to get a program-wide unique name for a
350 // decl with local linkage or no linkage?
351 if (!ED->isExternallyVisible())
352 return getChar();
353
354 SmallString<256> OutName;
355 llvm::raw_svector_ostream Out(OutName);
356 CGTypes.getCXXABI().getMangleContext().mangleCanonicalTypeName(
357 T: QualType(ETy, 0), Out);
358 return createScalarTypeNode(Name: OutName, Parent: getChar(), Size);
359 }
360
361 if (const auto *EIT = dyn_cast<BitIntType>(Val: Ty)) {
362 SmallString<256> OutName;
363 llvm::raw_svector_ostream Out(OutName);
364 // Don't specify signed/unsigned since integer types can alias despite sign
365 // differences.
366 Out << "_BitInt(" << EIT->getNumBits() << ')';
367 return createScalarTypeNode(Name: OutName, Parent: getChar(), Size);
368 }
369
370 // For now, handle any other kind of type conservatively.
371 return getChar();
372}
373
374llvm::MDNode *CodeGenTBAA::getTypeInfo(QualType QTy) {
375 // At -O0 or relaxed aliasing, TBAA is not emitted for regular types (unless
376 // we're running TypeSanitizer).
377 if (!Features.Sanitize.has(K: SanitizerKind::Type) &&
378 (CodeGenOpts.OptimizationLevel == 0 || CodeGenOpts.RelaxedAliasing))
379 return nullptr;
380
381 // If the type has the may_alias attribute (even on a typedef), it is
382 // effectively in the general char alias class.
383 if (TypeHasMayAlias(QTy))
384 return getChar();
385
386 // We need this function to not fall back to returning the "omnipotent char"
387 // type node for aggregate and union types. Otherwise, any dereference of an
388 // aggregate will result into the may-alias access descriptor, meaning all
389 // subsequent accesses to direct and indirect members of that aggregate will
390 // be considered may-alias too.
391 // TODO: Combine getTypeInfo() and getValidBaseTypeInfo() into a single
392 // function.
393 if (isValidBaseType(QTy))
394 return getValidBaseTypeInfo(QTy);
395
396 const Type *Ty = Context.getCanonicalType(T: QTy).getTypePtr();
397 if (llvm::MDNode *N = MetadataCache[Ty])
398 return N;
399
400 // Note that the following helper call is allowed to add new nodes to the
401 // cache, which invalidates all its previously obtained iterators. So we
402 // first generate the node for the type and then add that node to the cache.
403 llvm::MDNode *TypeNode = getTypeInfoHelper(Ty);
404 return MetadataCache[Ty] = TypeNode;
405}
406
407TBAAAccessInfo CodeGenTBAA::getAccessInfo(QualType AccessType) {
408 // Pointee values may have incomplete types, but they shall never be
409 // dereferenced.
410 if (AccessType->isIncompleteType())
411 return TBAAAccessInfo::getIncompleteInfo();
412
413 if (TypeHasMayAlias(QTy: AccessType))
414 return TBAAAccessInfo::getMayAliasInfo();
415
416 uint64_t Size = Context.getTypeSizeInChars(T: AccessType).getQuantity();
417 return TBAAAccessInfo(getTypeInfo(QTy: AccessType), Size);
418}
419
420TBAAAccessInfo CodeGenTBAA::getVTablePtrAccessInfo(llvm::Type *VTablePtrType) {
421 const llvm::DataLayout &DL = Module.getDataLayout();
422 unsigned Size = DL.getPointerTypeSize(Ty: VTablePtrType);
423 return TBAAAccessInfo(createScalarTypeNode(Name: "vtable pointer", Parent: getRoot(), Size),
424 Size);
425}
426
427bool
428CodeGenTBAA::CollectFields(uint64_t BaseOffset,
429 QualType QTy,
430 SmallVectorImpl<llvm::MDBuilder::TBAAStructField> &
431 Fields,
432 bool MayAlias) {
433 /* Things not handled yet include: C++ base classes, bitfields, */
434
435 if (const auto *TTy = QTy->getAsCanonical<RecordType>()) {
436 if (TTy->isUnionType()) {
437 uint64_t Size = Context.getTypeSizeInChars(T: QTy).getQuantity();
438 llvm::MDNode *TBAAType = getChar();
439 llvm::MDNode *TBAATag = getAccessTagInfo(Info: TBAAAccessInfo(TBAAType, Size));
440 Fields.push_back(
441 Elt: llvm::MDBuilder::TBAAStructField(BaseOffset, Size, TBAATag));
442 return true;
443 }
444 const RecordDecl *RD = TTy->getDecl()->getDefinition();
445 if (RD->hasFlexibleArrayMember())
446 return false;
447
448 // TODO: Handle C++ base classes.
449 if (const CXXRecordDecl *Decl = dyn_cast<CXXRecordDecl>(Val: RD))
450 if (!Decl->bases().empty())
451 return false;
452
453 const ASTRecordLayout &Layout = Context.getASTRecordLayout(D: RD);
454 const CGRecordLayout &CGRL = CGTypes.getCGRecordLayout(RD);
455
456 unsigned idx = 0;
457 for (RecordDecl::field_iterator i = RD->field_begin(), e = RD->field_end();
458 i != e; ++i, ++idx) {
459 if (CodeGenUtils::isEmptyFieldForLayout(Ctx: Context, FD: *i))
460 continue;
461
462 uint64_t Offset =
463 BaseOffset + Layout.getFieldOffset(FieldNo: idx) / Context.getCharWidth();
464
465 // Create a single field for consecutive named bitfields using char as
466 // base type.
467 if ((*i)->isBitField()) {
468 const CGBitFieldInfo &Info = CGRL.getBitFieldInfo(FD: *i);
469 // For big endian targets the first bitfield in the consecutive run is
470 // at the most-significant end; see CGRecordLowering::setBitFieldInfo
471 // for more information.
472 bool IsBE = Context.getTargetInfo().isBigEndian();
473 bool IsFirst = IsBE ? Info.StorageSize - (Info.Offset + Info.Size) == 0
474 : Info.Offset == 0;
475 if (!IsFirst)
476 continue;
477 unsigned CurrentBitFieldSize = Info.StorageSize;
478 uint64_t Size =
479 llvm::divideCeil(Numerator: CurrentBitFieldSize, Denominator: Context.getCharWidth());
480 llvm::MDNode *TBAAType = getChar();
481 llvm::MDNode *TBAATag =
482 getAccessTagInfo(Info: TBAAAccessInfo(TBAAType, Size));
483 Fields.push_back(
484 Elt: llvm::MDBuilder::TBAAStructField(Offset, Size, TBAATag));
485 continue;
486 }
487
488 QualType FieldQTy = i->getType();
489 if (!CollectFields(BaseOffset: Offset, QTy: FieldQTy, Fields,
490 MayAlias: MayAlias || TypeHasMayAlias(QTy: FieldQTy)))
491 return false;
492 }
493 return true;
494 }
495
496 /* Otherwise, treat whatever it is as a field. */
497 uint64_t Offset = BaseOffset;
498 uint64_t Size = Context.getTypeSizeInChars(T: QTy).getQuantity();
499 llvm::MDNode *TBAAType = MayAlias ? getChar() : getTypeInfo(QTy);
500 llvm::MDNode *TBAATag = getAccessTagInfo(Info: TBAAAccessInfo(TBAAType, Size));
501 Fields.push_back(Elt: llvm::MDBuilder::TBAAStructField(Offset, Size, TBAATag));
502 return true;
503}
504
505llvm::MDNode *
506CodeGenTBAA::getTBAAStructInfo(QualType QTy) {
507 if (CodeGenOpts.OptimizationLevel == 0 || CodeGenOpts.RelaxedAliasing)
508 return nullptr;
509
510 const Type *Ty = Context.getCanonicalType(T: QTy).getTypePtr();
511
512 if (llvm::MDNode *N = StructMetadataCache[Ty])
513 return N;
514
515 SmallVector<llvm::MDBuilder::TBAAStructField, 4> Fields;
516 if (CollectFields(BaseOffset: 0, QTy, Fields, MayAlias: TypeHasMayAlias(QTy)))
517 return MDHelper.createTBAAStructNode(Fields);
518
519 // For now, handle any other kind of type conservatively.
520 return StructMetadataCache[Ty] = nullptr;
521}
522
523llvm::MDNode *CodeGenTBAA::getBaseTypeInfoHelper(const Type *Ty) {
524 if (auto *TTy = dyn_cast<RecordType>(Val: Ty)) {
525 const RecordDecl *RD = TTy->getDecl()->getDefinition();
526 const ASTRecordLayout &Layout = Context.getASTRecordLayout(D: RD);
527 using TBAAStructField = llvm::MDBuilder::TBAAStructField;
528 SmallVector<TBAAStructField, 4> Fields;
529 if (const CXXRecordDecl *CXXRD = dyn_cast<CXXRecordDecl>(Val: RD)) {
530 // Handle C++ base classes. Non-virtual bases can treated a kind of
531 // field. Virtual bases are more complex and omitted, but avoid an
532 // incomplete view for NewStructPathTBAA.
533 if (CodeGenOpts.NewStructPathTBAA && CXXRD->getNumVBases() != 0)
534 return nullptr;
535 for (const CXXBaseSpecifier &B : CXXRD->bases()) {
536 if (B.isVirtual())
537 continue;
538 QualType BaseQTy = B.getType();
539 const CXXRecordDecl *BaseRD = BaseQTy->getAsCXXRecordDecl();
540 if (BaseRD->isEmpty())
541 continue;
542 llvm::MDNode *TypeNode = isValidBaseType(QTy: BaseQTy)
543 ? getValidBaseTypeInfo(QTy: BaseQTy)
544 : getTypeInfo(QTy: BaseQTy);
545 if (!TypeNode)
546 return nullptr;
547 uint64_t Offset = Layout.getBaseClassOffset(Base: BaseRD).getQuantity();
548 uint64_t Size =
549 Context.getASTRecordLayout(D: BaseRD).getDataSize().getQuantity();
550 Fields.push_back(
551 Elt: llvm::MDBuilder::TBAAStructField(Offset, Size, TypeNode));
552 }
553 // The order in which base class subobjects are allocated is unspecified,
554 // so may differ from declaration order. In particular, Itanium ABI will
555 // allocate a primary base first.
556 // Since we exclude empty subobjects, the objects are not overlapping and
557 // their offsets are unique.
558 llvm::sort(C&: Fields,
559 Comp: [](const TBAAStructField &A, const TBAAStructField &B) {
560 return A.Offset < B.Offset;
561 });
562 }
563 for (FieldDecl *Field : RD->fields()) {
564 if (Field->isZeroSize(Ctx: Context) || Field->isUnnamedBitField())
565 continue;
566 QualType FieldQTy = Field->getType();
567 llvm::MDNode *TypeNode = isValidBaseType(QTy: FieldQTy)
568 ? getValidBaseTypeInfo(QTy: FieldQTy)
569 : getTypeInfo(QTy: FieldQTy);
570 if (!TypeNode)
571 return nullptr;
572
573 uint64_t BitOffset = Layout.getFieldOffset(FieldNo: Field->getFieldIndex());
574 uint64_t Offset = Context.toCharUnitsFromBits(BitSize: BitOffset).getQuantity();
575 uint64_t Size = Context.getTypeSizeInChars(T: FieldQTy).getQuantity();
576 Fields.push_back(Elt: llvm::MDBuilder::TBAAStructField(Offset, Size,
577 TypeNode));
578 }
579
580 SmallString<256> OutName;
581 if (Features.CPlusPlus) {
582 // Don't use the mangler for C code.
583 llvm::raw_svector_ostream Out(OutName);
584 CGTypes.getCXXABI().getMangleContext().mangleCanonicalTypeName(
585 T: QualType(Ty, 0), Out);
586 } else {
587 OutName = RD->getName();
588 }
589
590 if (CodeGenOpts.NewStructPathTBAA) {
591 llvm::MDNode *Parent = getChar();
592 uint64_t Size = Context.getTypeSizeInChars(T: Ty).getQuantity();
593 llvm::Metadata *Id = MDHelper.createString(Str: OutName);
594 return MDHelper.createTBAATypeNode(Parent, Size, Id, Fields);
595 }
596
597 // Create the struct type node with a vector of pairs (offset, type).
598 SmallVector<std::pair<llvm::MDNode*, uint64_t>, 4> OffsetsAndTypes;
599 for (const auto &Field : Fields)
600 OffsetsAndTypes.push_back(Elt: std::make_pair(x: Field.Type, y: Field.Offset));
601 return MDHelper.createTBAAStructTypeNode(Name: OutName, Fields: OffsetsAndTypes);
602 }
603
604 return nullptr;
605}
606
607llvm::MDNode *CodeGenTBAA::getValidBaseTypeInfo(QualType QTy) {
608 assert(isValidBaseType(QTy) && "Must be a valid base type");
609
610 const Type *Ty = Context.getCanonicalType(T: QTy).getTypePtr();
611
612 // nullptr is a valid value in the cache, so use find rather than []
613 auto I = BaseTypeMetadataCache.find(Val: Ty);
614 if (I != BaseTypeMetadataCache.end())
615 return I->second;
616
617 // First calculate the metadata, before recomputing the insertion point, as
618 // the helper can recursively call us.
619 llvm::MDNode *TypeNode = getBaseTypeInfoHelper(Ty);
620 [[maybe_unused]] auto inserted = BaseTypeMetadataCache.insert(KV: {Ty, TypeNode});
621 assert(inserted.second && "BaseType metadata was already inserted");
622
623 return TypeNode;
624}
625
626llvm::MDNode *CodeGenTBAA::getBaseTypeInfo(QualType QTy) {
627 return isValidBaseType(QTy) ? getValidBaseTypeInfo(QTy) : nullptr;
628}
629
630llvm::MDNode *CodeGenTBAA::getAccessTagInfo(TBAAAccessInfo Info) {
631 assert(!Info.isIncomplete() && "Access to an object of an incomplete type!");
632
633 if (Info.isMayAlias())
634 Info = TBAAAccessInfo(getChar(), Info.Size);
635
636 if (!Info.AccessType)
637 return nullptr;
638
639 if (!CodeGenOpts.StructPathTBAA)
640 Info = TBAAAccessInfo(Info.AccessType, Info.Size);
641
642 llvm::MDNode *&N = AccessTagMetadataCache[Info];
643 if (N)
644 return N;
645
646 if (!Info.BaseType) {
647 Info.BaseType = Info.AccessType;
648 assert(!Info.Offset && "Nonzero offset for an access with no base type!");
649 }
650 if (CodeGenOpts.NewStructPathTBAA) {
651 return N = MDHelper.createTBAAAccessTag(BaseType: Info.BaseType, AccessType: Info.AccessType,
652 Offset: Info.Offset, Size: Info.Size);
653 }
654 return N = MDHelper.createTBAAStructTagNode(BaseType: Info.BaseType, AccessType: Info.AccessType,
655 Offset: Info.Offset);
656}
657
658TBAAAccessInfo CodeGenTBAA::mergeTBAAInfoForCast(TBAAAccessInfo SourceInfo,
659 TBAAAccessInfo TargetInfo) {
660 if (SourceInfo.isMayAlias() || TargetInfo.isMayAlias())
661 return TBAAAccessInfo::getMayAliasInfo();
662 return TargetInfo;
663}
664
665TBAAAccessInfo
666CodeGenTBAA::mergeTBAAInfoForConditionalOperator(TBAAAccessInfo InfoA,
667 TBAAAccessInfo InfoB) {
668 if (InfoA == InfoB)
669 return InfoA;
670
671 if (!InfoA || !InfoB)
672 return TBAAAccessInfo();
673
674 if (InfoA.isMayAlias() || InfoB.isMayAlias())
675 return TBAAAccessInfo::getMayAliasInfo();
676
677 // TODO: Implement the rest of the logic here. For example, two accesses
678 // with same final access types result in an access to an object of that final
679 // access type regardless of their base types.
680 return TBAAAccessInfo::getMayAliasInfo();
681}
682
683TBAAAccessInfo
684CodeGenTBAA::mergeTBAAInfoForMemoryTransfer(TBAAAccessInfo DestInfo,
685 TBAAAccessInfo SrcInfo) {
686 if (DestInfo == SrcInfo)
687 return DestInfo;
688
689 if (!DestInfo || !SrcInfo)
690 return TBAAAccessInfo();
691
692 if (DestInfo.isMayAlias() || SrcInfo.isMayAlias())
693 return TBAAAccessInfo::getMayAliasInfo();
694
695 // TODO: Implement the rest of the logic here. For example, two accesses
696 // with same final access types result in an access to an object of that final
697 // access type regardless of their base types.
698 return TBAAAccessInfo::getMayAliasInfo();
699}
700