1#include "llvm/ProfileData/MemProf.h"
2#include "llvm/ADT/SmallVector.h"
3#include "llvm/IR/Function.h"
4#include "llvm/ProfileData/InstrProf.h"
5#include "llvm/ProfileData/SampleProf.h"
6#include "llvm/Support/Endian.h"
7#include "llvm/Support/EndianStream.h"
8
9namespace llvm {
10namespace memprof {
11MemProfSchema getFullSchema() {
12 MemProfSchema List;
13#define MIBEntryDef(NameTag, Name, Type) List.push_back(Meta::Name);
14#include "llvm/ProfileData/MIBEntryDef.inc"
15#undef MIBEntryDef
16 return List;
17}
18
19MemProfSchema getHotColdSchema() {
20 return {Meta::AllocCount, Meta::TotalSize, Meta::TotalLifetime,
21 Meta::TotalLifetimeAccessDensity};
22}
23
24static size_t serializedSizeV3(const IndexedAllocationInfo &IAI,
25 const MemProfSchema &Schema) {
26 size_t Size = 0;
27 // The linear call stack ID.
28 Size += sizeof(LinearCallStackId);
29 // The size of the payload.
30 Size += PortableMemInfoBlock::serializedSize(Schema);
31 return Size;
32}
33
34size_t IndexedAllocationInfo::serializedSize(const MemProfSchema &Schema,
35 IndexedVersion Version) const {
36 switch (Version) {
37 // Combine V3 and V4 as the size calculation is the same
38 case Version3:
39 case Version4:
40 return serializedSizeV3(IAI: *this, Schema);
41 }
42 llvm_unreachable("unsupported MemProf version");
43}
44
45static size_t serializedSizeV3(const IndexedMemProfRecord &Record,
46 const MemProfSchema &Schema) {
47 // The number of alloc sites to serialize.
48 size_t Result = sizeof(uint64_t);
49 for (const IndexedAllocationInfo &N : Record.AllocSites)
50 Result += N.serializedSize(Schema, Version: Version3);
51
52 // The number of callsites we have information for.
53 Result += sizeof(uint64_t);
54 // The linear call stack ID.
55 // Note: V3 only stored the LinearCallStackId per call site.
56 Result += Record.CallSites.size() * sizeof(LinearCallStackId);
57 return Result;
58}
59
60static size_t serializedSizeV4(const IndexedMemProfRecord &Record,
61 const MemProfSchema &Schema) {
62 // The number of alloc sites to serialize.
63 size_t Result = sizeof(uint64_t);
64 for (const IndexedAllocationInfo &N : Record.AllocSites)
65 Result += N.serializedSize(Schema, Version: Version4);
66
67 // The number of callsites we have information for.
68 Result += sizeof(uint64_t);
69 for (const auto &CS : Record.CallSites)
70 Result += sizeof(LinearCallStackId) + sizeof(uint64_t) +
71 CS.CalleeGuids.size() * sizeof(GlobalValue::GUID);
72 return Result;
73}
74
75size_t IndexedMemProfRecord::serializedSize(const MemProfSchema &Schema,
76 IndexedVersion Version) const {
77 switch (Version) {
78 case Version3:
79 return serializedSizeV3(Record: *this, Schema);
80 case Version4:
81 return serializedSizeV4(Record: *this, Schema);
82 }
83 llvm_unreachable("unsupported MemProf version");
84}
85
86static void serializeV3(
87 const IndexedMemProfRecord &Record, const MemProfSchema &Schema,
88 raw_ostream &OS,
89 llvm::DenseMap<CallStackId, LinearCallStackId> &MemProfCallStackIndexes) {
90 using namespace support;
91
92 endian::Writer LE(OS, llvm::endianness::little);
93
94 LE.write<uint64_t>(Val: Record.AllocSites.size());
95 for (const IndexedAllocationInfo &N : Record.AllocSites) {
96 assert(MemProfCallStackIndexes.contains(N.CSId));
97 LE.write<LinearCallStackId>(Val: MemProfCallStackIndexes[N.CSId]);
98 N.Info.serialize(Schema, OS);
99 }
100
101 // Related contexts.
102 LE.write<uint64_t>(Val: Record.CallSites.size());
103 for (const auto &CS : Record.CallSites) {
104 assert(MemProfCallStackIndexes.contains(CS.CSId));
105 LE.write<LinearCallStackId>(Val: MemProfCallStackIndexes[CS.CSId]);
106 }
107}
108
109static void serializeV4(
110 const IndexedMemProfRecord &Record, const MemProfSchema &Schema,
111 raw_ostream &OS,
112 llvm::DenseMap<CallStackId, LinearCallStackId> &MemProfCallStackIndexes) {
113 using namespace support;
114
115 endian::Writer LE(OS, llvm::endianness::little);
116
117 LE.write<uint64_t>(Val: Record.AllocSites.size());
118 for (const IndexedAllocationInfo &N : Record.AllocSites) {
119 assert(MemProfCallStackIndexes.contains(N.CSId));
120 LE.write<LinearCallStackId>(Val: MemProfCallStackIndexes[N.CSId]);
121 N.Info.serialize(Schema, OS);
122 }
123
124 // Related contexts.
125 LE.write<uint64_t>(Val: Record.CallSites.size());
126 for (const auto &CS : Record.CallSites) {
127 assert(MemProfCallStackIndexes.contains(CS.CSId));
128 LE.write<LinearCallStackId>(Val: MemProfCallStackIndexes[CS.CSId]);
129 LE.write<uint64_t>(Val: CS.CalleeGuids.size());
130 for (const auto &Guid : CS.CalleeGuids)
131 LE.write<GlobalValue::GUID>(Val: Guid);
132 }
133}
134
135void IndexedMemProfRecord::serialize(
136 const MemProfSchema &Schema, raw_ostream &OS, IndexedVersion Version,
137 llvm::DenseMap<CallStackId, LinearCallStackId> *MemProfCallStackIndexes)
138 const {
139 switch (Version) {
140 case Version3:
141 serializeV3(Record: *this, Schema, OS, MemProfCallStackIndexes&: *MemProfCallStackIndexes);
142 return;
143 case Version4:
144 serializeV4(Record: *this, Schema, OS, MemProfCallStackIndexes&: *MemProfCallStackIndexes);
145 return;
146 }
147 llvm_unreachable("unsupported MemProf version");
148}
149
150static IndexedMemProfRecord deserializeV3(const MemProfSchema &Schema,
151 const unsigned char *Ptr) {
152 using namespace support;
153
154 IndexedMemProfRecord Record;
155
156 // Read the meminfo nodes.
157 const uint64_t NumNodes =
158 endian::readNext<uint64_t, llvm::endianness::little>(memory&: Ptr);
159 Record.AllocSites.reserve(N: NumNodes);
160 const size_t SerializedSize = PortableMemInfoBlock::serializedSize(Schema);
161 for (uint64_t I = 0; I < NumNodes; I++) {
162 IndexedAllocationInfo Node;
163 Node.CSId =
164 endian::readNext<LinearCallStackId, llvm::endianness::little>(memory&: Ptr);
165 Node.Info.deserialize(IncomingSchema: Schema, Ptr);
166 Ptr += SerializedSize;
167 Record.AllocSites.push_back(Elt: Node);
168 }
169
170 // Read the callsite information.
171 const uint64_t NumCtxs =
172 endian::readNext<uint64_t, llvm::endianness::little>(memory&: Ptr);
173 Record.CallSites.reserve(N: NumCtxs);
174 for (uint64_t J = 0; J < NumCtxs; J++) {
175 // We are storing LinearCallStackId in CallSiteIds, which is a vector of
176 // CallStackId. Assert that CallStackId is no smaller than
177 // LinearCallStackId.
178 static_assert(sizeof(LinearCallStackId) <= sizeof(CallStackId));
179 LinearCallStackId CSId =
180 endian::readNext<LinearCallStackId, llvm::endianness::little>(memory&: Ptr);
181 Record.CallSites.emplace_back(Args&: CSId);
182 }
183
184 return Record;
185}
186
187static IndexedMemProfRecord deserializeV4(const MemProfSchema &Schema,
188 const unsigned char *Ptr) {
189 using namespace support;
190
191 IndexedMemProfRecord Record;
192
193 // Read the meminfo nodes.
194 const uint64_t NumNodes =
195 endian::readNext<uint64_t, llvm::endianness::little>(memory&: Ptr);
196 Record.AllocSites.reserve(N: NumNodes);
197 const size_t SerializedSize = PortableMemInfoBlock::serializedSize(Schema);
198 for (uint64_t I = 0; I < NumNodes; I++) {
199 IndexedAllocationInfo Node;
200 Node.CSId =
201 endian::readNext<LinearCallStackId, llvm::endianness::little>(memory&: Ptr);
202 Node.Info.deserialize(IncomingSchema: Schema, Ptr);
203 Ptr += SerializedSize;
204 Record.AllocSites.push_back(Elt: Node);
205 }
206
207 // Read the callsite information.
208 const uint64_t NumCtxs =
209 endian::readNext<uint64_t, llvm::endianness::little>(memory&: Ptr);
210 Record.CallSites.reserve(N: NumCtxs);
211 for (uint64_t J = 0; J < NumCtxs; J++) {
212 static_assert(sizeof(LinearCallStackId) <= sizeof(CallStackId));
213 LinearCallStackId CSId =
214 endian::readNext<LinearCallStackId, llvm::endianness::little>(memory&: Ptr);
215 const uint64_t NumGuids =
216 endian::readNext<uint64_t, llvm::endianness::little>(memory&: Ptr);
217 SmallVector<GlobalValue::GUID, 1> Guids;
218 Guids.reserve(N: NumGuids);
219 for (uint64_t K = 0; K < NumGuids; ++K)
220 Guids.push_back(
221 Elt: endian::readNext<GlobalValue::GUID, llvm::endianness::little>(memory&: Ptr));
222 Record.CallSites.emplace_back(Args&: CSId, Args: std::move(Guids));
223 }
224
225 return Record;
226}
227
228IndexedMemProfRecord
229IndexedMemProfRecord::deserialize(const MemProfSchema &Schema,
230 const unsigned char *Ptr,
231 IndexedVersion Version) {
232 switch (Version) {
233 case Version3:
234 return deserializeV3(Schema, Ptr);
235 case Version4:
236 return deserializeV4(Schema, Ptr);
237 }
238 llvm_unreachable("unsupported MemProf version");
239}
240
241MemProfRecord IndexedMemProfRecord::toMemProfRecord(
242 llvm::function_ref<std::vector<Frame>(const CallStackId)> Callback) const {
243 MemProfRecord Record;
244
245 Record.AllocSites.reserve(N: AllocSites.size());
246 for (const IndexedAllocationInfo &IndexedAI : AllocSites) {
247 AllocationInfo AI;
248 AI.Info = IndexedAI.Info;
249 AI.CallStack = Callback(IndexedAI.CSId);
250 Record.AllocSites.push_back(Elt: std::move(AI));
251 }
252
253 Record.CallSites.reserve(N: CallSites.size());
254 for (const IndexedCallSiteInfo &CS : CallSites) {
255 std::vector<Frame> Frames = Callback(CS.CSId);
256 Record.CallSites.emplace_back(Args: std::move(Frames), Args: CS.CalleeGuids);
257 }
258
259 return Record;
260}
261
262GlobalValue::GUID getGUID(const StringRef FunctionName) {
263 // Canonicalize the function name to drop suffixes such as ".llvm.". Note
264 // we do not drop any ".__uniq." suffixes, as getCanonicalFnName does not drop
265 // those by default. This is by design to differentiate internal linkage
266 // functions during matching. By dropping the other suffixes we can then match
267 // functions in the profile use phase prior to their addition. Note that this
268 // applies to both instrumented and sampled function names.
269 StringRef CanonicalName =
270 sampleprof::FunctionSamples::getCanonicalFnName(FnName: FunctionName);
271
272 // We use the function guid which we expect to be a uint64_t. At
273 // this time, it is the lower 64 bits of the md5 of the canonical
274 // function name.
275 return Function::getGUIDAssumingExternalLinkage(GlobalName: CanonicalName);
276}
277
278Expected<MemProfSchema> readMemProfSchema(const unsigned char *&Buffer) {
279 using namespace support;
280
281 const unsigned char *Ptr = Buffer;
282 const uint64_t NumSchemaIds =
283 endian::readNext<uint64_t, llvm::endianness::little>(memory&: Ptr);
284 if (NumSchemaIds > static_cast<uint64_t>(Meta::Size)) {
285 return make_error<InstrProfError>(Args: instrprof_error::malformed,
286 Args: "memprof schema invalid");
287 }
288
289 MemProfSchema Result;
290 for (size_t I = 0; I < NumSchemaIds; I++) {
291 const uint64_t Tag =
292 endian::readNext<uint64_t, llvm::endianness::little>(memory&: Ptr);
293 if (Tag >= static_cast<uint64_t>(Meta::Size)) {
294 return make_error<InstrProfError>(Args: instrprof_error::malformed,
295 Args: "memprof schema invalid");
296 }
297 Result.push_back(Elt: static_cast<Meta>(Tag));
298 }
299 // Advance the buffer to one past the schema if we succeeded.
300 Buffer = Ptr;
301 return Result;
302}
303} // namespace memprof
304} // namespace llvm
305