| 1 | //===-- CodeGenData.cpp ---------------------------------------------------===// |
| 2 | // |
| 3 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
| 4 | // See https://llvm.org/LICENSE.txt for license information. |
| 5 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
| 6 | // |
| 7 | //===----------------------------------------------------------------------===// |
| 8 | // |
| 9 | // This file contains support for codegen data that has stable summary which |
| 10 | // can be used to optimize the code in the subsequent codegen. |
| 11 | // |
| 12 | //===----------------------------------------------------------------------===// |
| 13 | |
| 14 | #include "CGDataOptions.h" |
| 15 | #include "llvm/Bitcode/BitcodeWriter.h" |
| 16 | #include "llvm/CGData/CodeGenDataReader.h" |
| 17 | #include "llvm/CGData/OutlinedHashTreeRecord.h" |
| 18 | #include "llvm/CGData/StableFunctionMapRecord.h" |
| 19 | #include "llvm/Object/ObjectFile.h" |
| 20 | #include "llvm/Support/Caching.h" |
| 21 | #include "llvm/Support/WithColor.h" |
| 22 | |
| 23 | #define DEBUG_TYPE "cg-data" |
| 24 | |
| 25 | using namespace llvm; |
| 26 | using namespace cgdata; |
| 27 | |
| 28 | static std::string getCGDataErrString(cgdata_error Err, |
| 29 | const std::string &ErrMsg = "" ) { |
| 30 | std::string Msg; |
| 31 | raw_string_ostream OS(Msg); |
| 32 | |
| 33 | switch (Err) { |
| 34 | case cgdata_error::success: |
| 35 | OS << "success" ; |
| 36 | break; |
| 37 | case cgdata_error::eof: |
| 38 | OS << "end of File" ; |
| 39 | break; |
| 40 | case cgdata_error::bad_magic: |
| 41 | OS << "invalid codegen data (bad magic)" ; |
| 42 | break; |
| 43 | case cgdata_error::bad_header: |
| 44 | OS << "invalid codegen data (file header is corrupt)" ; |
| 45 | break; |
| 46 | case cgdata_error::empty_cgdata: |
| 47 | OS << "empty codegen data" ; |
| 48 | break; |
| 49 | case cgdata_error::malformed: |
| 50 | OS << "malformed codegen data" ; |
| 51 | break; |
| 52 | case cgdata_error::unsupported_version: |
| 53 | OS << "unsupported codegen data version" ; |
| 54 | break; |
| 55 | } |
| 56 | |
| 57 | // If optional error message is not empty, append it to the message. |
| 58 | if (!ErrMsg.empty()) |
| 59 | OS << ": " << ErrMsg; |
| 60 | |
| 61 | return OS.str(); |
| 62 | } |
| 63 | |
| 64 | namespace { |
| 65 | |
| 66 | // FIXME: This class is only here to support the transition to llvm::Error. It |
| 67 | // will be removed once this transition is complete. Clients should prefer to |
| 68 | // deal with the Error value directly, rather than converting to error_code. |
| 69 | class CGDataErrorCategoryType : public std::error_category { |
| 70 | const char *name() const noexcept override { return "llvm.cgdata" ; } |
| 71 | |
| 72 | std::string message(int IE) const override { |
| 73 | return getCGDataErrString(Err: static_cast<cgdata_error>(IE)); |
| 74 | } |
| 75 | }; |
| 76 | |
| 77 | } // end anonymous namespace |
| 78 | |
| 79 | const std::error_category &llvm::cgdata_category() { |
| 80 | static CGDataErrorCategoryType ErrorCategory; |
| 81 | return ErrorCategory; |
| 82 | } |
| 83 | |
| 84 | std::string CGDataError::message() const { |
| 85 | return getCGDataErrString(Err, ErrMsg: Msg); |
| 86 | } |
| 87 | |
| 88 | char CGDataError::ID = 0; |
| 89 | |
| 90 | namespace { |
| 91 | |
| 92 | const char *CodeGenDataSectNameCommon[] = { |
| 93 | #define CG_DATA_SECT_ENTRY(Kind, SectNameCommon, SectNameCoff, Prefix) \ |
| 94 | SectNameCommon, |
| 95 | #include "llvm/CGData/CodeGenData.inc" |
| 96 | }; |
| 97 | |
| 98 | const char *CodeGenDataSectNameCoff[] = { |
| 99 | #define CG_DATA_SECT_ENTRY(Kind, SectNameCommon, SectNameCoff, Prefix) \ |
| 100 | SectNameCoff, |
| 101 | #include "llvm/CGData/CodeGenData.inc" |
| 102 | }; |
| 103 | |
| 104 | const char *CodeGenDataSectNamePrefix[] = { |
| 105 | #define CG_DATA_SECT_ENTRY(Kind, SectNameCommon, SectNameCoff, Prefix) Prefix, |
| 106 | #include "llvm/CGData/CodeGenData.inc" |
| 107 | }; |
| 108 | |
| 109 | } // namespace |
| 110 | |
| 111 | bool llvm::cgdata::thinLTOTwoRounds() { |
| 112 | return CGDataOptions::Global.codegen_data_thinlto_two_rounds; |
| 113 | } |
| 114 | |
| 115 | namespace llvm { |
| 116 | |
| 117 | std::string getCodeGenDataSectionName(CGDataSectKind CGSK, |
| 118 | Triple::ObjectFormatType OF, |
| 119 | bool AddSegmentInfo) { |
| 120 | std::string SectName; |
| 121 | |
| 122 | if (OF == Triple::MachO && AddSegmentInfo) |
| 123 | SectName = CodeGenDataSectNamePrefix[CGSK]; |
| 124 | |
| 125 | if (OF == Triple::COFF) |
| 126 | SectName += CodeGenDataSectNameCoff[CGSK]; |
| 127 | else |
| 128 | SectName += CodeGenDataSectNameCommon[CGSK]; |
| 129 | |
| 130 | return SectName; |
| 131 | } |
| 132 | |
| 133 | std::unique_ptr<CodeGenData> CodeGenData::Instance = nullptr; |
| 134 | std::once_flag CodeGenData::OnceFlag; |
| 135 | |
| 136 | CodeGenData &CodeGenData::getInstance() { |
| 137 | std::call_once(once&: CodeGenData::OnceFlag, f: []() { |
| 138 | Instance = std::unique_ptr<CodeGenData>(new CodeGenData()); |
| 139 | |
| 140 | const CGDataOptions &Opts = CGDataOptions::Global; |
| 141 | if (Opts.codegen_data_generate || Opts.codegen_data_thinlto_two_rounds) |
| 142 | Instance->EmitCGData = true; |
| 143 | else if (!Opts.codegen_data_use_path.empty()) { |
| 144 | // Initialize the global CGData if the input file name is given. |
| 145 | // We do not error-out when failing to parse the input file. |
| 146 | // Instead, just emit an warning message and fall back as if no CGData |
| 147 | // were available. |
| 148 | auto FS = vfs::getRealFileSystem(); |
| 149 | auto ReaderOrErr = |
| 150 | CodeGenDataReader::create(Path: Opts.codegen_data_use_path, FS&: *FS, |
| 151 | LazyLoading: Opts.indexed_codegen_data_lazy_loading); |
| 152 | if (Error E = ReaderOrErr.takeError()) { |
| 153 | warn(E: std::move(E), Whence: Opts.codegen_data_use_path); |
| 154 | return; |
| 155 | } |
| 156 | // Publish each CGData based on the data type in the header. |
| 157 | auto Reader = ReaderOrErr->get(); |
| 158 | if (Reader->hasOutlinedHashTree()) |
| 159 | Instance->publishOutlinedHashTree(HashTree: Reader->releaseOutlinedHashTree()); |
| 160 | if (Reader->hasStableFunctionMap()) |
| 161 | Instance->publishStableFunctionMap(FunctionMap: Reader->releaseStableFunctionMap()); |
| 162 | } |
| 163 | }); |
| 164 | return *Instance; |
| 165 | } |
| 166 | |
| 167 | namespace IndexedCGData { |
| 168 | |
| 169 | Expected<Header> Header::(const unsigned char *Curr) { |
| 170 | using namespace support; |
| 171 | |
| 172 | static_assert(std::is_standard_layout_v<llvm::IndexedCGData::Header>, |
| 173 | "The header should be standard layout type since we use offset " |
| 174 | "of fields to read." ); |
| 175 | Header H; |
| 176 | H.Magic = endian::readNext<uint64_t, endianness::little, unaligned>(memory&: Curr); |
| 177 | if (H.Magic != IndexedCGData::Magic) |
| 178 | return make_error<CGDataError>(Args: cgdata_error::bad_magic); |
| 179 | H.Version = endian::readNext<uint32_t, endianness::little, unaligned>(memory&: Curr); |
| 180 | if (H.Version > IndexedCGData::CGDataVersion::CurrentVersion) |
| 181 | return make_error<CGDataError>(Args: cgdata_error::unsupported_version); |
| 182 | H.DataKind = endian::readNext<uint32_t, endianness::little, unaligned>(memory&: Curr); |
| 183 | |
| 184 | static_assert(IndexedCGData::CGDataVersion::CurrentVersion == Version4, |
| 185 | "Please update the offset computation below if a new field has " |
| 186 | "been added to the header." ); |
| 187 | H.OutlinedHashTreeOffset = |
| 188 | endian::readNext<uint64_t, endianness::little, unaligned>(memory&: Curr); |
| 189 | if (H.Version >= 2) |
| 190 | H.StableFunctionMapOffset = |
| 191 | endian::readNext<uint64_t, endianness::little, unaligned>(memory&: Curr); |
| 192 | |
| 193 | return H; |
| 194 | } |
| 195 | |
| 196 | } // end namespace IndexedCGData |
| 197 | |
| 198 | namespace cgdata { |
| 199 | |
| 200 | void warn(Twine Message, StringRef Whence, StringRef Hint) { |
| 201 | WithColor::warning(); |
| 202 | if (!Whence.empty()) |
| 203 | errs() << Whence << ": " ; |
| 204 | errs() << Message << "\n" ; |
| 205 | if (!Hint.empty()) |
| 206 | WithColor::note() << Hint << "\n" ; |
| 207 | } |
| 208 | |
| 209 | void warn(Error E, StringRef Whence) { |
| 210 | if (E.isA<CGDataError>()) { |
| 211 | handleAllErrors(E: std::move(E), Handlers: [&](const CGDataError &IPE) { |
| 212 | warn(Message: IPE.message(), Whence, Hint: "" ); |
| 213 | }); |
| 214 | } |
| 215 | } |
| 216 | |
| 217 | void saveModuleForTwoRounds(const Module &TheModule, unsigned Task, |
| 218 | AddStreamFn AddStream) { |
| 219 | LLVM_DEBUG(dbgs() << "Saving module: " << TheModule.getModuleIdentifier() |
| 220 | << " in Task " << Task << "\n" ); |
| 221 | Expected<std::unique_ptr<CachedFileStream>> StreamOrErr = |
| 222 | AddStream(Task, TheModule.getModuleIdentifier()); |
| 223 | if (Error Err = StreamOrErr.takeError()) |
| 224 | report_fatal_error(Err: std::move(Err)); |
| 225 | std::unique_ptr<CachedFileStream> &Stream = *StreamOrErr; |
| 226 | |
| 227 | WriteBitcodeToFile(M: TheModule, Out&: *Stream->OS, |
| 228 | /*ShouldPreserveUseListOrder=*/true); |
| 229 | |
| 230 | if (Error Err = Stream->commit()) |
| 231 | report_fatal_error(Err: std::move(Err)); |
| 232 | } |
| 233 | |
| 234 | std::unique_ptr<Module> loadModuleForTwoRounds(BitcodeModule &OrigModule, |
| 235 | unsigned Task, |
| 236 | LLVMContext &Context, |
| 237 | ArrayRef<StringRef> IRFiles) { |
| 238 | LLVM_DEBUG(dbgs() << "Loading module: " << OrigModule.getModuleIdentifier() |
| 239 | << " in Task " << Task << "\n" ); |
| 240 | auto FileBuffer = MemoryBuffer::getMemBuffer( |
| 241 | InputData: IRFiles[Task], BufferName: "in-memory IR file" , /*RequiresNullTerminator=*/false); |
| 242 | auto RestoredModule = parseBitcodeFile(Buffer: *FileBuffer, Context); |
| 243 | if (!RestoredModule) |
| 244 | report_fatal_error( |
| 245 | reason: Twine("Failed to parse optimized bitcode loaded for Task: " ) + |
| 246 | Twine(Task) + "\n" ); |
| 247 | |
| 248 | // Restore the original module identifier. |
| 249 | (*RestoredModule)->setModuleIdentifier(OrigModule.getModuleIdentifier()); |
| 250 | return std::move(*RestoredModule); |
| 251 | } |
| 252 | |
| 253 | Expected<stable_hash> mergeCodeGenData(ArrayRef<StringRef> ObjFiles) { |
| 254 | OutlinedHashTreeRecord GlobalOutlineRecord; |
| 255 | StableFunctionMapRecord GlobalStableFunctionMapRecord; |
| 256 | stable_hash CombinedHash = 0; |
| 257 | for (auto File : ObjFiles) { |
| 258 | if (File.empty()) |
| 259 | continue; |
| 260 | std::unique_ptr<MemoryBuffer> Buffer = MemoryBuffer::getMemBuffer( |
| 261 | InputData: File, BufferName: "in-memory object file" , /*RequiresNullTerminator=*/false); |
| 262 | Expected<std::unique_ptr<object::ObjectFile>> BinOrErr = |
| 263 | object::ObjectFile::createObjectFile(Object: Buffer->getMemBufferRef()); |
| 264 | if (!BinOrErr) |
| 265 | return BinOrErr.takeError(); |
| 266 | |
| 267 | std::unique_ptr<object::ObjectFile> &Obj = BinOrErr.get(); |
| 268 | if (auto E = CodeGenDataReader::mergeFromObjectFile( |
| 269 | Obj: Obj.get(), GlobalOutlineRecord, GlobalFunctionMapRecord&: GlobalStableFunctionMapRecord, |
| 270 | CombinedHash: &CombinedHash)) |
| 271 | return E; |
| 272 | } |
| 273 | |
| 274 | GlobalStableFunctionMapRecord.finalize(); |
| 275 | |
| 276 | if (!GlobalOutlineRecord.empty()) |
| 277 | cgdata::publishOutlinedHashTree(HashTree: std::move(GlobalOutlineRecord.HashTree)); |
| 278 | if (!GlobalStableFunctionMapRecord.empty()) |
| 279 | cgdata::publishStableFunctionMap( |
| 280 | FunctionMap: std::move(GlobalStableFunctionMapRecord.FunctionMap)); |
| 281 | |
| 282 | return CombinedHash; |
| 283 | } |
| 284 | |
| 285 | } // end namespace cgdata |
| 286 | |
| 287 | } // end namespace llvm |
| 288 | |