| 1 | //===--- UnicodeCharSetsGenerator.cpp - Unicode character sets generator --===// |
| 2 | // |
| 3 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
| 4 | // See https://llvm.org/LICENSE.txt for license information. |
| 5 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
| 6 | // |
| 7 | //===----------------------------------------------------------------------===// |
| 8 | // |
| 9 | // This file is used to generate clang/lib/Lex/UnicodeCharSetsGenerated.cpp and |
| 10 | // llvm/lib/Support/UnicodeCharSetsGenerated.cpp from the XML representation of |
| 11 | // the Unicode Character Database available at |
| 12 | // https://www.unicode.org/Public/UCD/latest/ucdxml/ |
| 13 | // |
| 14 | //===----------------------------------------------------------------------===// |
| 15 | |
| 16 | #include "llvm/ADT/STLExtras.h" |
| 17 | #include "llvm/ADT/SmallVector.h" |
| 18 | #include "llvm/ADT/Twine.h" |
| 19 | #include "llvm/Support/CommandLine.h" |
| 20 | #include "llvm/Support/FileSystem.h" |
| 21 | #include "llvm/Support/FormatVariadic.h" |
| 22 | #include "llvm/Support/InitLLVM.h" |
| 23 | #include "llvm/Support/ToolOutputFile.h" |
| 24 | #include "llvm/Support/UnicodeCharRanges.h" |
| 25 | #include "llvm/Support/WithColor.h" |
| 26 | |
| 27 | #include <libxml/xmlreader.h> |
| 28 | #include <vector> |
| 29 | |
| 30 | using namespace llvm; |
| 31 | |
| 32 | static cl::opt<std::string> UCDFile(cl::Positional, cl::Required, |
| 33 | cl::desc("<UCD XML file>" )); |
| 34 | static cl::opt<std::string> ClangOutput(cl::Positional, cl::Required, |
| 35 | cl::desc("<clang output file>" )); |
| 36 | static cl::opt<std::string> LLVMOutput(cl::Positional, cl::Required, |
| 37 | cl::desc("<LLVM output file>" )); |
| 38 | |
| 39 | static constexpr unsigned CodePointCount = 0x110000; |
| 40 | |
| 41 | static constexpr StringLiteral = R"( |
| 42 | //===----------------------------------------------------------------------===// |
| 43 | // |
| 44 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
| 45 | // See https://llvm.org/LICENSE.txt for license information. |
| 46 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
| 47 | // |
| 48 | //===----------------------------------------------------------------------===// |
| 49 | // |
| 50 | // This file was generated by llvm/utils/UnicodeData/UnicodeCharSetsGenerator |
| 51 | // from the Unicode {0} Character Database. Do not edit manually. |
| 52 | // |
| 53 | //===----------------------------------------------------------------------===// |
| 54 | |
| 55 | #include "llvm/Support/UnicodeCharRanges.h" |
| 56 | |
| 57 | using llvm::sys::UnicodeCharRange; |
| 58 | using CharRanges = llvm::sys::UnicodeCharSet::CharRanges; |
| 59 | |
| 60 | namespace {1} {{ |
| 61 | )" ; |
| 62 | |
| 63 | [[noreturn]] static void error(const Twine &Message) { |
| 64 | WithColor::error() << Message << '\n'; |
| 65 | exit(EXIT_FAILURE); |
| 66 | } |
| 67 | |
| 68 | namespace { |
| 69 | struct Properties { |
| 70 | unsigned XIDStart : 1; |
| 71 | unsigned XIDContinue : 1; |
| 72 | unsigned MathStart : 1; |
| 73 | unsigned MathContinue : 1; |
| 74 | unsigned Printable : 1; |
| 75 | unsigned Format : 1; |
| 76 | unsigned Combining : 1; |
| 77 | unsigned DoubleWidth : 1; |
| 78 | }; |
| 79 | |
| 80 | struct CharacterDatabase { |
| 81 | std::string Version; |
| 82 | std::vector<Properties> Chars; |
| 83 | }; |
| 84 | |
| 85 | struct XmlDeleter { |
| 86 | void operator()(xmlTextReader *Reader) { xmlFreeTextReader(reader: Reader); } |
| 87 | void operator()(xmlChar *Ptr) { xmlFree(Ptr); } |
| 88 | }; |
| 89 | } // namespace |
| 90 | |
| 91 | static StringRef fromXmlChar(const xmlChar *S) { |
| 92 | return S ? StringRef(reinterpret_cast<const char *>(S)) : StringRef(); |
| 93 | } |
| 94 | |
| 95 | static bool isPrintableGC(StringRef GC) { |
| 96 | return GC == "Zs" || (!GC.empty() && StringRef("LMNPS" ).contains(C: GC.front())); |
| 97 | } |
| 98 | |
| 99 | // Loads the properties we care about from the flat XML representation of the |
| 100 | // UCD, which describes each code point, or range of code points, with all of |
| 101 | // its properties, e.g. |
| 102 | // <char cp="0041" age="1.1" na="LATIN CAPITAL LETTER A" gc="Lu" ... /> |
| 103 | static CharacterDatabase loadDatabase(StringRef Path) { |
| 104 | std::unique_ptr<xmlTextReader, XmlDeleter> Reader(xmlReaderForFile( |
| 105 | filename: Path.str().c_str(), /*encoding=*/nullptr, options: XML_PARSE_NONET)); |
| 106 | if (!Reader) |
| 107 | error(Message: Path + ": cannot open file" ); |
| 108 | |
| 109 | CharacterDatabase Database; |
| 110 | Database.Chars.resize(new_size: CodePointCount); |
| 111 | int Status; |
| 112 | while ((Status = xmlTextReaderRead(reader: Reader.get())) == 1) { |
| 113 | if (xmlTextReaderNodeType(reader: Reader.get()) != XML_READER_TYPE_ELEMENT) |
| 114 | continue; |
| 115 | StringRef Element = fromXmlChar(S: xmlTextReaderConstLocalName(reader: Reader.get())); |
| 116 | if (Element == "description" ) { |
| 117 | // <description>Unicode 18.0.0</description> |
| 118 | std::unique_ptr<xmlChar, XmlDeleter> Description( |
| 119 | xmlTextReaderReadString(reader: Reader.get())); |
| 120 | if (Description) |
| 121 | Database.Version = fromXmlChar(S: Description.get()) |
| 122 | .rsplit(Separator: ' ') |
| 123 | .second.rsplit(Separator: '.') |
| 124 | .first.str(); |
| 125 | continue; |
| 126 | } |
| 127 | if (!is_contained(Set: {"char" , "reserved" , "noncharacter" , "surrogate" }, |
| 128 | Element)) |
| 129 | continue; |
| 130 | |
| 131 | unsigned First = 0, Last = 0; |
| 132 | bool HasRange = false; |
| 133 | unsigned Found = 0; |
| 134 | Properties P = {}; |
| 135 | while (xmlTextReaderMoveToNextAttribute(reader: Reader.get()) == 1) { |
| 136 | StringRef Name = fromXmlChar(S: xmlTextReaderConstLocalName(reader: Reader.get())); |
| 137 | StringRef Value = fromXmlChar(S: xmlTextReaderConstValue(reader: Reader.get())); |
| 138 | if (Name == "cp" || Name == "first-cp" ) { |
| 139 | if (Value.getAsInteger(Radix: 16, Result&: First)) |
| 140 | break; |
| 141 | if (Name == "cp" ) |
| 142 | Last = First; |
| 143 | HasRange = true; |
| 144 | } else if (Name == "last-cp" ) { |
| 145 | if (Value.getAsInteger(Radix: 16, Result&: Last)) |
| 146 | HasRange = false; |
| 147 | } else if (Name == "XIDS" ) { |
| 148 | P.XIDStart = Value == "Y" ; |
| 149 | ++Found; |
| 150 | } else if (Name == "XIDC" ) { |
| 151 | P.XIDContinue = Value == "Y" ; |
| 152 | ++Found; |
| 153 | } else if (Name == "ID_Compat_Math_Start" ) { |
| 154 | P.MathStart = Value == "Y" ; |
| 155 | ++Found; |
| 156 | } else if (Name == "ID_Compat_Math_Continue" ) { |
| 157 | P.MathContinue = Value == "Y" ; |
| 158 | ++Found; |
| 159 | } else if (Name == "gc" ) { |
| 160 | P.Printable = isPrintableGC(GC: Value); |
| 161 | P.Format = Value == "Cf" ; |
| 162 | P.Combining = Value == "Mn" || Value == "Me" ; |
| 163 | ++Found; |
| 164 | } else if (Name == "ea" ) { |
| 165 | P.DoubleWidth = Value == "F" || Value == "W" ; |
| 166 | ++Found; |
| 167 | } |
| 168 | } |
| 169 | |
| 170 | if (!HasRange || First > Last || Last >= CodePointCount || Found != 6) |
| 171 | error(Message: Path + ":" + Twine(xmlTextReaderGetParserLineNumber(reader: Reader.get())) + |
| 172 | ": invalid <" + Element + "> element" ); |
| 173 | llvm::fill(Range: MutableArrayRef(Database.Chars).slice(N: First, M: Last - First + 1), |
| 174 | Value&: P); |
| 175 | } |
| 176 | if (Status != 0) |
| 177 | error(Message: Path + ": failed to parse XML" ); |
| 178 | if (Database.Version.empty()) |
| 179 | error(Message: Path + ": missing Unicode version" ); |
| 180 | return Database; |
| 181 | } |
| 182 | |
| 183 | template <typename Predicate> |
| 184 | static void emitTable(raw_ostream &OS, StringRef Name, StringRef Description, |
| 185 | ArrayRef<Properties> Chars, Predicate Pred) { |
| 186 | SmallVector<sys::UnicodeCharRange> Ranges; |
| 187 | for (unsigned CodePoint = 0, E = Chars.size(); CodePoint != E; ++CodePoint) { |
| 188 | if (!Pred(Chars[CodePoint])) |
| 189 | continue; |
| 190 | if (!Ranges.empty() && Ranges.back().Upper + 1 == CodePoint) |
| 191 | Ranges.back().Upper = CodePoint; |
| 192 | else |
| 193 | Ranges.push_back(Elt: {.Lower: CodePoint, .Upper: CodePoint}); |
| 194 | } |
| 195 | |
| 196 | OS << "\n// " << Description << "\nstatic constexpr UnicodeCharRange " << Name |
| 197 | << "Data[] = {\n" ; |
| 198 | for (sys::UnicodeCharRange Range : Ranges) |
| 199 | OS << formatv(Fmt: " {{{0:X4}, {1:X4}},\n" , Vals&: Range.Lower, Vals&: Range.Upper); |
| 200 | OS << "};\nextern const CharRanges " << Name << " = " << Name << "Data;\n" ; |
| 201 | } |
| 202 | |
| 203 | static void writeFile(StringRef Path, StringRef Namespace, StringRef Version, |
| 204 | function_ref<void(raw_ostream &)> EmitTables) { |
| 205 | std::error_code EC; |
| 206 | ToolOutputFile Out(Path, EC, sys::fs::OF_Text); |
| 207 | if (EC) |
| 208 | error(Message: Path + ": " + EC.message()); |
| 209 | |
| 210 | Out.os() << formatv(Fmt: FileHeader.drop_front().data(), Vals&: Version, Vals&: Namespace); |
| 211 | EmitTables(Out.os()); |
| 212 | Out.os() << "\n} // namespace " << Namespace << '\n'; |
| 213 | Out.keep(); |
| 214 | } |
| 215 | |
| 216 | int main(int argc, char **argv) { |
| 217 | InitLLVM X(argc, argv); |
| 218 | cl::ParseCommandLineOptions( |
| 219 | argc, argv, |
| 220 | Overview: "Unicode character sets generator\n\n" |
| 221 | " Generates clang/lib/Lex/UnicodeCharSetsGenerated.cpp and\n" |
| 222 | " llvm/lib/Support/UnicodeCharSetsGenerated.cpp from the flat XML\n" |
| 223 | " representation of the Unicode Character Database, which can be\n" |
| 224 | " downloaded from https://www.unicode.org/Public/UCD/latest/ucdxml/\n" |
| 225 | " The generated files must then be formatted with clang-format.\n" ); |
| 226 | |
| 227 | CharacterDatabase UCD = loadDatabase(Path: UCDFile); |
| 228 | ArrayRef<Properties> Chars = UCD.Chars; |
| 229 | |
| 230 | writeFile(Path: ClangOutput, Namespace: "clang" , Version: UCD.Version, EmitTables: [&](raw_ostream &OS) { |
| 231 | emitTable(OS, Name: "GeneratedXIDStartRanges" , Description: "XID_Start" , Chars, |
| 232 | Pred: [](Properties P) { return P.XIDStart; }); |
| 233 | emitTable(OS, Name: "GeneratedXIDContinueRanges" , |
| 234 | Description: "XID_Continue, excluding XID_Start" , Chars, |
| 235 | Pred: [](Properties P) { return P.XIDContinue && !P.XIDStart; }); |
| 236 | emitTable(OS, Name: "GeneratedMathematicalNotationProfileIDStartRanges" , |
| 237 | Description: "ID_Compat_Math_Start" , Chars, |
| 238 | Pred: [](Properties P) { return P.MathStart; }); |
| 239 | emitTable(OS, Name: "GeneratedMathematicalNotationProfileIDContinueRanges" , |
| 240 | Description: "ID_Compat_Math_Continue, excluding ID_Compat_Math_Start" , Chars, |
| 241 | Pred: [](Properties P) { return P.MathContinue && !P.MathStart; }); |
| 242 | }); |
| 243 | writeFile( |
| 244 | Path: LLVMOutput, Namespace: "llvm::sys::unicode" , Version: UCD.Version, EmitTables: [&](raw_ostream &OS) { |
| 245 | emitTable(OS, Name: "GeneratedPrintableRanges" , |
| 246 | Description: "General_Category=L|M|N|P|S|Zs" , Chars, |
| 247 | Pred: [](Properties P) { return P.Printable; }); |
| 248 | emitTable(OS, Name: "GeneratedFormatCharacterRanges" , Description: "General_Category=Cf" , |
| 249 | Chars, Pred: [](Properties P) { return P.Format; }); |
| 250 | emitTable(OS, Name: "GeneratedCombiningCharacterRanges" , |
| 251 | Description: "General_Category=Mn|Me" , Chars, |
| 252 | Pred: [](Properties P) { return P.Combining; }); |
| 253 | emitTable(OS, Name: "GeneratedDoubleWidthCharacterRanges" , |
| 254 | Description: "East_Asian_Width=F|W" , Chars, |
| 255 | Pred: [](Properties P) { return P.DoubleWidth; }); |
| 256 | }); |
| 257 | } |
| 258 | |