1//===--- UnicodeCharSetsGenerator.cpp - Unicode character sets generator --===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file is used to generate clang/lib/Lex/UnicodeCharSetsGenerated.cpp and
10// llvm/lib/Support/UnicodeCharSetsGenerated.cpp from the XML representation of
11// the Unicode Character Database available at
12// https://www.unicode.org/Public/UCD/latest/ucdxml/
13//
14//===----------------------------------------------------------------------===//
15
16#include "llvm/ADT/STLExtras.h"
17#include "llvm/ADT/SmallVector.h"
18#include "llvm/ADT/Twine.h"
19#include "llvm/Support/CommandLine.h"
20#include "llvm/Support/FileSystem.h"
21#include "llvm/Support/FormatVariadic.h"
22#include "llvm/Support/InitLLVM.h"
23#include "llvm/Support/ToolOutputFile.h"
24#include "llvm/Support/UnicodeCharRanges.h"
25#include "llvm/Support/WithColor.h"
26
27#include <libxml/xmlreader.h>
28#include <vector>
29
30using namespace llvm;
31
32static cl::opt<std::string> UCDFile(cl::Positional, cl::Required,
33 cl::desc("<UCD XML file>"));
34static cl::opt<std::string> ClangOutput(cl::Positional, cl::Required,
35 cl::desc("<clang output file>"));
36static cl::opt<std::string> LLVMOutput(cl::Positional, cl::Required,
37 cl::desc("<LLVM output file>"));
38
39static constexpr unsigned CodePointCount = 0x110000;
40
41static constexpr StringLiteral FileHeader = R"(
42//===----------------------------------------------------------------------===//
43//
44// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
45// See https://llvm.org/LICENSE.txt for license information.
46// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
47//
48//===----------------------------------------------------------------------===//
49//
50// This file was generated by llvm/utils/UnicodeData/UnicodeCharSetsGenerator
51// from the Unicode {0} Character Database. Do not edit manually.
52//
53//===----------------------------------------------------------------------===//
54
55#include "llvm/Support/UnicodeCharRanges.h"
56
57using llvm::sys::UnicodeCharRange;
58using CharRanges = llvm::sys::UnicodeCharSet::CharRanges;
59
60namespace {1} {{
61)";
62
63[[noreturn]] static void error(const Twine &Message) {
64 WithColor::error() << Message << '\n';
65 exit(EXIT_FAILURE);
66}
67
68namespace {
69struct Properties {
70 unsigned XIDStart : 1;
71 unsigned XIDContinue : 1;
72 unsigned MathStart : 1;
73 unsigned MathContinue : 1;
74 unsigned Printable : 1;
75 unsigned Format : 1;
76 unsigned Combining : 1;
77 unsigned DoubleWidth : 1;
78};
79
80struct CharacterDatabase {
81 std::string Version;
82 std::vector<Properties> Chars;
83};
84
85struct XmlDeleter {
86 void operator()(xmlTextReader *Reader) { xmlFreeTextReader(reader: Reader); }
87 void operator()(xmlChar *Ptr) { xmlFree(Ptr); }
88};
89} // namespace
90
91static StringRef fromXmlChar(const xmlChar *S) {
92 return S ? StringRef(reinterpret_cast<const char *>(S)) : StringRef();
93}
94
95static bool isPrintableGC(StringRef GC) {
96 return GC == "Zs" || (!GC.empty() && StringRef("LMNPS").contains(C: GC.front()));
97}
98
99// Loads the properties we care about from the flat XML representation of the
100// UCD, which describes each code point, or range of code points, with all of
101// its properties, e.g.
102// <char cp="0041" age="1.1" na="LATIN CAPITAL LETTER A" gc="Lu" ... />
103static CharacterDatabase loadDatabase(StringRef Path) {
104 std::unique_ptr<xmlTextReader, XmlDeleter> Reader(xmlReaderForFile(
105 filename: Path.str().c_str(), /*encoding=*/nullptr, options: XML_PARSE_NONET));
106 if (!Reader)
107 error(Message: Path + ": cannot open file");
108
109 CharacterDatabase Database;
110 Database.Chars.resize(new_size: CodePointCount);
111 int Status;
112 while ((Status = xmlTextReaderRead(reader: Reader.get())) == 1) {
113 if (xmlTextReaderNodeType(reader: Reader.get()) != XML_READER_TYPE_ELEMENT)
114 continue;
115 StringRef Element = fromXmlChar(S: xmlTextReaderConstLocalName(reader: Reader.get()));
116 if (Element == "description") {
117 // <description>Unicode 18.0.0</description>
118 std::unique_ptr<xmlChar, XmlDeleter> Description(
119 xmlTextReaderReadString(reader: Reader.get()));
120 if (Description)
121 Database.Version = fromXmlChar(S: Description.get())
122 .rsplit(Separator: ' ')
123 .second.rsplit(Separator: '.')
124 .first.str();
125 continue;
126 }
127 if (!is_contained(Set: {"char", "reserved", "noncharacter", "surrogate"},
128 Element))
129 continue;
130
131 unsigned First = 0, Last = 0;
132 bool HasRange = false;
133 unsigned Found = 0;
134 Properties P = {};
135 while (xmlTextReaderMoveToNextAttribute(reader: Reader.get()) == 1) {
136 StringRef Name = fromXmlChar(S: xmlTextReaderConstLocalName(reader: Reader.get()));
137 StringRef Value = fromXmlChar(S: xmlTextReaderConstValue(reader: Reader.get()));
138 if (Name == "cp" || Name == "first-cp") {
139 if (Value.getAsInteger(Radix: 16, Result&: First))
140 break;
141 if (Name == "cp")
142 Last = First;
143 HasRange = true;
144 } else if (Name == "last-cp") {
145 if (Value.getAsInteger(Radix: 16, Result&: Last))
146 HasRange = false;
147 } else if (Name == "XIDS") {
148 P.XIDStart = Value == "Y";
149 ++Found;
150 } else if (Name == "XIDC") {
151 P.XIDContinue = Value == "Y";
152 ++Found;
153 } else if (Name == "ID_Compat_Math_Start") {
154 P.MathStart = Value == "Y";
155 ++Found;
156 } else if (Name == "ID_Compat_Math_Continue") {
157 P.MathContinue = Value == "Y";
158 ++Found;
159 } else if (Name == "gc") {
160 P.Printable = isPrintableGC(GC: Value);
161 P.Format = Value == "Cf";
162 P.Combining = Value == "Mn" || Value == "Me";
163 ++Found;
164 } else if (Name == "ea") {
165 P.DoubleWidth = Value == "F" || Value == "W";
166 ++Found;
167 }
168 }
169
170 if (!HasRange || First > Last || Last >= CodePointCount || Found != 6)
171 error(Message: Path + ":" + Twine(xmlTextReaderGetParserLineNumber(reader: Reader.get())) +
172 ": invalid <" + Element + "> element");
173 llvm::fill(Range: MutableArrayRef(Database.Chars).slice(N: First, M: Last - First + 1),
174 Value&: P);
175 }
176 if (Status != 0)
177 error(Message: Path + ": failed to parse XML");
178 if (Database.Version.empty())
179 error(Message: Path + ": missing Unicode version");
180 return Database;
181}
182
183template <typename Predicate>
184static void emitTable(raw_ostream &OS, StringRef Name, StringRef Description,
185 ArrayRef<Properties> Chars, Predicate Pred) {
186 SmallVector<sys::UnicodeCharRange> Ranges;
187 for (unsigned CodePoint = 0, E = Chars.size(); CodePoint != E; ++CodePoint) {
188 if (!Pred(Chars[CodePoint]))
189 continue;
190 if (!Ranges.empty() && Ranges.back().Upper + 1 == CodePoint)
191 Ranges.back().Upper = CodePoint;
192 else
193 Ranges.push_back(Elt: {.Lower: CodePoint, .Upper: CodePoint});
194 }
195
196 OS << "\n// " << Description << "\nstatic constexpr UnicodeCharRange " << Name
197 << "Data[] = {\n";
198 for (sys::UnicodeCharRange Range : Ranges)
199 OS << formatv(Fmt: " {{{0:X4}, {1:X4}},\n", Vals&: Range.Lower, Vals&: Range.Upper);
200 OS << "};\nextern const CharRanges " << Name << " = " << Name << "Data;\n";
201}
202
203static void writeFile(StringRef Path, StringRef Namespace, StringRef Version,
204 function_ref<void(raw_ostream &)> EmitTables) {
205 std::error_code EC;
206 ToolOutputFile Out(Path, EC, sys::fs::OF_Text);
207 if (EC)
208 error(Message: Path + ": " + EC.message());
209
210 Out.os() << formatv(Fmt: FileHeader.drop_front().data(), Vals&: Version, Vals&: Namespace);
211 EmitTables(Out.os());
212 Out.os() << "\n} // namespace " << Namespace << '\n';
213 Out.keep();
214}
215
216int main(int argc, char **argv) {
217 InitLLVM X(argc, argv);
218 cl::ParseCommandLineOptions(
219 argc, argv,
220 Overview: "Unicode character sets generator\n\n"
221 " Generates clang/lib/Lex/UnicodeCharSetsGenerated.cpp and\n"
222 " llvm/lib/Support/UnicodeCharSetsGenerated.cpp from the flat XML\n"
223 " representation of the Unicode Character Database, which can be\n"
224 " downloaded from https://www.unicode.org/Public/UCD/latest/ucdxml/\n"
225 " The generated files must then be formatted with clang-format.\n");
226
227 CharacterDatabase UCD = loadDatabase(Path: UCDFile);
228 ArrayRef<Properties> Chars = UCD.Chars;
229
230 writeFile(Path: ClangOutput, Namespace: "clang", Version: UCD.Version, EmitTables: [&](raw_ostream &OS) {
231 emitTable(OS, Name: "GeneratedXIDStartRanges", Description: "XID_Start", Chars,
232 Pred: [](Properties P) { return P.XIDStart; });
233 emitTable(OS, Name: "GeneratedXIDContinueRanges",
234 Description: "XID_Continue, excluding XID_Start", Chars,
235 Pred: [](Properties P) { return P.XIDContinue && !P.XIDStart; });
236 emitTable(OS, Name: "GeneratedMathematicalNotationProfileIDStartRanges",
237 Description: "ID_Compat_Math_Start", Chars,
238 Pred: [](Properties P) { return P.MathStart; });
239 emitTable(OS, Name: "GeneratedMathematicalNotationProfileIDContinueRanges",
240 Description: "ID_Compat_Math_Continue, excluding ID_Compat_Math_Start", Chars,
241 Pred: [](Properties P) { return P.MathContinue && !P.MathStart; });
242 });
243 writeFile(
244 Path: LLVMOutput, Namespace: "llvm::sys::unicode", Version: UCD.Version, EmitTables: [&](raw_ostream &OS) {
245 emitTable(OS, Name: "GeneratedPrintableRanges",
246 Description: "General_Category=L|M|N|P|S|Zs", Chars,
247 Pred: [](Properties P) { return P.Printable; });
248 emitTable(OS, Name: "GeneratedFormatCharacterRanges", Description: "General_Category=Cf",
249 Chars, Pred: [](Properties P) { return P.Format; });
250 emitTable(OS, Name: "GeneratedCombiningCharacterRanges",
251 Description: "General_Category=Mn|Me", Chars,
252 Pred: [](Properties P) { return P.Combining; });
253 emitTable(OS, Name: "GeneratedDoubleWidthCharacterRanges",
254 Description: "East_Asian_Width=F|W", Chars,
255 Pred: [](Properties P) { return P.DoubleWidth; });
256 });
257}
258