1//===- llvm/Support/Unicode.cpp - Unicode character properties -*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements functions that allow querying certain properties of
10// Unicode characters.
11//
12//===----------------------------------------------------------------------===//
13
14#include "llvm/Support/Unicode.h"
15#include "llvm/Support/ConvertUTF.h"
16#include "llvm/Support/UnicodeCharRanges.h"
17
18namespace llvm {
19namespace sys {
20namespace unicode {
21
22extern const UnicodeCharSet::CharRanges GeneratedPrintableRanges;
23extern const UnicodeCharSet::CharRanges GeneratedFormatCharacterRanges;
24extern const UnicodeCharSet::CharRanges GeneratedCombiningCharacterRanges;
25extern const UnicodeCharSet::CharRanges GeneratedDoubleWidthCharacterRanges;
26
27/// Unicode code points of the categories L, M, N, P, S and Zs are considered
28/// printable.
29/// In addition, U+00AD SOFT HYPHEN is also considered printable, as
30/// it's actually displayed on most terminals. \return true if the character is
31/// considered printable.
32bool isPrintable(int UCS) {
33 static const UnicodeCharSet Printables(GeneratedPrintableRanges);
34 // Clang special cases 0x00AD (SOFT HYPHEN) which is rendered as an actual
35 // hyphen in most terminals.
36 return UCS == 0x00AD || Printables.contains(C: UCS);
37}
38
39/// Unicode code points of the Cf category are considered
40/// formatting characters.
41bool isFormatting(int UCS) {
42 static const UnicodeCharSet Format(GeneratedFormatCharacterRanges);
43 return Format.contains(C: UCS);
44}
45
46/// Gets the number of positions a character is likely to occupy when output
47/// on a terminal ("character width"). This depends on the implementation of the
48/// terminal, and there's no standard definition of character width.
49/// The implementation defines it in a way that is expected to be compatible
50/// with a generic Unicode-capable terminal.
51/// \return Character width:
52/// * ErrorNonPrintableCharacter (-1) for non-printable characters (as
53/// identified by isPrintable);
54/// * 0 for non-spacing and enclosing combining marks;
55/// * 2 for CJK characters excluding halfwidth forms;
56/// * 1 for all remaining characters.
57static inline int charWidth(int UCS) {
58 if (!isPrintable(UCS))
59 return ErrorNonPrintableCharacter;
60
61 static const UnicodeCharSet CombiningCharacters(
62 GeneratedCombiningCharacterRanges);
63
64 if (CombiningCharacters.contains(C: UCS))
65 return 0;
66
67 static const UnicodeCharSet DoubleWidthCharacters(
68 GeneratedDoubleWidthCharacterRanges);
69
70 if (DoubleWidthCharacters.contains(C: UCS))
71 return 2;
72 return 1;
73}
74
75static bool isprintableascii(char c) { return c > 31 && c < 127; }
76
77int columnWidthUTF8(StringRef Text) {
78 unsigned ColumnWidth = 0;
79 unsigned Length;
80 for (size_t i = 0, e = Text.size(); i < e; i += Length) {
81 Length = getNumBytesForUTF8(firstByte: Text[i]);
82
83 // fast path for ASCII characters
84 if (Length == 1) {
85 if (!isprintableascii(c: Text[i]))
86 return ErrorNonPrintableCharacter;
87 ColumnWidth += 1;
88 continue;
89 }
90
91 if (Length <= 0 || i + Length > Text.size())
92 return ErrorInvalidUTF8;
93 UTF32 buf[1];
94 const UTF8 *Start = reinterpret_cast<const UTF8 *>(Text.data() + i);
95 UTF32 *Target = &buf[0];
96 if (conversionOK != ConvertUTF8toUTF32(sourceStart: &Start, sourceEnd: Start + Length, targetStart: &Target,
97 targetEnd: Target + 1, flags: strictConversion))
98 return ErrorInvalidUTF8;
99 int Width = charWidth(UCS: buf[0]);
100 if (Width < 0)
101 return ErrorNonPrintableCharacter;
102 ColumnWidth += Width;
103 }
104 return ColumnWidth;
105}
106
107} // namespace unicode
108} // namespace sys
109} // namespace llvm
110