1 | //===--- ConvertEBCDIC.cpp - UTF8/EBCDIC CharSet Conversion -----*- C++ -*-===// |
2 | // |
3 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
4 | // See https://llvm.org/LICENSE.txt for license information. |
5 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
6 | // |
7 | //===----------------------------------------------------------------------===// |
8 | /// |
9 | /// \file |
10 | /// This file provides utility functions for converting between EBCDIC-1047 and |
11 | /// UTF-8. |
12 | /// |
13 | /// |
14 | //===----------------------------------------------------------------------===// |
15 | |
16 | #include "llvm/Support/ConvertEBCDIC.h" |
17 | |
18 | using namespace llvm; |
19 | |
20 | static const unsigned char ISO88591ToIBM1047[256] = { |
21 | 0x00, 0x01, 0x02, 0x03, 0x37, 0x2d, 0x2e, 0x2f, 0x16, 0x05, 0x15, 0x0b, |
22 | 0x0c, 0x0d, 0x0e, 0x0f, 0x10, 0x11, 0x12, 0x13, 0x3c, 0x3d, 0x32, 0x26, |
23 | 0x18, 0x19, 0x3f, 0x27, 0x1c, 0x1d, 0x1e, 0x1f, 0x40, 0x5a, 0x7f, 0x7b, |
24 | 0x5b, 0x6c, 0x50, 0x7d, 0x4d, 0x5d, 0x5c, 0x4e, 0x6b, 0x60, 0x4b, 0x61, |
25 | 0xf0, 0xf1, 0xf2, 0xf3, 0xf4, 0xf5, 0xf6, 0xf7, 0xf8, 0xf9, 0x7a, 0x5e, |
26 | 0x4c, 0x7e, 0x6e, 0x6f, 0x7c, 0xc1, 0xc2, 0xc3, 0xc4, 0xc5, 0xc6, 0xc7, |
27 | 0xc8, 0xc9, 0xd1, 0xd2, 0xd3, 0xd4, 0xd5, 0xd6, 0xd7, 0xd8, 0xd9, 0xe2, |
28 | 0xe3, 0xe4, 0xe5, 0xe6, 0xe7, 0xe8, 0xe9, 0xad, 0xe0, 0xbd, 0x5f, 0x6d, |
29 | 0x79, 0x81, 0x82, 0x83, 0x84, 0x85, 0x86, 0x87, 0x88, 0x89, 0x91, 0x92, |
30 | 0x93, 0x94, 0x95, 0x96, 0x97, 0x98, 0x99, 0xa2, 0xa3, 0xa4, 0xa5, 0xa6, |
31 | 0xa7, 0xa8, 0xa9, 0xc0, 0x4f, 0xd0, 0xa1, 0x07, 0x20, 0x21, 0x22, 0x23, |
32 | 0x24, 0x25, 0x06, 0x17, 0x28, 0x29, 0x2a, 0x2b, 0x2c, 0x09, 0x0a, 0x1b, |
33 | 0x30, 0x31, 0x1a, 0x33, 0x34, 0x35, 0x36, 0x08, 0x38, 0x39, 0x3a, 0x3b, |
34 | 0x04, 0x14, 0x3e, 0xff, 0x41, 0xaa, 0x4a, 0xb1, 0x9f, 0xb2, 0x6a, 0xb5, |
35 | 0xbb, 0xb4, 0x9a, 0x8a, 0xb0, 0xca, 0xaf, 0xbc, 0x90, 0x8f, 0xea, 0xfa, |
36 | 0xbe, 0xa0, 0xb6, 0xb3, 0x9d, 0xda, 0x9b, 0x8b, 0xb7, 0xb8, 0xb9, 0xab, |
37 | 0x64, 0x65, 0x62, 0x66, 0x63, 0x67, 0x9e, 0x68, 0x74, 0x71, 0x72, 0x73, |
38 | 0x78, 0x75, 0x76, 0x77, 0xac, 0x69, 0xed, 0xee, 0xeb, 0xef, 0xec, 0xbf, |
39 | 0x80, 0xfd, 0xfe, 0xfb, 0xfc, 0xba, 0xae, 0x59, 0x44, 0x45, 0x42, 0x46, |
40 | 0x43, 0x47, 0x9c, 0x48, 0x54, 0x51, 0x52, 0x53, 0x58, 0x55, 0x56, 0x57, |
41 | 0x8c, 0x49, 0xcd, 0xce, 0xcb, 0xcf, 0xcc, 0xe1, 0x70, 0xdd, 0xde, 0xdb, |
42 | 0xdc, 0x8d, 0x8e, 0xdf}; |
43 | |
44 | static const unsigned char IBM1047ToISO88591[256] = { |
45 | 0x00, 0x01, 0x02, 0x03, 0x9c, 0x09, 0x86, 0x7f, 0x97, 0x8d, 0x8e, 0x0b, |
46 | 0x0c, 0x0d, 0x0e, 0x0f, 0x10, 0x11, 0x12, 0x13, 0x9d, 0x0a, 0x08, 0x87, |
47 | 0x18, 0x19, 0x92, 0x8f, 0x1c, 0x1d, 0x1e, 0x1f, 0x80, 0x81, 0x82, 0x83, |
48 | 0x84, 0x85, 0x17, 0x1b, 0x88, 0x89, 0x8a, 0x8b, 0x8c, 0x05, 0x06, 0x07, |
49 | 0x90, 0x91, 0x16, 0x93, 0x94, 0x95, 0x96, 0x04, 0x98, 0x99, 0x9a, 0x9b, |
50 | 0x14, 0x15, 0x9e, 0x1a, 0x20, 0xa0, 0xe2, 0xe4, 0xe0, 0xe1, 0xe3, 0xe5, |
51 | 0xe7, 0xf1, 0xa2, 0x2e, 0x3c, 0x28, 0x2b, 0x7c, 0x26, 0xe9, 0xea, 0xeb, |
52 | 0xe8, 0xed, 0xee, 0xef, 0xec, 0xdf, 0x21, 0x24, 0x2a, 0x29, 0x3b, 0x5e, |
53 | 0x2d, 0x2f, 0xc2, 0xc4, 0xc0, 0xc1, 0xc3, 0xc5, 0xc7, 0xd1, 0xa6, 0x2c, |
54 | 0x25, 0x5f, 0x3e, 0x3f, 0xf8, 0xc9, 0xca, 0xcb, 0xc8, 0xcd, 0xce, 0xcf, |
55 | 0xcc, 0x60, 0x3a, 0x23, 0x40, 0x27, 0x3d, 0x22, 0xd8, 0x61, 0x62, 0x63, |
56 | 0x64, 0x65, 0x66, 0x67, 0x68, 0x69, 0xab, 0xbb, 0xf0, 0xfd, 0xfe, 0xb1, |
57 | 0xb0, 0x6a, 0x6b, 0x6c, 0x6d, 0x6e, 0x6f, 0x70, 0x71, 0x72, 0xaa, 0xba, |
58 | 0xe6, 0xb8, 0xc6, 0xa4, 0xb5, 0x7e, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78, |
59 | 0x79, 0x7a, 0xa1, 0xbf, 0xd0, 0x5b, 0xde, 0xae, 0xac, 0xa3, 0xa5, 0xb7, |
60 | 0xa9, 0xa7, 0xb6, 0xbc, 0xbd, 0xbe, 0xdd, 0xa8, 0xaf, 0x5d, 0xb4, 0xd7, |
61 | 0x7b, 0x41, 0x42, 0x43, 0x44, 0x45, 0x46, 0x47, 0x48, 0x49, 0xad, 0xf4, |
62 | 0xf6, 0xf2, 0xf3, 0xf5, 0x7d, 0x4a, 0x4b, 0x4c, 0x4d, 0x4e, 0x4f, 0x50, |
63 | 0x51, 0x52, 0xb9, 0xfb, 0xfc, 0xf9, 0xfa, 0xff, 0x5c, 0xf7, 0x53, 0x54, |
64 | 0x55, 0x56, 0x57, 0x58, 0x59, 0x5a, 0xb2, 0xd4, 0xd6, 0xd2, 0xd3, 0xd5, |
65 | 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, 0x38, 0x39, 0xb3, 0xdb, |
66 | 0xdc, 0xd9, 0xda, 0x9f}; |
67 | |
68 | std::error_code |
69 | ConverterEBCDIC::convertToEBCDIC(StringRef Source, |
70 | SmallVectorImpl<char> &Result) { |
71 | assert(Result.empty() && "Result must be empty!" ); |
72 | const unsigned char *Table = ISO88591ToIBM1047; |
73 | const unsigned char *Ptr = |
74 | reinterpret_cast<const unsigned char *>(Source.data()); |
75 | size_t Length = Source.size(); |
76 | Result.reserve(N: Length); |
77 | while (Length--) { |
78 | unsigned char Ch = *Ptr++; |
79 | // Handle UTF-8 2-byte-sequences in input. |
80 | if (Ch >= 128) { |
81 | // Only two-byte sequences can be decoded. |
82 | if (Ch != 0xc2 && Ch != 0xc3) |
83 | return std::make_error_code(e: std::errc::illegal_byte_sequence); |
84 | // Is buffer truncated? |
85 | if (!Length) |
86 | return std::make_error_code(e: std::errc::invalid_argument); |
87 | unsigned char Ch2 = *Ptr++; |
88 | // Is second byte well-formed? |
89 | if ((Ch2 & 0xc0) != 0x80) |
90 | return std::make_error_code(e: std::errc::illegal_byte_sequence); |
91 | Ch = Ch2 | (Ch << 6); |
92 | Length--; |
93 | } |
94 | // Translate the character. |
95 | Ch = Table[Ch]; |
96 | Result.push_back(Elt: static_cast<char>(Ch)); |
97 | } |
98 | return std::error_code(); |
99 | } |
100 | |
101 | void ConverterEBCDIC::convertToUTF8(StringRef Source, |
102 | SmallVectorImpl<char> &Result) { |
103 | assert(Result.empty() && "Result must be empty!" ); |
104 | |
105 | const unsigned char *Table = IBM1047ToISO88591; |
106 | const unsigned char *Ptr = |
107 | reinterpret_cast<const unsigned char *>(Source.data()); |
108 | size_t Length = Source.size(); |
109 | Result.reserve(N: Length); |
110 | while (Length--) { |
111 | unsigned char Ch = *Ptr++; |
112 | // Translate the character. |
113 | Ch = Table[Ch]; |
114 | // Handle UTF-8 2-byte-sequences in output. |
115 | if (Ch >= 128) { |
116 | // First byte prefixed with either 0xc2 or 0xc3. |
117 | Result.push_back(Elt: static_cast<char>(0xc0 | (Ch >> 6))); |
118 | // Second byte is either the same as the ASCII byte or ASCII byte -64. |
119 | Ch = Ch & 0xbf; |
120 | } |
121 | Result.push_back(Elt: static_cast<char>(Ch)); |
122 | } |
123 | } |
124 | |