Upload Kmake

This commit is contained in:
Gorochu
2026-05-26 23:36:42 -07:00
parent ba051b2f74
commit 555ec72358
41615 changed files with 13344630 additions and 1 deletions

View File

@ -0,0 +1,521 @@
// Copyright (C) 2016 and later: Unicode, Inc. and others. License & terms of use: http://www.unicode.org/copyright.html
// generated by tblgen. You weren't going to edit it by hand, were you?
static const char cp1047_8859_1[256] = {
static_cast<char>(0x00), /* 00 */
static_cast<char>(0x01), /* 01 */
static_cast<char>(0x02), /* 02 */
static_cast<char>(0x03), /* 03 */
static_cast<char>(0x9C), /* 04 */
static_cast<char>(0x09), /* 05 */
static_cast<char>(0x86), /* 06 */
static_cast<char>(0x7F), /* 07 */
static_cast<char>(0x97), /* 08 */
static_cast<char>(0x8D), /* 09 */
static_cast<char>(0x8E), /* 0A */
static_cast<char>(0x0B), /* 0B */
static_cast<char>(0x0C), /* 0C */
static_cast<char>(0x0D), /* 0D */
static_cast<char>(0x0E), /* 0E */
static_cast<char>(0x0F), /* 0F */
static_cast<char>(0x10), /* 10 */
static_cast<char>(0x11), /* 11 */
static_cast<char>(0x12), /* 12 */
static_cast<char>(0x13), /* 13 */
static_cast<char>(0x9D), /* 14 */
static_cast<char>(0x85), /* 15 */
static_cast<char>(0x08), /* 16 */
static_cast<char>(0x87), /* 17 */
static_cast<char>(0x18), /* 18 */
static_cast<char>(0x19), /* 19 */
static_cast<char>(0x92), /* 1A */
static_cast<char>(0x8F), /* 1B */
static_cast<char>(0x1C), /* 1C */
static_cast<char>(0x1D), /* 1D */
static_cast<char>(0x1E), /* 1E */
static_cast<char>(0x1F), /* 1F */
static_cast<char>(0x80), /* 20 */
static_cast<char>(0x81), /* 21 */
static_cast<char>(0x82), /* 22 */
static_cast<char>(0x83), /* 23 */
static_cast<char>(0x84), /* 24 */
static_cast<char>(0x0A), /* 25 */
static_cast<char>(0x17), /* 26 */
static_cast<char>(0x1B), /* 27 */
static_cast<char>(0x88), /* 28 */
static_cast<char>(0x89), /* 29 */
static_cast<char>(0x8A), /* 2A */
static_cast<char>(0x8B), /* 2B */
static_cast<char>(0x8C), /* 2C */
static_cast<char>(0x05), /* 2D */
static_cast<char>(0x06), /* 2E */
static_cast<char>(0x07), /* 2F */
static_cast<char>(0x90), /* 30 */
static_cast<char>(0x91), /* 31 */
static_cast<char>(0x16), /* 32 */
static_cast<char>(0x93), /* 33 */
static_cast<char>(0x94), /* 34 */
static_cast<char>(0x95), /* 35 */
static_cast<char>(0x96), /* 36 */
static_cast<char>(0x04), /* 37 */
static_cast<char>(0x98), /* 38 */
static_cast<char>(0x99), /* 39 */
static_cast<char>(0x9A), /* 3A */
static_cast<char>(0x9B), /* 3B */
static_cast<char>(0x14), /* 3C */
static_cast<char>(0x15), /* 3D */
static_cast<char>(0x9E), /* 3E */
static_cast<char>(0x1A), /* 3F */
static_cast<char>(0x20), /* 40 */
static_cast<char>(0xA0), /* 41 */
static_cast<char>(0xE2), /* 42 */
static_cast<char>(0xE4), /* 43 */
static_cast<char>(0xE0), /* 44 */
static_cast<char>(0xE1), /* 45 */
static_cast<char>(0xE3), /* 46 */
static_cast<char>(0xE5), /* 47 */
static_cast<char>(0xE7), /* 48 */
static_cast<char>(0xF1), /* 49 */
static_cast<char>(0xA2), /* 4A */
static_cast<char>(0x2E), /* 4B */
static_cast<char>(0x3C), /* 4C */
static_cast<char>(0x28), /* 4D */
static_cast<char>(0x2B), /* 4E */
static_cast<char>(0x7C), /* 4F */
static_cast<char>(0x26), /* 50 */
static_cast<char>(0xE9), /* 51 */
static_cast<char>(0xEA), /* 52 */
static_cast<char>(0xEB), /* 53 */
static_cast<char>(0xE8), /* 54 */
static_cast<char>(0xED), /* 55 */
static_cast<char>(0xEE), /* 56 */
static_cast<char>(0xEF), /* 57 */
static_cast<char>(0xEC), /* 58 */
static_cast<char>(0xDF), /* 59 */
static_cast<char>(0x21), /* 5A */
static_cast<char>(0x24), /* 5B */
static_cast<char>(0x2A), /* 5C */
static_cast<char>(0x29), /* 5D */
static_cast<char>(0x3B), /* 5E */
static_cast<char>(0x5E), /* 5F */
static_cast<char>(0x2D), /* 60 */
static_cast<char>(0x2F), /* 61 */
static_cast<char>(0xC2), /* 62 */
static_cast<char>(0xC4), /* 63 */
static_cast<char>(0xC0), /* 64 */
static_cast<char>(0xC1), /* 65 */
static_cast<char>(0xC3), /* 66 */
static_cast<char>(0xC5), /* 67 */
static_cast<char>(0xC7), /* 68 */
static_cast<char>(0xD1), /* 69 */
static_cast<char>(0xA6), /* 6A */
static_cast<char>(0x2C), /* 6B */
static_cast<char>(0x25), /* 6C */
static_cast<char>(0x5F), /* 6D */
static_cast<char>(0x3E), /* 6E */
static_cast<char>(0x3F), /* 6F */
static_cast<char>(0xF8), /* 70 */
static_cast<char>(0xC9), /* 71 */
static_cast<char>(0xCA), /* 72 */
static_cast<char>(0xCB), /* 73 */
static_cast<char>(0xC8), /* 74 */
static_cast<char>(0xCD), /* 75 */
static_cast<char>(0xCE), /* 76 */
static_cast<char>(0xCF), /* 77 */
static_cast<char>(0xCC), /* 78 */
static_cast<char>(0x60), /* 79 */
static_cast<char>(0x3A), /* 7A */
static_cast<char>(0x23), /* 7B */
static_cast<char>(0x40), /* 7C */
static_cast<char>(0x27), /* 7D */
static_cast<char>(0x3D), /* 7E */
static_cast<char>(0x22), /* 7F */
static_cast<char>(0xD8), /* 80 */
static_cast<char>(0x61), /* 81 */
static_cast<char>(0x62), /* 82 */
static_cast<char>(0x63), /* 83 */
static_cast<char>(0x64), /* 84 */
static_cast<char>(0x65), /* 85 */
static_cast<char>(0x66), /* 86 */
static_cast<char>(0x67), /* 87 */
static_cast<char>(0x68), /* 88 */
static_cast<char>(0x69), /* 89 */
static_cast<char>(0xAB), /* 8A */
static_cast<char>(0xBB), /* 8B */
static_cast<char>(0xF0), /* 8C */
static_cast<char>(0xFD), /* 8D */
static_cast<char>(0xFE), /* 8E */
static_cast<char>(0xB1), /* 8F */
static_cast<char>(0xB0), /* 90 */
static_cast<char>(0x6A), /* 91 */
static_cast<char>(0x6B), /* 92 */
static_cast<char>(0x6C), /* 93 */
static_cast<char>(0x6D), /* 94 */
static_cast<char>(0x6E), /* 95 */
static_cast<char>(0x6F), /* 96 */
static_cast<char>(0x70), /* 97 */
static_cast<char>(0x71), /* 98 */
static_cast<char>(0x72), /* 99 */
static_cast<char>(0xAA), /* 9A */
static_cast<char>(0xBA), /* 9B */
static_cast<char>(0xE6), /* 9C */
static_cast<char>(0xB8), /* 9D */
static_cast<char>(0xC6), /* 9E */
static_cast<char>(0xA4), /* 9F */
static_cast<char>(0xB5), /* A0 */
static_cast<char>(0x7E), /* A1 */
static_cast<char>(0x73), /* A2 */
static_cast<char>(0x74), /* A3 */
static_cast<char>(0x75), /* A4 */
static_cast<char>(0x76), /* A5 */
static_cast<char>(0x77), /* A6 */
static_cast<char>(0x78), /* A7 */
static_cast<char>(0x79), /* A8 */
static_cast<char>(0x7A), /* A9 */
static_cast<char>(0xA1), /* AA */
static_cast<char>(0xBF), /* AB */
static_cast<char>(0xD0), /* AC */
static_cast<char>(0x5B), /* AD */
static_cast<char>(0xDE), /* AE */
static_cast<char>(0xAE), /* AF */
static_cast<char>(0xAC), /* B0 */
static_cast<char>(0xA3), /* B1 */
static_cast<char>(0xA5), /* B2 */
static_cast<char>(0xB7), /* B3 */
static_cast<char>(0xA9), /* B4 */
static_cast<char>(0xA7), /* B5 */
static_cast<char>(0xB6), /* B6 */
static_cast<char>(0xBC), /* B7 */
static_cast<char>(0xBD), /* B8 */
static_cast<char>(0xBE), /* B9 */
static_cast<char>(0xDD), /* BA */
static_cast<char>(0xA8), /* BB */
static_cast<char>(0xAF), /* BC */
static_cast<char>(0x5D), /* BD */
static_cast<char>(0xB4), /* BE */
static_cast<char>(0xD7), /* BF */
static_cast<char>(0x7B), /* C0 */
static_cast<char>(0x41), /* C1 */
static_cast<char>(0x42), /* C2 */
static_cast<char>(0x43), /* C3 */
static_cast<char>(0x44), /* C4 */
static_cast<char>(0x45), /* C5 */
static_cast<char>(0x46), /* C6 */
static_cast<char>(0x47), /* C7 */
static_cast<char>(0x48), /* C8 */
static_cast<char>(0x49), /* C9 */
static_cast<char>(0xAD), /* CA */
static_cast<char>(0xF4), /* CB */
static_cast<char>(0xF6), /* CC */
static_cast<char>(0xF2), /* CD */
static_cast<char>(0xF3), /* CE */
static_cast<char>(0xF5), /* CF */
static_cast<char>(0x7D), /* D0 */
static_cast<char>(0x4A), /* D1 */
static_cast<char>(0x4B), /* D2 */
static_cast<char>(0x4C), /* D3 */
static_cast<char>(0x4D), /* D4 */
static_cast<char>(0x4E), /* D5 */
static_cast<char>(0x4F), /* D6 */
static_cast<char>(0x50), /* D7 */
static_cast<char>(0x51), /* D8 */
static_cast<char>(0x52), /* D9 */
static_cast<char>(0xB9), /* DA */
static_cast<char>(0xFB), /* DB */
static_cast<char>(0xFC), /* DC */
static_cast<char>(0xF9), /* DD */
static_cast<char>(0xFA), /* DE */
static_cast<char>(0xFF), /* DF */
static_cast<char>(0x5C), /* E0 */
static_cast<char>(0xF7), /* E1 */
static_cast<char>(0x53), /* E2 */
static_cast<char>(0x54), /* E3 */
static_cast<char>(0x55), /* E4 */
static_cast<char>(0x56), /* E5 */
static_cast<char>(0x57), /* E6 */
static_cast<char>(0x58), /* E7 */
static_cast<char>(0x59), /* E8 */
static_cast<char>(0x5A), /* E9 */
static_cast<char>(0xB2), /* EA */
static_cast<char>(0xD4), /* EB */
static_cast<char>(0xD6), /* EC */
static_cast<char>(0xD2), /* ED */
static_cast<char>(0xD3), /* EE */
static_cast<char>(0xD5), /* EF */
static_cast<char>(0x30), /* F0 */
static_cast<char>(0x31), /* F1 */
static_cast<char>(0x32), /* F2 */
static_cast<char>(0x33), /* F3 */
static_cast<char>(0x34), /* F4 */
static_cast<char>(0x35), /* F5 */
static_cast<char>(0x36), /* F6 */
static_cast<char>(0x37), /* F7 */
static_cast<char>(0x38), /* F8 */
static_cast<char>(0x39), /* F9 */
static_cast<char>(0xB3), /* FA */
static_cast<char>(0xDB), /* FB */
static_cast<char>(0xDC), /* FC */
static_cast<char>(0xD9), /* FD */
static_cast<char>(0xDA), /* FE */
static_cast<char>(0x9F), /* FF */
};
static const bool oldIllegal[256] = {
false, /* U+0000 */
false, /* U+0001 */
false, /* U+0002 */
false, /* U+0003 */
false, /* U+0004 */
false, /* U+0005 */
false, /* U+0006 */
false, /* U+0007 */
false, /* U+0008 */
false, /* U+0009 */
false, /* U+000A */
false, /* U+000B */
false, /* U+000C */
false, /* U+000D */
false, /* U+000E */
false, /* U+000F */
false, /* U+0010 */
false, /* U+0011 */
false, /* U+0012 */
false, /* U+0013 */
false, /* U+0014 */
false, /* U+0015 */
false, /* U+0016 */
false, /* U+0017 */
false, /* U+0018 */
false, /* U+0019 */
false, /* U+001A */
false, /* U+001B */
false, /* U+001C */
false, /* U+001D */
false, /* U+001E */
false, /* U+001F */
true, /* U+0020 */
true, /* U+0021 */
true, /* U+0022 */
true, /* U+0023 */
false, /* U+0024 */
true, /* U+0025 */
true, /* U+0026 */
true, /* U+0027 */
true, /* U+0028 */
true, /* U+0029 */
true, /* U+002A */
true, /* U+002B */
true, /* U+002C */
true, /* U+002D */
true, /* U+002E */
true, /* U+002F */
true, /* U+0030 */
true, /* U+0031 */
true, /* U+0032 */
true, /* U+0033 */
true, /* U+0034 */
true, /* U+0035 */
true, /* U+0036 */
true, /* U+0037 */
true, /* U+0038 */
true, /* U+0039 */
true, /* U+003A */
true, /* U+003B */
true, /* U+003C */
true, /* U+003D */
true, /* U+003E */
true, /* U+003F */
false, /* U+0040 */
true, /* U+0041 */
true, /* U+0042 */
true, /* U+0043 */
true, /* U+0044 */
true, /* U+0045 */
true, /* U+0046 */
true, /* U+0047 */
true, /* U+0048 */
true, /* U+0049 */
true, /* U+004A */
true, /* U+004B */
true, /* U+004C */
true, /* U+004D */
true, /* U+004E */
true, /* U+004F */
true, /* U+0050 */
true, /* U+0051 */
true, /* U+0052 */
true, /* U+0053 */
true, /* U+0054 */
true, /* U+0055 */
true, /* U+0056 */
true, /* U+0057 */
true, /* U+0058 */
true, /* U+0059 */
true, /* U+005A */
true, /* U+005B */
false, /* U+005C */
true, /* U+005D */
true, /* U+005E */
true, /* U+005F */
false, /* U+0060 */
true, /* U+0061 */
true, /* U+0062 */
true, /* U+0063 */
true, /* U+0064 */
true, /* U+0065 */
true, /* U+0066 */
true, /* U+0067 */
true, /* U+0068 */
true, /* U+0069 */
true, /* U+006A */
true, /* U+006B */
true, /* U+006C */
true, /* U+006D */
true, /* U+006E */
true, /* U+006F */
true, /* U+0070 */
true, /* U+0071 */
true, /* U+0072 */
true, /* U+0073 */
true, /* U+0074 */
true, /* U+0075 */
true, /* U+0076 */
true, /* U+0077 */
true, /* U+0078 */
true, /* U+0079 */
true, /* U+007A */
true, /* U+007B */
true, /* U+007C */
true, /* U+007D */
true, /* U+007E */
false, /* U+007F */
false, /* U+0080 */
false, /* U+0081 */
false, /* U+0082 */
false, /* U+0083 */
false, /* U+0084 */
false, /* U+0085 */
false, /* U+0086 */
false, /* U+0087 */
false, /* U+0088 */
false, /* U+0089 */
false, /* U+008A */
false, /* U+008B */
false, /* U+008C */
false, /* U+008D */
false, /* U+008E */
false, /* U+008F */
false, /* U+0090 */
false, /* U+0091 */
false, /* U+0092 */
false, /* U+0093 */
false, /* U+0094 */
false, /* U+0095 */
false, /* U+0096 */
false, /* U+0097 */
false, /* U+0098 */
false, /* U+0099 */
false, /* U+009A */
false, /* U+009B */
false, /* U+009C */
false, /* U+009D */
false, /* U+009E */
false, /* U+009F */
false, /* U+00A0 */
false, /* U+00A1 */
false, /* U+00A2 */
false, /* U+00A3 */
false, /* U+00A4 */
false, /* U+00A5 */
false, /* U+00A6 */
false, /* U+00A7 */
false, /* U+00A8 */
false, /* U+00A9 */
false, /* U+00AA */
false, /* U+00AB */
false, /* U+00AC */
false, /* U+00AD */
false, /* U+00AE */
false, /* U+00AF */
false, /* U+00B0 */
false, /* U+00B1 */
false, /* U+00B2 */
false, /* U+00B3 */
false, /* U+00B4 */
false, /* U+00B5 */
false, /* U+00B6 */
false, /* U+00B7 */
false, /* U+00B8 */
false, /* U+00B9 */
false, /* U+00BA */
false, /* U+00BB */
false, /* U+00BC */
false, /* U+00BD */
false, /* U+00BE */
false, /* U+00BF */
false, /* U+00C0 */
false, /* U+00C1 */
false, /* U+00C2 */
false, /* U+00C3 */
false, /* U+00C4 */
false, /* U+00C5 */
false, /* U+00C6 */
false, /* U+00C7 */
false, /* U+00C8 */
false, /* U+00C9 */
false, /* U+00CA */
false, /* U+00CB */
false, /* U+00CC */
false, /* U+00CD */
false, /* U+00CE */
false, /* U+00CF */
false, /* U+00D0 */
false, /* U+00D1 */
false, /* U+00D2 */
false, /* U+00D3 */
false, /* U+00D4 */
false, /* U+00D5 */
false, /* U+00D6 */
false, /* U+00D7 */
false, /* U+00D8 */
false, /* U+00D9 */
false, /* U+00DA */
false, /* U+00DB */
false, /* U+00DC */
false, /* U+00DD */
false, /* U+00DE */
false, /* U+00DF */
false, /* U+00E0 */
false, /* U+00E1 */
false, /* U+00E2 */
false, /* U+00E3 */
false, /* U+00E4 */
false, /* U+00E5 */
false, /* U+00E6 */
false, /* U+00E7 */
false, /* U+00E8 */
false, /* U+00E9 */
false, /* U+00EA */
false, /* U+00EB */
false, /* U+00EC */
false, /* U+00ED */
false, /* U+00EE */
false, /* U+00EF */
false, /* U+00F0 */
false, /* U+00F1 */
false, /* U+00F2 */
false, /* U+00F3 */
false, /* U+00F4 */
false, /* U+00F5 */
false, /* U+00F6 */
false, /* U+00F7 */
false, /* U+00F8 */
false, /* U+00F9 */
false, /* U+00FA */
false, /* U+00FB */
false, /* U+00FC */
false, /* U+00FD */
false, /* U+00FE */
false, /* U+00FF */
};

View File

@ -0,0 +1,427 @@
// © 2016 and later: Unicode, Inc. and others.
// License & terms of use: http://www.unicode.org/copyright.html
#include <stdio.h>
#include <string>
#include <stdlib.h>
#include <errno.h>
#include <string.h>
#include <iostream>
#include <fstream>
// We only use U8_* macros, which are entirely inline.
#include "unicode/utf8.h"
// This contains a codepage and ISO 14882:1998 illegality table.
// Use "make gen-table" to rebuild it.
#include "cptbl.h"
/**
* What is this?
*
* "This" is a preprocessor that makes an attempt to convert fully valid C++11 source code
* in utf-8 into something consumable by certain compilers (Solaris, xlC)
* which aren't quite standards compliant.
*
* - u"<unicode>" or u'<unicode>' gets converted to u"\uNNNN" or u'\uNNNN'
* - u8"<unicode>" gets converted to "\xAA\xBB\xCC\xDD" etc.
* (some compilers do not support the u8 prefix correctly.)
* - if the system is EBCDIC-based, that is used to correct the input characters.
*
* Usage:
* escapesrc infile.cpp outfile.cpp
* Normally this is invoked by the build stage, with a rule such as:
*
* _%.cpp: $(srcdir)/%.cpp
* @$(BINDIR)/escapesrc$(EXEEXT) $< $@
* %.o: _%.cpp
* $(COMPILE.cc) ... $@ $<
*
* In the Makefiles, SKIP_ESCAPING=YES is used to prevent escapesrc.cpp
* from being itself escaped.
*/
static const char
kSPACE = 0x20,
kTAB = 0x09,
kLF = 0x0A,
kCR = 0x0D;
// For convenience
# define cp1047_to_8859(c) cp1047_8859_1[c]
// Our app's name
std::string prog;
/**
* Give the usual 1-line documentation and exit
*/
void usage() {
fprintf(stderr, "%s: usage: %s infile.cpp outfile.cpp\n", prog.c_str(), prog.c_str());
}
/**
* Delete the output file (if any)
* We want to delete even if we didn't generate, because it might be stale.
*/
int cleanup(const std::string &outfile) {
const char *outstr = outfile.c_str();
if(outstr && *outstr) {
int rc = std::remove(outstr);
if(rc == 0) {
fprintf(stderr, "%s: deleted %s\n", prog.c_str(), outstr);
return 0;
} else {
if( errno == ENOENT ) {
return 0; // File did not exist - no error.
} else {
perror("std::remove");
return 1;
}
}
}
return 0;
}
/**
* Skip across any known whitespace.
* @param p startpoint
* @param e limit
* @return first non-whitespace char
*/
inline const char *skipws(const char *p, const char *e) {
for(;p<e;p++) {
switch(*p) {
case kSPACE:
case kTAB:
case kLF:
case kCR:
break;
default:
return p; // non ws
}
}
return p;
}
/**
* Append a byte, hex encoded
* @param outstr sstring to append to
* @param byte the byte to append
*/
void appendByte(std::string &outstr,
uint8_t byte) {
char tmp2[5];
snprintf(tmp2, sizeof(tmp2), "\\x%02X", 0xFF & static_cast<int>(byte));
outstr += tmp2;
}
/**
* Append the bytes from 'linestr' into outstr, with escaping
* @param outstr the output buffer
* @param linestr the input buffer
* @param pos in/out: the current char under consideration
* @param chars the number of chars to consider
* @return true on failure
*/
bool appendUtf8(std::string &outstr,
const std::string &linestr,
size_t &pos,
size_t chars) {
char tmp[9];
for(size_t i=0;i<chars;i++) {
tmp[i] = linestr[++pos];
}
tmp[chars] = 0;
unsigned int c;
sscanf(tmp, "%X", &c);
UChar32 ch = c & 0x1FFFFF;
// now to append \\x%% etc
uint8_t bytesNeeded = U8_LENGTH(ch);
if(bytesNeeded == 0) {
fprintf(stderr, "Illegal code point U+%X\n", ch);
return true;
}
uint8_t bytes[4];
uint8_t *s = bytes;
size_t i = 0;
U8_APPEND_UNSAFE(s, i, ch);
for(size_t t = 0; t<i; t++) {
appendByte(outstr, s[t]);
}
return false;
}
/**
* Fixup u8"x"
* @param linestr string to mutate. Already escaped into \u format.
* @param origpos beginning, points to 'u8"'
* @param pos end, points to "
* @return false for no-problem, true for failure!
*/
bool fixu8(std::string &linestr, size_t origpos, size_t &endpos) {
size_t pos = origpos + 3;
std::string outstr;
outstr += '\"'; // local encoding
for(;pos<endpos;pos++) {
char c = linestr[pos];
if(c == '\\') {
char c2 = linestr[++pos];
switch(c2) {
case '\'':
case '"':
#if (U_CHARSET_FAMILY == U_EBCDIC_FAMILY)
c2 = cp1047_to_8859(c2);
#endif
appendByte(outstr, c2);
break;
case 'u':
appendUtf8(outstr, linestr, pos, 4);
break;
case 'U':
appendUtf8(outstr, linestr, pos, 8);
break;
}
} else {
#if (U_CHARSET_FAMILY == U_EBCDIC_FAMILY)
c = cp1047_to_8859(c);
#endif
appendByte(outstr, c);
}
}
outstr += ('\"');
linestr.replace(origpos, (endpos-origpos+1), outstr);
return false; // OK
}
/**
* fix the u"x"/u'x'/u8"x" string at the position
* u8'x' is not supported, sorry.
* @param linestr the input string
* @param pos the position
* @return false = no err, true = had err
*/
bool fixAt(std::string &linestr, size_t pos) {
size_t origpos = pos;
if(linestr[pos] != 'u') {
fprintf(stderr, "Not a 'u'?");
return true;
}
pos++; // past 'u'
bool utf8 = false;
if(linestr[pos] == '8') { // u8"
utf8 = true;
pos++;
}
char quote = linestr[pos];
if(quote != '\'' && quote != '\"') {
fprintf(stderr, "Quote is '%c' - not sure what to do.\n", quote);
return true;
}
if(quote == '\'' && utf8) {
fprintf(stderr, "Cannot do u8'...'\n");
return true;
}
pos ++;
//printf("u%c…%c\n", quote, quote);
for(; pos < linestr.size(); pos++) {
if(linestr[pos] == quote) {
if(utf8) {
return fixu8(linestr, origpos, pos); // fix u8"..."
} else {
return false; // end of quote
}
}
if(linestr[pos] == '\\') {
pos++;
if(linestr[pos] == quote) continue; // quoted quote
if(linestr[pos] == 'u') continue; // for now ... unicode escape
if(linestr[pos] == '\\') continue;
// some other escape… ignore
} else {
size_t old_pos = pos;
int32_t i = pos;
#if (U_CHARSET_FAMILY == U_EBCDIC_FAMILY)
// mogrify 1-4 bytes from 1047 'back' to utf-8
char old_byte = linestr[pos];
linestr[pos] = cp1047_to_8859(linestr[pos]);
// how many more?
int32_t trail = U8_COUNT_TRAIL_BYTES(linestr[pos]);
for(size_t pos2 = pos+1; trail>0; pos2++,trail--) {
linestr[pos2] = cp1047_to_8859(linestr[pos2]);
if(linestr[pos2] == 0x0A) {
linestr[pos2] = 0x85; // NL is ambiguous here
}
}
#endif
// Proceed to decode utf-8
const uint8_t* s = reinterpret_cast<const uint8_t*>(linestr.c_str());
int32_t length = linestr.size();
UChar32 c;
if(U8_IS_SINGLE((uint8_t)s[i]) && oldIllegal[s[i]]) {
#if (U_CHARSET_FAMILY == U_EBCDIC_FAMILY)
linestr[pos] = old_byte; // put it back
#endif
continue; // single code point not previously legal for \u escaping
}
// otherwise, convert it to \u / \U
{
U8_NEXT(s, i, length, c);
}
if(c<0) {
fprintf(stderr, "Illegal utf-8 sequence at Column: %d\n", static_cast<int>(old_pos));
fprintf(stderr, "Line: >>%s<<\n", linestr.c_str());
return true;
}
size_t seqLen = (i-pos);
//printf("U+%04X pos %d [len %d]\n", c, pos, seqLen);fflush(stdout);
char newSeq[20];
if( c <= 0xFFFF) {
snprintf(newSeq, sizeof(newSeq), "\\u%04X", c);
} else {
snprintf(newSeq, sizeof(newSeq), "\\U%08X", c);
}
linestr.replace(pos, seqLen, newSeq);
pos += strlen(newSeq) - 1;
}
}
return false;
}
/**
* Fixup an entire line
* false = no err
* true = had err
* @param no the line number (not used)
* @param linestr the string to fix
* @return true if any err, else false
*/
bool fixLine(int /*no*/, std::string &linestr) {
const char *line = linestr.c_str();
size_t len = linestr.size();
// no u' in the line?
if(!strstr(line, "u'") && !strstr(line, "u\"") && !strstr(line, "u8\"")) {
return false; // Nothing to do. No u' or u" detected
}
// start from the end and find all u" cases
size_t pos = len = linestr.size();
if(len>INT32_MAX/2) {
return true;
}
while((pos>0) && (pos = linestr.rfind("u\"", pos)) != std::string::npos) {
//printf("found doublequote at %d\n", pos);
if(fixAt(linestr, pos)) return true;
if(pos == 0) break;
pos--;
}
// reset and find all u' cases
pos = len = linestr.size();
while((pos>0) && (pos = linestr.rfind("u'", pos)) != std::string::npos) {
//printf("found singlequote at %d\n", pos);
if(fixAt(linestr, pos)) return true;
if(pos == 0) break;
pos--;
}
// reset and find all u8" cases
pos = len = linestr.size();
while((pos>0) && (pos = linestr.rfind("u8\"", pos)) != std::string::npos) {
if(fixAt(linestr, pos)) return true;
if(pos == 0) break;
pos--;
}
//fprintf(stderr, "%d - fixed\n", no);
return false;
}
/**
* Convert a whole file
* @param infile
* @param outfile
* @return 1 on err, 0 otherwise
*/
int convert(const std::string &infile, const std::string &outfile) {
fprintf(stderr, "escapesrc: %s -> %s\n", infile.c_str(), outfile.c_str());
std::ifstream inf;
inf.open(infile.c_str(), std::ios::in);
if(!inf.is_open()) {
fprintf(stderr, "%s: could not open input file %s\n", prog.c_str(), infile.c_str());
cleanup(outfile);
return 1;
}
std::ofstream outf;
outf.open(outfile.c_str(), std::ios::out);
if(!outf.is_open()) {
fprintf(stderr, "%s: could not open output file %s\n", prog.c_str(), outfile.c_str());
return 1;
}
// TODO: any platform variations of #line?
outf << "#line 1 \"" << infile << "\"" << '\n';
int no = 0;
std::string linestr;
while( getline( inf, linestr)) {
no++;
if(fixLine(no, linestr)) {
goto fail;
}
outf << linestr << '\n';
}
if(inf.eof()) {
return 0;
}
fail:
outf.close();
fprintf(stderr, "%s:%d: Fixup failed by %s\n", infile.c_str(), no, prog.c_str());
cleanup(outfile);
return 1;
}
/**
* Main function
*/
int main(int argc, const char *argv[]) {
prog = argv[0];
if(argc != 3) {
usage();
return 1;
}
std::string infile = argv[1];
std::string outfile = argv[2];
return convert(infile, outfile);
}

View File

@ -0,0 +1,17 @@
// © 2016 and later: Unicode, Inc. and others.
// License & terms of use: http://www.unicode.org/copyright.html
u"sa\u0127\u0127a";
u'\u6587';
u"\U000219F2";
u"\u039C\u03C5\u03C3\u03C4\u03AE\u03C1\u03B9\u03BF";
u"sa\u0127\u0127a";
u'\u6587'; u"\U000219F2";
"\x20\xCC\x81";
"\xCC\x88\x20";
"\x73\x61\xC4\xA7\xC4\xA7\x61";
"\xE6\x96\x87";
"\xF0\xA1\xA7\xB2";
"\x73\x61\xC4\xA7\xC4\xA7\x61";

View File

@ -0,0 +1,83 @@
// © 2016 and later: Unicode, Inc. and others.
// License & terms of use: http://www.unicode.org/copyright.html
#include "unicode/utypes.h"
#include "unicode/ucnv.h"
#include "unicode/uniset.h"
#include <stdio.h>
using icu::LocalUConverterPointer;
using icu::UnicodeSet;
static const char *kConverter = "ibm-1047";
int main(int argc, const char *argv[]) {
printf("// %s\n", U_COPYRIGHT_STRING);
printf("// generated by tblgen. You weren't going to edit it by hand, were you?\n");
printf("\n");
UErrorCode status = U_ZERO_ERROR;
LocalUConverterPointer cnv(ucnv_open(kConverter, &status));
if(U_FAILURE(status)) {
fprintf(stderr, "Failed to open %s: %s\n", kConverter, u_errorName(status));
return 1;
}
printf("static const char cp1047_8859_1[256] = { \n");
for(int i=0x00; i<0x100; i++) {
char cp1047[1];
cp1047[0] = i;
char16_t u[1];
char16_t *target = u;
const char *source = cp1047;
ucnv_toUnicode(cnv.getAlias(), &target, u+1, &source, cp1047+1, nullptr, true, &status);
if(U_FAILURE(status)) {
fprintf(stderr, "Conversion failure at #%X: %s\n", i, u_errorName(status));
return 2;
}
printf(" (char)0x%02X, /* %02X */\n", u[0], i);
}
printf("};\n\n");
//
// UnicodeSet oldIllegal("[:print:]", status); // [a-zA-Z0-9_}{#)(><%:;.?*+-/^&|~!=,\\u005b\\u005d\\u005c]", status);
UnicodeSet oldIllegal("[0-9 a-z A-Z "
"_ \\{ \\} \\[ \\] # \\( \\) < > % \\: ; . "
"? * + \\- / \\^ \\& | ~ ! = , \\ \" ' ]", status);
/*
http://www.lirmm.fr/~ducour/Doc-objets/ISO+IEC+14882-1998.pdf ( note: 1998 ) page 10, section 2.2 says:
1 The basic source character set consists of 96 characters: the space character, the control characters repre- 15)
senting horizontal tab, vertical tab, form feed, and new-line, plus the following 91 graphical characters:
a b c d e f g h i j k l m n opqrstuvwxyz
A B C D E F G H I J K L M N OPQRSTUVWXYZ
0 12 3 4 5 6 7 8 9
_ { } [ ] # ( ) < > % : ; . ?*+-/^&|~!=,\"
2 The universal-character-name construct provides a way to name other characters. hex-quad:
hexadecimal-digit hexadecimal-digit hexadecimal-digit hexadecimal-digit
universal-character-name: \u hex-quad
\U hex-quad hex-quad
The character designated by the universal-character-name \UNNNNNNNN is that character whose character short name in ISO/IEC 10646 is NNNNNNNN; the character designated by the universal-character-name \uNNNN is that character whose character short name in ISO/IEC 10646 is 0000NNNN. If the hexadecimal value for a universal character name is less than 0x20 or in the range 0x7F-0x9F (inclusive), or if the uni- versal character name designates a character in the basic source character set, then the program is ill- formed.
So basically: printable ASCII plus 0x00-0x1F, 0x7F-0x9F, was all illegal.
Some discussion at http://unicode.org/mail-arch/unicode-ml/y2003-m10/0471.html
*/
printf("static const bool oldIllegal[256] = { \n");
for(char16_t i=0x00; i<0x100;i++) {
printf(" %s, /* U+%04X */\n",
(oldIllegal.contains(i))?" true":"false",
i);
}
printf("};\n\n");
return 0;
}

View File

@ -0,0 +1,5 @@
// © 2016 and later: Unicode, Inc. and others.
// License & terms of use: http://www.unicode.org/copyright.html
// This is a source file with no changes needed in it.
// In fact, the only non-ASCII character is the comment line at top.

View File

@ -0,0 +1,17 @@
// © 2016 and later: Unicode, Inc. and others.
// License & terms of use: http://www.unicode.org/copyright.html
u"saħħa";
u'';
u"𡧲";
u"Μυστήριο";
u"saħħa";
u''; u"𡧲";
u8" \u0301";
u8"\u0308 ";
u8"saħħa";
u8"";
u8"𡧲";
u8"saħ\u0127a";