From f3da7ce00a7e5153c959ceb5c421a038b02b1474 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 20 Jul 2013 21:46:18 -0700 Subject: [PATCH] Add UTF16 and Shift-JIS decode/encoders. May not entirely be perfect, not enough error checking. Shift-JIS encoder is not complete but does most things. --- util/text/shiftjis.h | 133 +++++++++++++++++++++++++++++++++++++++++++ util/text/utf16.h | 73 ++++++++++++++++++++++++ util/text/utf8.h | 8 +++ 3 files changed, 214 insertions(+) create mode 100644 util/text/shiftjis.h create mode 100644 util/text/utf16.h diff --git a/util/text/shiftjis.h b/util/text/shiftjis.h new file mode 100644 index 0000000000..72f9438c48 --- /dev/null +++ b/util/text/shiftjis.h @@ -0,0 +1,133 @@ +#pragma once + +#include "base/basictypes.h" + +// Warning: decodes/encodes JIS, not Unicode. +// Use a table to map. +struct ShiftJIS { + static const uint32_t INVALID = (uint32_t) -1; + + ShiftJIS(const char *c) : c_(c), index_(0) {} + + uint32_t next() { + uint32_t j = (uint8_t)c_[index_++]; + + int row; + bool emojiAdjust = false; + switch (j >> 4) { + case 0x8: + if (j == 0x80) { + return INVALID; + } + // Intentional fall-through. + case 0x9: + case 0xE: + row = ((j & 0x3F) << 1) - 0x01; + break; + + case 0xF: + emojiAdjust = true; + if (j < 0xF4) { + row = ((j & 0x7F) << 1) - 0x59; + } else if (j < 0xFD) { + row = ((j & 0x7F) << 1) - 0x1B; + } else { + return j; + } + break; + + // Anything else (i.e. <= 0x7x, 0xAx, 0xBx, 0xCx, and 0xDx) is JIS X 0201, return directly. + default: + return j; + } + + // Okay, if we didn't return, it's time for the second byte (the cell.) + j = (uint8_t)c_[index_++]; + // Not a valid second byte. + if (j < 0x40 || j == 0x7F || j >= 0xFD) { + return INVALID; + } + + if (j >= 0x9F) { + // This range means the row was even. + ++row; + j -= 0x7E; + } else { + if (j >= 0x80) { + j -= 0x20; + } else { + // Yuck. They wrapped around 0x7F, so we subtract one less. + j -= 0x20 - 1; + } + + if (emojiAdjust) { + // These are shoved in where they'll fit. + if (row == 0x87) { + // First byte was 0xF0. + row = 0x81; + } else if (row == 0x8B) { + // First byte was 0xF2. + row = 0x85; + } else if (row == 0xCD) { + // First byte was 0xF4. + row = 0x8F; + } + } + } + + // j is already the cell + 0x20. + return ((row + 0x20) << 8) | j; + } + + bool end() const { + return c_[index_] == 0; + } + + int length() const { + int len = 0; + for (ShiftJIS dec(c_); !dec.end(); dec.next()) + ++len; + return len; + } + + int byteIndex() const { + return index_; + } + + static int encode(char *dest, uint32_t j) { + int row = (j >> 8) - 0x20; + int offsetCell = j & 0xFF; + + // JIS X 0201. + if ((j & ~0xFF) == 0) { + *dest = j; + return 1; + } + + if (row < 0x3F) { + *dest++ = 0x80 + ((row + 1) >> 1); + } else if (row < 0x5F) { + // Reduce by 0x40 to account for the above range. + *dest++ = 0xE0 + ((row - 0x40 + 1) >> 1); + } else if (row >= 0x80) { + // TODO + } + + if (row & 1) { + if (offsetCell < 0x60) { + // Subtract one to shift around 0x7F. + *dest++ = offsetCell + 0x20 - 1; + } else { + *dest++ = offsetCell + 0x20; + } + } else { + *dest++ = offsetCell + 0x7E; + } + + return 2; + } + +private: + const char *c_; + int index_; +}; \ No newline at end of file diff --git a/util/text/utf16.h b/util/text/utf16.h new file mode 100644 index 0000000000..e84ca82079 --- /dev/null +++ b/util/text/utf16.h @@ -0,0 +1,73 @@ +#pragma once + +#include "base/basictypes.h" + +#if defined(__LITTLE_ENDIAN__) +#define UTF16_IS_LITTLE_ENDIAN 1 +#elif defined(__BIG_ENDIAN__) +#define UTF16_IS_LITTLE_ENDIAN 0 +#else +// Should optimize out. +#define UTF16_IS_LITTLE_ENDIAN (*(const uint16_t *)"\0\xff" >= 0x100) +#endif + +template +uint16_t UTF16_Swap(uint16_t u) { + if (is_little) { + return UTF16_IS_LITTLE_ENDIAN ? u : swap16(u); + } else { + return UTF16_IS_LITTLE_ENDIAN ? swap16(u) : u; + } +} + +template +struct UTF16_Type { +public: + static const uint32_t INVALID = (uint32_t)-1; + + UTF16_Type(const uint16_t *c) : c_(c), index_(0) {} + + uint32_t next() { + const uint32_t u = UTF16_Swap(c_[index_++]); + + // Surrogate pair. UTF-16 is so simple. We assume it's valid. + if ((u & 0xD800) == 0xD800) { + return 0x10000 + (((u & 0x3FF) << 10) | (UTF16_Swap(c_[index_++]) & 0x3FF)); + } + return u; + } + + bool end() const { + return c_[index_] == 0; + } + + int length() const { + int len = 0; + for (UTF16_Type dec(c_); !dec.end(); dec.next()) + ++len; + return len; + } + + int byteIndex() const { + return index_; + } + + static int encode(uint16_t *dest, uint32_t u) { + if (u >= 0x10000) { + u -= 0x10000; + *dest = UTF16_Swap(0xD800 + ((u >> 10) & 0x3FF)); + *dest = UTF16_Swap(0xDC00 + ((u >> 0) & 0x3FF)); + return 2; + } else { + *dest = UTF16_Swap((uint16_t)u); + return 1; + } + } + +private: + const uint16_t *c_; + int index_; +}; + +typedef UTF16_Type UTF16LE; +typedef UTF16_Type UTF16BE; diff --git a/util/text/utf8.h b/util/text/utf8.h index bd320979f8..5d46a67648 100644 --- a/util/text/utf8.h +++ b/util/text/utf8.h @@ -17,10 +17,12 @@ #include "base/basictypes.h" uint32_t u8_nextchar(const char *s, int *i); +int u8_wc_toutf8(char *dest, uint32_t ch); int u8_strlen(const char *s); class UTF8 { public: + static const uint32_t INVALID = (uint32_t)-1; UTF8(const char *c) : c_(c), index_(0) {} bool end() const { return c_[index_] == 0; } uint32_t next() { @@ -29,6 +31,12 @@ public: int length() const { return u8_strlen(c_); } + int byteIndex() const { + return index_; + } + static int encode(char *dest, uint32_t ch) { + return u8_wc_toutf8(dest, ch); + } private: const char *c_;