Add UTF16 and Shift-JIS decode/encoders.

May not entirely be perfect, not enough error checking.
Shift-JIS encoder is not complete but does most things.
This commit is contained in:
Unknown W. Brackets committed 2013-07-20 21:46:18 -07:00
1 parent d70ac3bba5
commit f3da7ce00a
3 files changed
+214

No files matched your search

+133
View File
@@ -0,0 +1,133 @@
#pragma once
#include "base/basictypes.h"
// Warning: decodes/encodes JIS, not Unicode.
// Use a table to map.
struct ShiftJIS {
static const uint32_t INVALID = (uint32_t) -1;
ShiftJIS(const char *c) : c_(c), index_(0) {}
uint32_t next() {
uint32_t j = (uint8_t)c_[index_++];
int row;
bool emojiAdjust = false;
switch (j >> 4) {
case 0x8:
if (j == 0x80) {
return INVALID;
}
// Intentional fall-through.
case 0x9:
case 0xE:
row = ((j & 0x3F) << 1) - 0x01;
break;
case 0xF:
emojiAdjust = true;
if (j < 0xF4) {
row = ((j & 0x7F) << 1) - 0x59;
} else if (j < 0xFD) {
row = ((j & 0x7F) << 1) - 0x1B;
} else {
return j;
}
break;
// Anything else (i.e. <= 0x7x, 0xAx, 0xBx, 0xCx, and 0xDx) is JIS X 0201, return directly.
default:
return j;
}
// Okay, if we didn't return, it's time for the second byte (the cell.)
j = (uint8_t)c_[index_++];
// Not a valid second byte.
if (j < 0x40 || j == 0x7F || j >= 0xFD) {
return INVALID;
}
if (j >= 0x9F) {
// This range means the row was even.
++row;
j -= 0x7E;
} else {
if (j >= 0x80) {
j -= 0x20;
} else {
// Yuck. They wrapped around 0x7F, so we subtract one less.
j -= 0x20 - 1;
}
if (emojiAdjust) {
// These are shoved in where they'll fit.
if (row == 0x87) {
// First byte was 0xF0.
row = 0x81;
} else if (row == 0x8B) {
// First byte was 0xF2.
row = 0x85;
} else if (row == 0xCD) {
// First byte was 0xF4.
row = 0x8F;
}
}
}
// j is already the cell + 0x20.
return ((row + 0x20) << 8) | j;
}
bool end() const {
return c_[index_] == 0;
}
int length() const {
int len = 0;
for (ShiftJIS dec(c_); !dec.end(); dec.next())
++len;
return len;
}
int byteIndex() const {
return index_;
}
static int encode(char *dest, uint32_t j) {
int row = (j >> 8) - 0x20;
int offsetCell = j & 0xFF;
// JIS X 0201.
if ((j & ~0xFF) == 0) {
*dest = j;
return 1;
}
if (row < 0x3F) {
*dest++ = 0x80 + ((row + 1) >> 1);
} else if (row < 0x5F) {
// Reduce by 0x40 to account for the above range.
*dest++ = 0xE0 + ((row - 0x40 + 1) >> 1);
} else if (row >= 0x80) {
// TODO
}
if (row & 1) {
if (offsetCell < 0x60) {
// Subtract one to shift around 0x7F.
*dest++ = offsetCell + 0x20 - 1;
} else {
*dest++ = offsetCell + 0x20;
}
} else {
*dest++ = offsetCell + 0x7E;
}
return 2;
}
private:
const char *c_;
int index_;
};
+73
View File
@@ -0,0 +1,73 @@
#pragma once
#include "base/basictypes.h"
#if defined(__LITTLE_ENDIAN__)
#define UTF16_IS_LITTLE_ENDIAN 1
#elif defined(__BIG_ENDIAN__)
#define UTF16_IS_LITTLE_ENDIAN 0
#else
// Should optimize out.
#define UTF16_IS_LITTLE_ENDIAN (*(const uint16_t *)"\0\xff" >= 0x100)
#endif
template <bool is_little>
uint16_t UTF16_Swap(uint16_t u) {
if (is_little) {
return UTF16_IS_LITTLE_ENDIAN ? u : swap16(u);
} else {
return UTF16_IS_LITTLE_ENDIAN ? swap16(u) : u;
}
}
template <bool is_little>
struct UTF16_Type {
public:
static const uint32_t INVALID = (uint32_t)-1;
UTF16_Type(const uint16_t *c) : c_(c), index_(0) {}
uint32_t next() {
const uint32_t u = UTF16_Swap<is_little>(c_[index_++]);
// Surrogate pair. UTF-16 is so simple. We assume it's valid.
if ((u & 0xD800) == 0xD800) {
return 0x10000 + (((u & 0x3FF) << 10) | (UTF16_Swap<is_little>(c_[index_++]) & 0x3FF));
}
return u;
}
bool end() const {
return c_[index_] == 0;
}
int length() const {
int len = 0;
for (UTF16_Type<is_little> dec(c_); !dec.end(); dec.next())
++len;
return len;
}
int byteIndex() const {
return index_;
}
static int encode(uint16_t *dest, uint32_t u) {
if (u >= 0x10000) {
u -= 0x10000;
*dest = UTF16_Swap<is_little>(0xD800 + ((u >> 10) & 0x3FF));
*dest = UTF16_Swap<is_little>(0xDC00 + ((u >> 0) & 0x3FF));
return 2;
} else {
*dest = UTF16_Swap<is_little>((uint16_t)u);
return 1;
}
}
private:
const uint16_t *c_;
int index_;
};
typedef UTF16_Type<true> UTF16LE;
typedef UTF16_Type<false> UTF16BE;
+8
View File
@@ -17,10 +17,12 @@
#include "base/basictypes.h"
uint32_t u8_nextchar(const char *s, int *i);
int u8_wc_toutf8(char *dest, uint32_t ch);
int u8_strlen(const char *s);
class UTF8 {
public:
static const uint32_t INVALID = (uint32_t)-1;
UTF8(const char *c) : c_(c), index_(0) {}
bool end() const { return c_[index_] == 0; }
uint32_t next() {
@@ -29,6 +31,12 @@ public:
int length() const {
return u8_strlen(c_);
}
int byteIndex() const {
return index_;
}
static int encode(char *dest, uint32_t ch) {
return u8_wc_toutf8(dest, ch);
}
private:
const char *c_;