mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-05 10:36:17 +02:00
Add UTF16 and Shift-JIS decode/encoders.
May not entirely be perfect, not enough error checking. Shift-JIS encoder is not complete but does most things.
This commit is contained in:
1 parent
d70ac3bba5
commit
f3da7ce00a
3 files changed
+214
No files matched your search
@@ -0,0 +1,133 @@
|
||||
#pragma once
|
||||
|
||||
#include "base/basictypes.h"
|
||||
|
||||
// Warning: decodes/encodes JIS, not Unicode.
|
||||
// Use a table to map.
|
||||
struct ShiftJIS {
|
||||
static const uint32_t INVALID = (uint32_t) -1;
|
||||
|
||||
ShiftJIS(const char *c) : c_(c), index_(0) {}
|
||||
|
||||
uint32_t next() {
|
||||
uint32_t j = (uint8_t)c_[index_++];
|
||||
|
||||
int row;
|
||||
bool emojiAdjust = false;
|
||||
switch (j >> 4) {
|
||||
case 0x8:
|
||||
if (j == 0x80) {
|
||||
return INVALID;
|
||||
}
|
||||
// Intentional fall-through.
|
||||
case 0x9:
|
||||
case 0xE:
|
||||
row = ((j & 0x3F) << 1) - 0x01;
|
||||
break;
|
||||
|
||||
case 0xF:
|
||||
emojiAdjust = true;
|
||||
if (j < 0xF4) {
|
||||
row = ((j & 0x7F) << 1) - 0x59;
|
||||
} else if (j < 0xFD) {
|
||||
row = ((j & 0x7F) << 1) - 0x1B;
|
||||
} else {
|
||||
return j;
|
||||
}
|
||||
break;
|
||||
|
||||
// Anything else (i.e. <= 0x7x, 0xAx, 0xBx, 0xCx, and 0xDx) is JIS X 0201, return directly.
|
||||
default:
|
||||
return j;
|
||||
}
|
||||
|
||||
// Okay, if we didn't return, it's time for the second byte (the cell.)
|
||||
j = (uint8_t)c_[index_++];
|
||||
// Not a valid second byte.
|
||||
if (j < 0x40 || j == 0x7F || j >= 0xFD) {
|
||||
return INVALID;
|
||||
}
|
||||
|
||||
if (j >= 0x9F) {
|
||||
// This range means the row was even.
|
||||
++row;
|
||||
j -= 0x7E;
|
||||
} else {
|
||||
if (j >= 0x80) {
|
||||
j -= 0x20;
|
||||
} else {
|
||||
// Yuck. They wrapped around 0x7F, so we subtract one less.
|
||||
j -= 0x20 - 1;
|
||||
}
|
||||
|
||||
if (emojiAdjust) {
|
||||
// These are shoved in where they'll fit.
|
||||
if (row == 0x87) {
|
||||
// First byte was 0xF0.
|
||||
row = 0x81;
|
||||
} else if (row == 0x8B) {
|
||||
// First byte was 0xF2.
|
||||
row = 0x85;
|
||||
} else if (row == 0xCD) {
|
||||
// First byte was 0xF4.
|
||||
row = 0x8F;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// j is already the cell + 0x20.
|
||||
return ((row + 0x20) << 8) | j;
|
||||
}
|
||||
|
||||
bool end() const {
|
||||
return c_[index_] == 0;
|
||||
}
|
||||
|
||||
int length() const {
|
||||
int len = 0;
|
||||
for (ShiftJIS dec(c_); !dec.end(); dec.next())
|
||||
++len;
|
||||
return len;
|
||||
}
|
||||
|
||||
int byteIndex() const {
|
||||
return index_;
|
||||
}
|
||||
|
||||
static int encode(char *dest, uint32_t j) {
|
||||
int row = (j >> 8) - 0x20;
|
||||
int offsetCell = j & 0xFF;
|
||||
|
||||
// JIS X 0201.
|
||||
if ((j & ~0xFF) == 0) {
|
||||
*dest = j;
|
||||
return 1;
|
||||
}
|
||||
|
||||
if (row < 0x3F) {
|
||||
*dest++ = 0x80 + ((row + 1) >> 1);
|
||||
} else if (row < 0x5F) {
|
||||
// Reduce by 0x40 to account for the above range.
|
||||
*dest++ = 0xE0 + ((row - 0x40 + 1) >> 1);
|
||||
} else if (row >= 0x80) {
|
||||
// TODO
|
||||
}
|
||||
|
||||
if (row & 1) {
|
||||
if (offsetCell < 0x60) {
|
||||
// Subtract one to shift around 0x7F.
|
||||
*dest++ = offsetCell + 0x20 - 1;
|
||||
} else {
|
||||
*dest++ = offsetCell + 0x20;
|
||||
}
|
||||
} else {
|
||||
*dest++ = offsetCell + 0x7E;
|
||||
}
|
||||
|
||||
return 2;
|
||||
}
|
||||
|
||||
private:
|
||||
const char *c_;
|
||||
int index_;
|
||||
};
|
||||
@@ -0,0 +1,73 @@
|
||||
#pragma once
|
||||
|
||||
#include "base/basictypes.h"
|
||||
|
||||
#if defined(__LITTLE_ENDIAN__)
|
||||
#define UTF16_IS_LITTLE_ENDIAN 1
|
||||
#elif defined(__BIG_ENDIAN__)
|
||||
#define UTF16_IS_LITTLE_ENDIAN 0
|
||||
#else
|
||||
// Should optimize out.
|
||||
#define UTF16_IS_LITTLE_ENDIAN (*(const uint16_t *)"\0\xff" >= 0x100)
|
||||
#endif
|
||||
|
||||
template <bool is_little>
|
||||
uint16_t UTF16_Swap(uint16_t u) {
|
||||
if (is_little) {
|
||||
return UTF16_IS_LITTLE_ENDIAN ? u : swap16(u);
|
||||
} else {
|
||||
return UTF16_IS_LITTLE_ENDIAN ? swap16(u) : u;
|
||||
}
|
||||
}
|
||||
|
||||
template <bool is_little>
|
||||
struct UTF16_Type {
|
||||
public:
|
||||
static const uint32_t INVALID = (uint32_t)-1;
|
||||
|
||||
UTF16_Type(const uint16_t *c) : c_(c), index_(0) {}
|
||||
|
||||
uint32_t next() {
|
||||
const uint32_t u = UTF16_Swap<is_little>(c_[index_++]);
|
||||
|
||||
// Surrogate pair. UTF-16 is so simple. We assume it's valid.
|
||||
if ((u & 0xD800) == 0xD800) {
|
||||
return 0x10000 + (((u & 0x3FF) << 10) | (UTF16_Swap<is_little>(c_[index_++]) & 0x3FF));
|
||||
}
|
||||
return u;
|
||||
}
|
||||
|
||||
bool end() const {
|
||||
return c_[index_] == 0;
|
||||
}
|
||||
|
||||
int length() const {
|
||||
int len = 0;
|
||||
for (UTF16_Type<is_little> dec(c_); !dec.end(); dec.next())
|
||||
++len;
|
||||
return len;
|
||||
}
|
||||
|
||||
int byteIndex() const {
|
||||
return index_;
|
||||
}
|
||||
|
||||
static int encode(uint16_t *dest, uint32_t u) {
|
||||
if (u >= 0x10000) {
|
||||
u -= 0x10000;
|
||||
*dest = UTF16_Swap<is_little>(0xD800 + ((u >> 10) & 0x3FF));
|
||||
*dest = UTF16_Swap<is_little>(0xDC00 + ((u >> 0) & 0x3FF));
|
||||
return 2;
|
||||
} else {
|
||||
*dest = UTF16_Swap<is_little>((uint16_t)u);
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
const uint16_t *c_;
|
||||
int index_;
|
||||
};
|
||||
|
||||
typedef UTF16_Type<true> UTF16LE;
|
||||
typedef UTF16_Type<false> UTF16BE;
|
||||
@@ -17,10 +17,12 @@
|
||||
#include "base/basictypes.h"
|
||||
|
||||
uint32_t u8_nextchar(const char *s, int *i);
|
||||
int u8_wc_toutf8(char *dest, uint32_t ch);
|
||||
int u8_strlen(const char *s);
|
||||
|
||||
class UTF8 {
|
||||
public:
|
||||
static const uint32_t INVALID = (uint32_t)-1;
|
||||
UTF8(const char *c) : c_(c), index_(0) {}
|
||||
bool end() const { return c_[index_] == 0; }
|
||||
uint32_t next() {
|
||||
@@ -29,6 +31,12 @@ public:
|
||||
int length() const {
|
||||
return u8_strlen(c_);
|
||||
}
|
||||
int byteIndex() const {
|
||||
return index_;
|
||||
}
|
||||
static int encode(char *dest, uint32_t ch) {
|
||||
return u8_wc_toutf8(dest, ch);
|
||||
}
|
||||
|
||||
private:
|
||||
const char *c_;
|
||||
|
||||
Reference in new issue
Block a user