mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-10 21:16:24 +02:00
PPGe: Interpret invalid UTF-8 sequences better.
This matches PSP firmware behavior per tests.
This commit is contained in:
1 parent
7d085966f1
commit
5ef8762c32
3 files changed
+40
-4
No files matched your search
@@ -234,6 +234,32 @@ uint32_t u8_nextchar(const char *s, int *i)
|
||||
return ch;
|
||||
}
|
||||
|
||||
uint32_t u8_nextchar_unsafe(const char *s, int *i) {
|
||||
uint32_t ch = (unsigned char)s[(*i)++];
|
||||
int sz = 1;
|
||||
|
||||
if (ch >= 0xF0) {
|
||||
sz++;
|
||||
ch &= ~0x10;
|
||||
}
|
||||
if (ch >= 0xE0) {
|
||||
sz++;
|
||||
ch &= ~0x20;
|
||||
}
|
||||
if (ch >= 0xC0) {
|
||||
sz++;
|
||||
ch &= ~0xC0;
|
||||
}
|
||||
|
||||
// Just assume the bytes must be there. This is the logic used on the PSP.
|
||||
for (int j = 1; j < sz; ++j) {
|
||||
ch <<= 6;
|
||||
ch += ((unsigned char)s[(*i)++]) & 0x3F;
|
||||
}
|
||||
|
||||
return ch;
|
||||
}
|
||||
|
||||
void u8_inc(const char *s, int *i)
|
||||
{
|
||||
(void)(isutf(s[++(*i)]) || isutf(s[++(*i)]) ||
|
||||
@@ -489,9 +515,10 @@ std::string SanitizeUTF8(const std::string &utf8string) {
|
||||
// Worst case.
|
||||
s.resize(utf8string.size() * 4);
|
||||
|
||||
// This stops at invalid start bytes.
|
||||
size_t pos = 0;
|
||||
while (!utf.end_or_overlong_end()) {
|
||||
int c = utf.next();
|
||||
while (!utf.end() && !utf.invalid()) {
|
||||
int c = utf.next_unsafe();
|
||||
pos += UTF8::encode(&s[pos], c);
|
||||
}
|
||||
s.resize(pos);
|
||||
|
||||
Reference in new issue
Block a user