Encoding: Refactor mblen() to accept first byte instead of pointer

This commit is contained in:
kichikuou
2024-12-20 13:59:30 +09:00
parent b000fecf92
commit 24c6041915
6 changed files with 27 additions and 32 deletions
+13 -13
View File
@@ -11,12 +11,12 @@ namespace {
class SjisEncoding : public Encoding {
public:
int mblen(const unsigned char* s)
int mblen(unsigned char first_byte) override
{
return is_2byte(*s) ? 2 : 1;
return is_2byte(first_byte) ? 2 : 1;
}
int next_codepoint(const unsigned char** s)
int next_codepoint(const unsigned char** s) override
{
int code = *(*s)++;
if (is_2byte(code))
@@ -24,7 +24,7 @@ public:
return sjis_to_unicode(code);
}
char* fromUtf8(const char* str)
char* fromUtf8(const char* str) override
{
unsigned char* src = (unsigned char*)str;
unsigned char* dst = (unsigned char*)malloc(strlen(str) + 1);
@@ -65,7 +65,7 @@ public:
return (char*)dst;
}
char* toUtf8(const char* str)
char* toUtf8(const char* str) override
{
unsigned char* src = (unsigned char*)str;
unsigned char* dst = (unsigned char*)malloc(strlen(str) * 3 + 1);
@@ -138,18 +138,18 @@ private:
class Utf8Encoding : public Encoding {
public:
int mblen(const unsigned char* s)
int mblen(unsigned char first_byte) override
{
if (*s <= 0xbf)
if (first_byte <= 0xbf)
return 1;
if (*s <= 0xdf)
if (first_byte <= 0xdf)
return 2;
if (*s <= 0xef)
if (first_byte <= 0xef)
return 3;
return 4;
}
int next_codepoint(const unsigned char** str)
int next_codepoint(const unsigned char** str) override
{
int code;
const unsigned char *s = *str;
@@ -179,12 +179,12 @@ public:
return code;
}
char* fromUtf8(const char* s)
char* fromUtf8(const char* s) override
{
return strdup(s);
}
char* toUtf8(const char* s)
char* toUtf8(const char* s) override
{
return strdup(s);
}
@@ -194,7 +194,7 @@ int Encoding::mbslen(const unsigned char* s)
{
int len = 0;
while (*s) {
s += mblen(s);
s += mblen(*s);
len++;
}
return len;
+2 -5
View File
@@ -19,11 +19,8 @@ class Encoding {
int next_codepoint(const char** s) {
return next_codepoint(reinterpret_cast<const unsigned char **>(s));
}
// Returns byte length of the first character of s.
virtual int mblen(const unsigned char* s) = 0;
int mblen(const char* s) {
return mblen(reinterpret_cast<const unsigned char*>(s));
}
// Determines the byte length of a character based on the first byte.
virtual int mblen(unsigned char first_byte) = 0;
// Returns the number of characters in s.
int mbslen(const unsigned char* s);
int mbslen(const char* s) {
+1 -1
View File
@@ -299,7 +299,7 @@ void NACT::message(uint8 terminator)
const uint8* begin = sco.ptr();
const uint8* p = begin;
while (is_message(*p))
p += encoding->mblen(p);
p += encoding->mblen(*p);
int len = p - begin;
sco.skip(len);
+1 -2
View File
@@ -218,8 +218,7 @@ void NACT_Sys1::cmd_branch()
} else if (cmd == '\'' || cmd == '"') { // SysEng
sco.skip_syseng_string(encoding.get(), cmd);
} else if (is_message(cmd)) {
sco.ungetd();
sco.skip(encoding->mblen(sco.ptr()));
sco.skip(encoding->mblen(cmd) - 1);
} else {
sco.unknown_command(cmd);
}
+1 -2
View File
@@ -251,8 +251,7 @@ void NACT_Sys2::cmd_branch()
} else if (cmd == '\'' || cmd == '"') { // SysEng
sco.skip_syseng_string(encoding.get(), cmd);
} else if (is_message(cmd)) {
sco.ungetd();
sco.skip(encoding->mblen(sco.ptr()));
sco.skip(encoding->mblen(cmd) - 1);
} else {
sco.unknown_command(cmd);
}
+9 -9
View File
@@ -19,9 +19,9 @@ uint8_t Scenario::fetch_command()
void Scenario::skip_syseng_string(Encoding *enc, uint8_t terminator)
{
for (uint8_t c = getd(); c != terminator; c = getd()) {
if (c != '\\')
ungetd();
skip(enc->mblen(ptr()));
if (c == '\\')
c = getd();
skip(enc->mblen(c) - 1);
}
}
@@ -29,14 +29,14 @@ void Scenario::get_syseng_string(char* buf, int size, Encoding *enc, uint8_t ter
{
int i = 0;
for (uint8_t c = getd(); c != terminator; c = getd()) {
if (c != '\\')
ungetd();
int len = enc->mblen(ptr());
if (c == '\\')
c = getd();
int len = enc->mblen(c);
if (i + len >= size)
sys_error("String buffer overrun at %d:%04x", page_, cmd_addr_);
memcpy(&buf[i], ptr(), len);
i += len;
skip(len);
buf[i++] = c;
for (int j = 1; j < len; ++j)
buf[i++] = getd();
}
buf[i] = '\0';
}