From 7834004648efad7fdc5b069809934bf0fa018808 Mon Sep 17 00:00:00 2001 From: kichikuou Date: Sun, 5 Sep 2021 10:02:31 +0900 Subject: [PATCH] Fix unicode conversion on Android / Emscripten This also moves functions in utfsjis.cpp to encoding.cpp. --- src/CMakeLists.txt | 1 - src/android/nact_android.cpp | 11 +-- src/emscripten/nact_emscripten.cpp | 6 +- src/encoding.cpp | 126 ++++++++++++++++++++++++++++- src/encoding.h | 8 ++ src/sys/ags_text.cpp | 1 - src/utfsjis.cpp | 112 ------------------------- src/utfsjis.h | 14 ---- 8 files changed, 142 insertions(+), 137 deletions(-) delete mode 100644 src/utfsjis.cpp delete mode 100644 src/utfsjis.h diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index a0643dc..6d9b81b 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -140,7 +140,6 @@ target_sources(system3 PRIVATE config.cpp fileio.cpp resource.cpp - utfsjis.cpp encoding.cpp texthook.cpp fm/makofm.cpp diff --git a/src/android/nact_android.cpp b/src/android/nact_android.cpp index dd7e83f..2f29483 100644 --- a/src/android/nact_android.cpp +++ b/src/android/nact_android.cpp @@ -1,6 +1,6 @@ #include #include "nact.h" -#include "utfsjis.h" +#include "encoding.h" #include "jnihelper.h" #define STRING "Ljava/lang/String;" @@ -11,8 +11,9 @@ void NACT::text_dialog() if (!jni.env()) return; - char *oldstr = sjis2utf(tvar[tvar_index - 1]); + char *oldstr = encoding->toUtf8(tvar[tvar_index - 1]); jstring joldstr = jni.env()->NewStringUTF(oldstr); + free(oldstr); if (!joldstr) { WARNING("Failed to allocate a string"); return; @@ -22,9 +23,9 @@ void NACT::text_dialog() if (!jnewstr) return; const char* newstr_utf8 = jni.env()->GetStringUTFChars(jnewstr, NULL); - char* newstr_sjis = utf2sjis((char*)newstr_utf8); - strcpy_s(tvar[tvar_index - 1], 22, newstr_sjis); - free(newstr_sjis); + char* newstr = encoding->fromUtf8(newstr_utf8); + strcpy_s(tvar[tvar_index - 1], 22, newstr); + free(newstr); jni.env()->ReleaseStringUTFChars(jnewstr, newstr_utf8); } diff --git a/src/emscripten/nact_emscripten.cpp b/src/emscripten/nact_emscripten.cpp index ea0c115..0687e7c 100644 --- a/src/emscripten/nact_emscripten.cpp +++ b/src/emscripten/nact_emscripten.cpp @@ -1,6 +1,6 @@ #include #include "nact.h" -#include "utfsjis.h" +#include "encoding.h" #include "msgskip.h" #include @@ -19,7 +19,7 @@ Uint32 custom_event_type = static_cast(-1); void NACT::text_dialog() { static char buf[256]; - char *oldstr = sjis2utf(tvar[tvar_index - 1]); + char *oldstr = encoding->toUtf8(tvar[tvar_index - 1]); int ok = EM_ASM_({ var r = xsystem35.shell.inputString("文字列を入力してください", UTF8ToString($0), $1); if (r) { @@ -30,7 +30,7 @@ void NACT::text_dialog() }, oldstr, tvar_maxlen, buf, sizeof buf); free(oldstr); if (ok) { - char *newstr = utf2sjis(buf); + char *newstr = encoding->fromUtf8(buf); strcpy_s(tvar[tvar_index - 1], 22, newstr); free(newstr); } diff --git a/src/encoding.cpp b/src/encoding.cpp index 7bbf890..e652174 100644 --- a/src/encoding.cpp +++ b/src/encoding.cpp @@ -1,5 +1,13 @@ +#include +#include +#include "common.h" #include "encoding.h" -#include "utfsjis.h" + +namespace { + +#include "s2utbl.h" + +} // namespace class SjisEncoding : public Encoding { public: @@ -16,10 +24,116 @@ public: return sjis_to_unicode(code); } + char* fromUtf8(const char* str) + { + unsigned char* src = (unsigned char*)str; + unsigned char* dst = (unsigned char*)malloc(strlen(str) + 1); + unsigned char* dstp = dst; + + while (*src) { + if (*src <= 0x7f) { + *dstp++ = *src++; + continue; + } + + int u; + if (*src <= 0xdf) { + u = (src[0] & 0x1f) << 6 | (src[1] & 0x3f); + src += 2; + } else if (*src <= 0xef) { + u = (src[0] & 0xf) << 12 | (src[1] & 0x3f) << 6 | (src[2] & 0x3f); + src += 3; + } else { + *dstp++ = '?'; + do src++; while ((*src & 0xc0) == 0x80); + continue; + } + + if (u > 0xff60 && u <= 0xff9f) { + *dstp++ = u - 0xff60 + 0xa0; + } else { + int c = unicode_to_sjis(u); + if (c) { + *dstp++ = c >> 8; + *dstp++ = c & 0xff; + } else { + *dstp++ = '?'; + } + } + } + *dstp = '\0'; + return (char*)dst; + } + + char* toUtf8(const char* str) + { + unsigned char* src = (unsigned char*)str; + unsigned char* dst = (unsigned char*)malloc(strlen(str) * 3 + 1); + unsigned char* dstp = dst; + + while (*src) { + if (*src <= 0x7f) { + *dstp++ = *src++; + continue; + } + + int c; + if (*src >= 0xa0 && *src <= 0xdf) { + c = 0xff60 + *src - 0xa0; + src++; + } else { + c = s2u[*src - 0x80][*(src+1) - 0x40]; + src += 2; + } + + if (c <= 0x7f) { + *dstp++ = c; + } else if (c <= 0x7ff) { + *dstp++ = 0xc0 | c >> 6; + *dstp++ = 0x80 | (c & 0x3f); + } else { + *dstp++ = 0xe0 | c >> 12; + *dstp++ = 0x80 | (c >> 6 & 0x3f); + *dstp++ = 0x80 | (c & 0x3f); + } + } + *dstp = '\0'; + return (char*)dst; + } + private: bool is_2byte(unsigned char c) { return (0x81 <= c && c <= 0x9f) || 0xe0 <= c; } + + static int unicode_to_sjis(int u) { + for (int b1 = 0x80; b1 <= 0xff; b1++) { + if (b1 >= 0xa0 && b1 <= 0xdf) + continue; + for (int b2 = 0x40; b2 <= 0xff; b2++) { + if (u == s2u[b1 - 0x80][b2 - 0x40]) + return b1 << 8 | b2; + } + } + return 0; + } + + static uint16 sjis_to_unicode(uint16 code) + { + // ASCII characters. + if (code < 0x80) + return code; + // 1-byte kana characters. + if (code >= 0xa0 && code <= 0xdf) + return 0xff60 + code - 0xa0; + // Gaiji characters. + if (0xeb9f <= code && code <= 0xebfc) + return code - 0xeb9f + GAIJI_FIRST; + if (0xec40 <= code && code <= 0xec9e) + return code - 0xec40 + 94 + GAIJI_FIRST; + + return s2u[(code >> 8) - 0x80][(code & 0xff) - 0x40]; + } }; class Utf8Encoding : public Encoding { @@ -64,6 +178,16 @@ public: *str = s; return code; } + + char* fromUtf8(const char* s) + { + return strdup(s); + } + + char* toUtf8(const char* s) + { + return strdup(s); + } }; int Encoding::mbslen(const unsigned char* s) diff --git a/src/encoding.h b/src/encoding.h index 4e1ae05..83bf9e1 100644 --- a/src/encoding.h +++ b/src/encoding.h @@ -3,6 +3,10 @@ #include +// Gaiji characters are mapped to Unicode Private Use Area U+E000-U+E0BB. +const int GAIJI_FIRST = 0xE000; +const int GAIJI_LAST = 0xE0BB; + class Encoding { public: static std::unique_ptr create(const char* name); @@ -25,6 +29,10 @@ class Encoding { int mbslen(const char* s) { return mbslen(reinterpret_cast(s)); } + + // Convert from/to utf-8 encoding. Caller must free() the returned buffer. + virtual char* fromUtf8(const char* s) = 0; + virtual char* toUtf8(const char* s) = 0; }; #endif // _ENCODING_H_ diff --git a/src/sys/ags_text.cpp b/src/sys/ags_text.cpp index 26774e3..74f0dae 100644 --- a/src/sys/ags_text.cpp +++ b/src/sys/ags_text.cpp @@ -7,7 +7,6 @@ #include #include "ags.h" #include "encoding.h" -#include "utfsjis.h" #include "texthook.h" namespace { diff --git a/src/utfsjis.cpp b/src/utfsjis.cpp deleted file mode 100644 index 8da335b..0000000 --- a/src/utfsjis.cpp +++ /dev/null @@ -1,112 +0,0 @@ -#include -#include -#include "utfsjis.h" - -namespace { -#include "s2utbl.h" - -int unicode_to_sjis(int u) { - for (int b1 = 0x80; b1 <= 0xff; b1++) { - if (b1 >= 0xa0 && b1 <= 0xdf) - continue; - for (int b2 = 0x40; b2 <= 0xff; b2++) { - if (u == s2u[b1 - 0x80][b2 - 0x40]) - return b1 << 8 | b2; - } - } - return 0; -} - -} // namespace - -uint16 sjis_to_unicode(uint16 code) -{ - // ASCII characters. - if (code < 0x80) - return code; - // 1-byte kana characters. - if (code >= 0xa0 && code <= 0xdf) - return 0xff60 + code - 0xa0; - // Gaiji characters. - if (0xeb9f <= code && code <= 0xebfc) - return code - 0xeb9f + GAIJI_FIRST; - if (0xec40 <= code && code <= 0xec9e) - return code - 0xec40 + 94 + GAIJI_FIRST; - - return s2u[(code >> 8) - 0x80][(code & 0xff) - 0x40]; -} - -char* sjis2utf(char* str) { - unsigned char* src = (unsigned char*)str; - unsigned char* dst = (unsigned char*)malloc(strlen(str) * 3 + 1); - unsigned char* dstp = dst; - - while (*src) { - if (*src <= 0x7f) { - *dstp++ = *src++; - continue; - } - - int c; - if (*src >= 0xa0 && *src <= 0xdf) { - c = 0xff60 + *src - 0xa0; - src++; - } else { - c = s2u[*src - 0x80][*(src+1) - 0x40]; - src += 2; - } - - if (c <= 0x7f) { - *dstp++ = c; - } else if (c <= 0x7ff) { - *dstp++ = 0xc0 | c >> 6; - *dstp++ = 0x80 | (c & 0x3f); - } else { - *dstp++ = 0xe0 | c >> 12; - *dstp++ = 0x80 | (c >> 6 & 0x3f); - *dstp++ = 0x80 | (c & 0x3f); - } - } - *dstp = '\0'; - return (char*)dst; -} - -char* utf2sjis(char* str) { - unsigned char* src = (unsigned char*)str; - unsigned char* dst = (unsigned char*)malloc(strlen(str) + 1); - unsigned char* dstp = dst; - - while (*src) { - if (*src <= 0x7f) { - *dstp++ = *src++; - continue; - } - - int u; - if (*src <= 0xdf) { - u = (src[0] & 0x1f) << 6 | (src[1] & 0x3f); - src += 2; - } else if (*src <= 0xef) { - u = (src[0] & 0xf) << 12 | (src[1] & 0x3f) << 6 | (src[2] & 0x3f); - src += 3; - } else { - *dstp++ = '?'; - do src++; while ((*src & 0xc0) == 0x80); - continue; - } - - if (u > 0xff60 && u <= 0xff9f) { - *dstp++ = u - 0xff60 + 0xa0; - } else { - int c = unicode_to_sjis(u); - if (c) { - *dstp++ = c >> 8; - *dstp++ = c & 0xff; - } else { - *dstp++ = '?'; - } - } - } - *dstp = '\0'; - return (char*)dst; -} diff --git a/src/utfsjis.h b/src/utfsjis.h deleted file mode 100644 index d640217..0000000 --- a/src/utfsjis.h +++ /dev/null @@ -1,14 +0,0 @@ -#ifndef _UTFSJIS_H_ -#define _UTFSJIS_H_ - -#include "common.h" - -// Gaiji characters are mapped to Unicode Private Use Area U+E000-U+E0BB. -const int GAIJI_FIRST = 0xE000; -const int GAIJI_LAST = 0xE0BB; - -uint16 sjis_to_unicode(uint16 code); -char* sjis2utf(char* src); -char* utf2sjis(char* src); - -#endif // _UTFSJIS_H_