diff --git a/CMakeLists.txt b/CMakeLists.txt index d838f64fde8..3764a2f5ba3 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -65,6 +65,7 @@ else() endif() include(cmake/config.cmake) +include(cmake/icu.cmake) include(cmake/gamespy.cmake) include(cmake/lzhl.cmake) include(cmake/stb.cmake) @@ -77,6 +78,7 @@ endif() add_subdirectory(Dependencies/Precompiled) add_subdirectory(Dependencies/Utility) +add_subdirectory(Dependencies/ICU) add_subdirectory(Dependencies/Bink) add_subdirectory(Dependencies/Miles) if (WIN32) diff --git a/Core/GameEngine/CMakeLists.txt b/Core/GameEngine/CMakeLists.txt index ba6c63ac21e..28af20fbde9 100644 --- a/Core/GameEngine/CMakeLists.txt +++ b/Core/GameEngine/CMakeLists.txt @@ -1212,6 +1212,7 @@ target_link_libraries(corei_gameengine_public INTERFACE core_browserdispatch dbghelploader #core_wwvegas + core_icu d3d8lib gamespy::gamespy stlport diff --git a/Core/GameEngine/Source/Common/System/AsciiString.cpp b/Core/GameEngine/Source/Common/System/AsciiString.cpp index 6bf644428ee..653b4a6396c 100644 --- a/Core/GameEngine/Source/Common/System/AsciiString.cpp +++ b/Core/GameEngine/Source/Common/System/AsciiString.cpp @@ -45,7 +45,7 @@ #include "PreRTS.h" // This must go first in EVERY cpp file in the GameEngine #include "Common/CriticalSection.h" -#include "WWLib/utf8.h" +#include "ICU/utf8.h" // ----------------------------------------------------- @@ -306,18 +306,27 @@ void AsciiString::translate(const UnicodeString& stringSrc) { validate(); // TheSuperHackers @fix bobtista 02/04/2026 Implement UTF-8 conversion replacing 7-bit ASCII only implementation + // TheSuperHackers @bugfix CryoTheRenegade 04/08/2026 Convert wide text to UTF-8 with ICU4C. const WideChar* src = stringSrc.str(); - const size_t srcLen = wcslen(src); + const size_t srcLen = stringSrc.getLength(); const size_t dstLen = Wide_To_Utf8_Len(src, srcLen); if (dstLen == 0) { clear(); } + else if (dstLen >= static_cast(MAX_LEN)) + { + DEBUG_ASSERTCRASH(false, + ("AsciiString::translate exceeds max string length %d with required UTF-8 length %u", + MAX_LEN, static_cast(dstLen))); + clear(); + } else { - ensureUniqueBufferOfSize((Int)dstLen + 1, false, nullptr, nullptr); + ensureUniqueBufferOfSize(static_cast(dstLen) + 1, false, nullptr, nullptr); Wide_To_Utf8(peek(), dstLen + 1, src, srcLen); } + validate(); } diff --git a/Core/GameEngine/Source/Common/System/UnicodeString.cpp b/Core/GameEngine/Source/Common/System/UnicodeString.cpp index c8e12a9b185..859d6d48642 100644 --- a/Core/GameEngine/Source/Common/System/UnicodeString.cpp +++ b/Core/GameEngine/Source/Common/System/UnicodeString.cpp @@ -45,7 +45,7 @@ #include "PreRTS.h" // This must go first in EVERY cpp file in the GameEngine #include "Common/CriticalSection.h" -#include "WWLib/utf8.h" +#include "ICU/utf8.h" // ----------------------------------------------------- @@ -222,31 +222,32 @@ void UnicodeString::translate(const AsciiString& stringSrc) // TheSuperHackers @fix bobtista 02/04/2026 Convert UTF-8 to wide, replacing the 7-bit ASCII only // implementation. Data that is not valid UTF-8 (e.g. legacy CP1252) falls back to a 1:1 byte cast // to preserve the original characters instead of producing replacement characters. + // TheSuperHackers @bugfix CryoTheRenegade 04/08/2026 Convert UTF-8 to wide text with ICU4C. const char* src = stringSrc.str(); - const size_t srcLen = strlen(src); + const size_t srcLen = stringSrc.getLength(); const size_t dstLen = Utf8_To_Wide_Len(src, srcLen); - if (dstLen != UTF8_INVALID) + if (srcLen == 0) { - if (dstLen == 0) - { - clear(); - } - else - { - ensureUniqueBufferOfSize((Int)dstLen + 1, false, nullptr, nullptr); - Utf8_To_Wide(peek(), dstLen + 1, src, srcLen); - } + clear(); } - else + else if (dstLen == UTF8_INVALID) { - ensureUniqueBufferOfSize((Int)srcLen + 1, false, nullptr, nullptr); + // Preserve legacy non-UTF-8 data with the original one-byte-to-one-wide-unit behavior. + ensureUniqueBufferOfSize(static_cast(srcLen) + 1, false, nullptr, nullptr); WideChar* buf = peek(); for (size_t i = 0; i < srcLen; ++i) { - buf[i] = (WideChar)(unsigned char)src[i]; + buf[i] = static_cast(static_cast(src[i])); } + buf[srcLen] = 0; } + else + { + ensureUniqueBufferOfSize(static_cast(dstLen) + 1, false, nullptr, nullptr); + Utf8_To_Wide(peek(), dstLen + 1, src, srcLen); + } + validate(); } diff --git a/Core/GameEngine/Source/GameNetwork/GameInfo.cpp b/Core/GameEngine/Source/GameNetwork/GameInfo.cpp index d7ec6355124..8dd9b58c961 100644 --- a/Core/GameEngine/Source/GameNetwork/GameInfo.cpp +++ b/Core/GameEngine/Source/GameNetwork/GameInfo.cpp @@ -44,7 +44,7 @@ #include "GameNetwork/LANAPI.h" // for testing packet size #include "GameNetwork/LANAPICallbacks.h" // for testing packet size #include "WWLib/strtok_r.h" -#include "WWLib/utf8.h" +#include "ICU/utf8.h" diff --git a/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp b/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp index 84ce0c9b19a..9699a427d71 100644 --- a/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp +++ b/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp @@ -28,26 +28,28 @@ #include "PreRTS.h" // This must go first in EVERY cpp file in the GameEngine -#include "WWLib/utf8.h" +#include "ICU/utf8.h" //------------------------------------------------------------------------- // TheSuperHackers @refactor bobtista 02/04/2026 Use WWLib UTF-8 functions instead of raw Win32 API calls -std::wstring MultiByteToWideCharSingleLine( const char *orig ) +// TheSuperHackers @refactor CryoTheRenegade 04/08/2026 Use the shared ICU4C UTF conversion functions. +std::wstring MultiByteToWideCharSingleLine( const char* orig ) { const size_t srcLen = strlen(orig); const size_t dstLen = Utf8_To_Wide_Len(orig, srcLen); if (dstLen == 0) + { return std::wstring(); + } + std::wstring ret; if (dstLen == UTF8_INVALID) { - // Not UTF-8. Fall back to a 1:1 byte cast so legacy data keeps its characters, matching - // UnicodeString::translate. ret.resize(srcLen); for (size_t i = 0; i < srcLen; ++i) { - ret[i] = (WideChar)(unsigned char)orig[i]; + ret[i] = static_cast(static_cast(orig[i])); } } else @@ -55,35 +57,27 @@ std::wstring MultiByteToWideCharSingleLine( const char *orig ) ret.resize(dstLen); Utf8_To_Wide(&ret[0], dstLen, orig, srcLen); } - WideChar *c = nullptr; - do - { - c = wcschr(&ret[0], L'\n'); - if (c) - { - *c = L' '; - } - } - while ( c != nullptr ); - do + + for (size_t i = 0; i < ret.size(); ++i) { - c = wcschr(&ret[0], L'\r'); - if (c) + if (ret[i] == L'\n' || ret[i] == L'\x0D') { - *c = L' '; + ret[i] = L' '; } } - while ( c != nullptr ); return ret; } -std::string WideCharStringToMultiByte( const WideChar *orig ) +std::string WideCharStringToMultiByte( const WideChar* orig ) { const size_t srcLen = wcslen(orig); const size_t dstLen = Wide_To_Utf8_Len(orig, srcLen); if (dstLen == 0) + { return std::string(); + } + std::string ret; ret.resize(dstLen); Wide_To_Utf8(&ret[0], dstLen, orig, srcLen); diff --git a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt index 23d32b1dc66..cbca9c63baa 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt +++ b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt @@ -126,8 +126,6 @@ set(WWLIB_SRC trim.cpp trim.h uarray.h - utf8.cpp - utf8.h vector.cpp Vector.h visualc.h @@ -180,3 +178,7 @@ target_link_libraries(core_wwlib PRIVATE core_wwcommon corei_always ) + +target_link_libraries(core_wwlib PUBLIC + core_icu +) diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp deleted file mode 100644 index 578299fbe3d..00000000000 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp +++ /dev/null @@ -1,304 +0,0 @@ -/* -** Command & Conquer Generals Zero Hour(tm) -** Copyright 2026 TheSuperHackers -** -** This program is free software: you can redistribute it and/or modify -** it under the terms of the GNU General Public License as published by -** the Free Software Foundation, either version 3 of the License, or -** (at your option) any later version. -** -** This program is distributed in the hope that it will be useful, -** but WITHOUT ANY WARRANTY; without even the implied warranty of -** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -** GNU General Public License for more details. -** -** You should have received a copy of the GNU General Public License -** along with this program. If not, see . -*/ - -#include "always.h" -#include "utf8.h" - -// wchar_t is a 16-bit UTF-16 code unit on Windows and a 32-bit UTF-32 codepoint on most other -// platforms. WCHAR_MAX lets us distinguish the two at compile time so the surrogate-pair paths -// are excluded entirely (not just constant-folded) where wchar_t is wide enough to hold a codepoint. -#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) -#define UTF8_WCHAR_IS_UTF16 1 -#else -#define UTF8_WCHAR_IS_UTF16 0 -#endif - -static const unsigned int UTF8_CODEPOINT_MAX = 0x10FFFF; -static const unsigned int UTF8_SURROGATE_MIN = 0xD800; -static const unsigned int UTF8_SURROGATE_MAX = 0xDFFF; -static const unsigned int UTF8_REPLACEMENT_CHAR = 0xFFFD; - -// Number of UTF-8 bytes required to encode a codepoint. -static size_t Utf8_Encoded_Length(unsigned int cp) -{ - if (cp < 0x80) - { - return 1; - } - if (cp < 0x800) - { - return 2; - } - if (cp < 0x10000) - { - return 3; - } - return 4; -} - -// Encode a codepoint to dest, which is assumed to have room. Returns the number of bytes written. -static size_t Utf8_Encode(char* dest, unsigned int cp) -{ - if (cp < 0x80) - { - dest[0] = (char)cp; - return 1; - } - if (cp < 0x800) - { - dest[0] = (char)(0xC0 | (cp >> 6)); - dest[1] = (char)(0x80 | (cp & 0x3F)); - return 2; - } - if (cp < 0x10000) - { - dest[0] = (char)(0xE0 | (cp >> 12)); - dest[1] = (char)(0x80 | ((cp >> 6) & 0x3F)); - dest[2] = (char)(0x80 | (cp & 0x3F)); - return 3; - } - dest[0] = (char)(0xF0 | (cp >> 18)); - dest[1] = (char)(0x80 | ((cp >> 12) & 0x3F)); - dest[2] = (char)(0x80 | ((cp >> 6) & 0x3F)); - dest[3] = (char)(0x80 | (cp & 0x3F)); - return 4; -} - -// Decode one UTF-8 sequence at src, with srcLen bytes remaining. On success returns the number of -// bytes consumed (1-4) and sets cp. Returns 0 on any malformed, overlong, out-of-range or surrogate -// encoding. -static size_t Utf8_Decode(const char* src, size_t srcLen, unsigned int& cp) -{ - const unsigned char lead = (unsigned char)src[0]; - if (lead < 0x80) - { - cp = lead; - return 1; - } - - size_t count; - unsigned int lowerBound; - if ((lead & 0xE0) == 0xC0) - { - count = 2; - cp = lead & 0x1F; - lowerBound = 0x80; - } - else if ((lead & 0xF0) == 0xE0) - { - count = 3; - cp = lead & 0x0F; - lowerBound = 0x800; - } - else if ((lead & 0xF8) == 0xF0) - { - count = 4; - cp = lead & 0x07; - lowerBound = 0x10000; - } - else - { - return 0; // a continuation byte or a 5/6-byte form cannot start a sequence - } - - if (srcLen < count) - { - return 0; // truncated sequence - } - for (size_t i = 1; i < count; ++i) - { - const unsigned char trail = (unsigned char)src[i]; - if ((trail & 0xC0) != 0x80) - { - return 0; // not a continuation byte - } - cp = (cp << 6) | (trail & 0x3F); - } - - if (cp < lowerBound || cp > UTF8_CODEPOINT_MAX || (cp >= UTF8_SURROGATE_MIN && cp <= UTF8_SURROGATE_MAX)) - { - return 0; // overlong, out of range, or a surrogate codepoint - } - return count; -} - -// Read one codepoint at src, with srcLen wide characters remaining. Returns the number of wide -// characters consumed (1-2) and sets cp. Combines UTF-16 surrogate pairs where wchar_t is 16-bit; -// treats each element as a whole codepoint where wchar_t is 32-bit. Wide data that has no UTF-8 -// representation is reported as U+FFFD, so the encoder never emits a sequence that the decoder -// would reject. -static size_t Wide_Read(const wchar_t* src, size_t srcLen, unsigned int& cp) -{ - size_t consumed = 1; -#if UTF8_WCHAR_IS_UTF16 - cp = (unsigned int)src[0] & 0xFFFF; - if (cp >= UTF8_SURROGATE_MIN && cp <= 0xDBFF && srcLen > 1) - { - const unsigned int low = (unsigned int)src[1] & 0xFFFF; - if (low >= 0xDC00 && low <= UTF8_SURROGATE_MAX) - { - cp = 0x10000 + ((cp - UTF8_SURROGATE_MIN) << 10) + (low - 0xDC00); - consumed = 2; - } - } -#else - (void)srcLen; - cp = (unsigned int)src[0]; -#endif - if (cp > UTF8_CODEPOINT_MAX || (cp >= UTF8_SURROGATE_MIN && cp <= UTF8_SURROGATE_MAX)) - { - cp = UTF8_REPLACEMENT_CHAR; - } - return consumed; -} - -// Number of wide characters required to store a codepoint. -static size_t Wide_Encoded_Length(unsigned int cp) -{ -#if UTF8_WCHAR_IS_UTF16 - return (cp >= 0x10000) ? 2 : 1; -#else - (void)cp; - return 1; -#endif -} - -// Write one codepoint to a wide buffer, which is assumed to have room. Returns wide characters written. -static size_t Wide_Write(wchar_t* dest, unsigned int cp) -{ -#if UTF8_WCHAR_IS_UTF16 - if (cp >= 0x10000) - { - cp -= 0x10000; - dest[0] = (wchar_t)(0xD800 + (cp >> 10)); - dest[1] = (wchar_t)(0xDC00 + (cp & 0x3FF)); - return 2; - } -#endif - dest[0] = (wchar_t)cp; - return 1; -} - -size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) -{ - size_t needed = 0; - size_t i = 0; - while (i < srcLen) - { - unsigned int cp; - i += Wide_Read(src + i, srcLen - i, cp); - needed += Utf8_Encoded_Length(cp); - } - return needed; -} - -size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) -{ - size_t needed = 0; - size_t i = 0; - while (i < srcLen) - { - unsigned int cp; - const size_t consumed = Utf8_Decode(src + i, srcLen - i, cp); - if (consumed == 0) - { - return UTF8_INVALID; - } - i += consumed; - needed += Wide_Encoded_Length(cp); - } - return needed; -} - -size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) -{ - size_t needed = 0; - size_t out = 0; - size_t i = 0; - while (i < srcLen) - { - unsigned int cp; - i += Wide_Read(src + i, srcLen - i, cp); - const size_t need = Utf8_Encoded_Length(cp); - // Stop writing at the first codepoint that does not fit, but keep counting for the caller. - if (needed == out && out + need <= destLen) - { - out += Utf8_Encode(dest + out, cp); - } - needed += need; - } - if (out < destLen) - { - dest[out] = '\0'; - } - return needed; -} - -size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) -{ - size_t needed = 0; - size_t out = 0; - size_t i = 0; - while (i < srcLen) - { - unsigned int cp; - const size_t consumed = Utf8_Decode(src + i, srcLen - i, cp); - if (consumed == 0) - { - if (destLen > 0) - { - dest[0] = L'\0'; - } - return UTF8_INVALID; - } - i += consumed; - const size_t need = Wide_Encoded_Length(cp); - // Stop writing at the first codepoint that does not fit, but keep counting for the caller. - if (needed == out && out + need <= destLen) - { - out += Wide_Write(dest + out, cp); - } - needed += need; - } - if (out < destLen) - { - dest[out] = L'\0'; - } - return needed; -} - -// A UTF-8 continuation byte matches 10xxxxxx, so it can never start a sequence. -static bool Utf8_Is_Continuation_Byte(char c) -{ - return ((unsigned char)c & 0xC0) == 0x80; -} - -size_t Utf8_Truncate_Len(const char* src, size_t srcLen, size_t maxLen) -{ - if (srcLen <= maxLen) - { - return srcLen; - } - - size_t len = maxLen; - while (len > 0 && Utf8_Is_Continuation_Byte(src[len])) - { - --len; - } - return len; -} diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.h b/Core/Libraries/Source/WWVegas/WWLib/utf8.h deleted file mode 100644 index 0a4f92f95ee..00000000000 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.h +++ /dev/null @@ -1,63 +0,0 @@ -/* -** Command & Conquer Generals Zero Hour(tm) -** Copyright 2026 TheSuperHackers -** -** This program is free software: you can redistribute it and/or modify -** it under the terms of the GNU General Public License as published by -** the Free Software Foundation, either version 3 of the License, or -** (at your option) any later version. -** -** This program is distributed in the hope that it will be useful, -** but WITHOUT ANY WARRANTY; without even the implied warranty of -** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -** GNU General Public License for more details. -** -** You should have received a copy of the GNU General Public License -** along with this program. If not, see . -*/ - -#pragma once - -#include -#include - -// UTF-8 <-> wide-character transcoding, hand-rolled per RFC 3629, using no platform text APIs. -// The wide side is wchar_t, whose width is platform-dependent: on Windows it is a 16-bit UTF-16 -// code unit (astral codepoints use surrogate pairs); on most other platforms it is a 32-bit -// UTF-32 codepoint. Both are handled transparently based on the width of wchar_t. - -// Returned by the decoding functions when the source is not well-formed UTF-8. A return of 0 means -// an empty result, which is a success and must not be confused with a decoding failure. -const size_t UTF8_INVALID = (size_t)-1; - -// Returns the number of UTF-8 bytes needed for the UTF-8 representation of srcLen wide characters -// from src, not counting a null terminator. Returns 0 if srcLen is 0. Wide values that have no -// UTF-8 representation are counted as U+FFFD. -size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen); - -// Returns the number of wide characters needed for the wide representation of srcLen bytes from the -// UTF-8 string src, not counting a null terminator. Returns 0 if srcLen is 0, or UTF8_INVALID if -// src is not well-formed UTF-8. -size_t Utf8_To_Wide_Len(const char* src, size_t srcLen); - -// Converts srcLen wide characters from src to UTF-8. destLen is the destination buffer capacity in -// bytes. Writes a null terminator if room remains, otherwise not. Wide values that have no UTF-8 -// representation are written as U+FFFD, so the output always decodes back through Utf8_To_Wide. -// Returns the number of bytes the whole conversion needs, not counting a null terminator. A return -// greater than destLen means the output was truncated on a codepoint boundary; retry with that many -// bytes plus one for the terminator. Pass destLen 0 to measure without writing. -size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen); - -// Converts srcLen bytes from the UTF-8 string src to wide characters. destLen is the destination -// buffer capacity in wide characters. Writes a null terminator if room remains, otherwise not. -// Returns the number of wide characters the whole conversion needs, not counting a null terminator. -// A return greater than destLen means the output was truncated on a codepoint boundary; retry with -// that many wide characters plus one for the terminator. Pass destLen 0 to measure without writing. -// Returns UTF8_INVALID if src is not well-formed UTF-8, setting dest[0] to L'\0' if destLen > 0. -size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen); - -// Returns the largest length not greater than maxLen at which the srcLen bytes of the UTF-8 string -// src can be cut without splitting a multibyte sequence, by backing off the continuation bytes at -// the cut point. Returns srcLen when the string already fits in maxLen. Returns 0 when no whole -// sequence fits, which is also what malformed UTF-8 yields once it has no lead byte to back off to. -size_t Utf8_Truncate_Len(const char* src, size_t srcLen, size_t maxLen); diff --git a/Dependencies/ICU/CMakeLists.txt b/Dependencies/ICU/CMakeLists.txt new file mode 100644 index 00000000000..245a5ba357a --- /dev/null +++ b/Dependencies/ICU/CMakeLists.txt @@ -0,0 +1,11 @@ +# ICU selection and public link settings are configured in cmake/icu.cmake. +target_sources(core_icu PRIVATE + ICU/IcuLoader.cpp + ICU/IcuLoader.h + ICU/IcuSupport.h + ICU/utf8.cpp + ICU/utf8.h +) + +target_include_directories(core_icu PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}) +target_link_libraries(core_icu PRIVATE core_config core_utility stlport) diff --git a/Dependencies/ICU/ICU/IcuLoader.cpp b/Dependencies/ICU/ICU/IcuLoader.cpp new file mode 100644 index 00000000000..5a50cf6d0d3 --- /dev/null +++ b/Dependencies/ICU/ICU/IcuLoader.cpp @@ -0,0 +1,202 @@ +/* +** Command & Conquer Generals Zero Hour(tm) +** Copyright 2026 TheSuperHackers +** +** This program is free software: you can redistribute it and/or modify +** it under the terms of the GNU General Public License as published by +** the Free Software Foundation, either version 3 of the License, or +** (at your option) any later version. +** +** This program is distributed in the hope that it will be useful, +** but WITHOUT ANY WARRANTY; without even the implied warranty of +** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +** GNU General Public License for more details. +** +** You should have received a copy of the GNU General Public License +** along with this program. If not, see . +*/ + +#include "ICU/IcuLoader.h" + +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + +#include +#include +#include +#include + +namespace +{ + +typedef IcuLoader::Char* (__cdecl* StrFromUtf8)(IcuLoader::Char*, int, int*, const char*, int, IcuLoader::ErrorCode*); +typedef char* (__cdecl* StrToUtf8WithSub)(char*, int, int*, const IcuLoader::Char*, int, + IcuLoader::Char32, int*, IcuLoader::ErrorCode*); + +bool LoadAttempted = false; +HMODULE Module = nullptr; +StrFromUtf8 FromUtf8 = nullptr; +StrToUtf8WithSub ToUtf8WithSub = nullptr; + +CRITICAL_SECTION CriticalSection; +LONG CriticalSectionState = 0; + +void freeResources(); + +void cleanup() +{ + // Conversion workers must have stopped before process shutdown. + freeResources(); + LoadAttempted = false; + DeleteCriticalSection(&CriticalSection); + InterlockedExchange(&CriticalSectionState, 0); +} + +void initializeCriticalSection() +{ + // Startup logging can convert strings before global constructors have run. + // VC6 also lacks synchronized local statics. Publish the native lock once; + // this loop only waits during initialization, not during ICU conversions. + while (InterlockedCompareExchange(&CriticalSectionState, 2, 2) != 2) + { + if (InterlockedCompareExchange(&CriticalSectionState, 1, 0) == 0) + { + InitializeCriticalSection(&CriticalSection); + atexit(cleanup); + InterlockedExchange(&CriticalSectionState, 2); + return; + } + + Sleep(0); + } +} + +class LoaderLock +{ +public: + LoaderLock() + { + initializeCriticalSection(); + EnterCriticalSection(&CriticalSection); + } + + ~LoaderLock() + { + LeaveCriticalSection(&CriticalSection); + } + +private: + LoaderLock(const LoaderLock&); + LoaderLock& operator=(const LoaderLock&); +}; + +HMODULE loadModule() +{ + // Use absolute paths so -setCwd and PATH cannot supply an unexpected DLL. + // Wide paths also allow app-local ICU in non-ASCII installation directories. + wchar_t path[MAX_PATH]; + const wchar_t name[] = L"icu.dll"; + const DWORD length = GetModuleFileNameW(nullptr, path, MAX_PATH); + if (length > 0 && length < MAX_PATH) + { + wchar_t* separator = wcsrchr(path, L'\\'); + if (separator != nullptr && separator + 1 - path + sizeof(name) / sizeof(name[0]) <= MAX_PATH) + { + memcpy(separator + 1, name, sizeof(name)); + HMODULE module = LoadLibraryW(path); + if (module != nullptr) + { + return module; + } + } + } + + const UINT systemLength = GetSystemDirectoryW(path, MAX_PATH); + if (systemLength > 0 && systemLength + 1 + sizeof(name) / sizeof(name[0]) <= MAX_PATH) + { + path[systemLength] = L'\\'; + memcpy(path + systemLength + 1, name, sizeof(name)); + return LoadLibraryW(path); + } + + return nullptr; +} + +void freeResources() +{ + if (Module != nullptr) + { + FreeLibrary(Module); + Module = nullptr; + } + + FromUtf8 = nullptr; + ToUtf8WithSub = nullptr; +} + +// The caller holds the critical section across loading and the ICU call. +bool load() +{ + if (LoadAttempted) + { + return Module != nullptr; + } + + LoadAttempted = true; + Module = loadModule(); + if (Module == nullptr) + { + return false; + } + + FromUtf8 = reinterpret_cast(GetProcAddress(Module, "u_strFromUTF8")); + ToUtf8WithSub = reinterpret_cast(GetProcAddress(Module, "u_strToUTF8WithSub")); + if (FromUtf8 == nullptr || ToUtf8WithSub == nullptr) + { + freeResources(); + return false; + } + + return true; +} + +} // namespace + +bool IcuLoader::isAvailable() +{ + LoaderLock lock; + return load(); +} + +void IcuLoader::unload() +{ + LoaderLock lock; + freeResources(); + LoadAttempted = false; +} + +bool IcuLoader::fromUtf8(Char* dest, int capacity, int* length, const char* src, int srcLength, ErrorCode* error) +{ + LoaderLock lock; + if (!load()) + { + return false; + } + + FromUtf8(dest, capacity, length, src, srcLength, error); + return true; +} + +bool IcuLoader::toUtf8WithSub(char* dest, int capacity, int* length, const Char* src, int srcLength, + Char32 substitution, int* substitutions, ErrorCode* error) +{ + LoaderLock lock; + if (!load()) + { + return false; + } + + ToUtf8WithSub(dest, capacity, length, src, srcLength, substitution, substitutions, error); + return true; +} + +#endif diff --git a/Dependencies/ICU/ICU/IcuLoader.h b/Dependencies/ICU/ICU/IcuLoader.h new file mode 100644 index 00000000000..b3b383806eb --- /dev/null +++ b/Dependencies/ICU/ICU/IcuLoader.h @@ -0,0 +1,52 @@ +/* +** Command & Conquer Generals Zero Hour(tm) +** Copyright 2026 TheSuperHackers +** +** This program is free software: you can redistribute it and/or modify +** it under the terms of the GNU General Public License as published by +** the Free Software Foundation, either version 3 of the License, or +** (at your option) any later version. +** +** This program is distributed in the hope that it will be useful, +** but WITHOUT ANY WARRANTY; without even the implied warranty of +** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +** GNU General Public License for more details. +** +** You should have received a copy of the GNU General Public License +** along with this program. If not, see . +*/ + +#pragma once + +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + +#include + +// Loads icu.dll on the first conversion and caches success or failure until unload. +// Conversion calls and unload are serialized with a Windows critical section, +// including on VC6. Function pointers stay private and cannot outlive the DLL. +// An explicit unload releases the DLL and allows the next conversion to retry. +class IcuLoader +{ +public: + // These types match the Windows ICU C ABI, including its 32-bit error enum. + typedef unsigned short Char; + typedef int Char32; + typedef int ErrorCode; + + static bool isAvailable(); + static void unload(); + + // Return false when ICU is unavailable; otherwise call ICU and return true. + // The caller checks error for the conversion result, including preflight. + static bool fromUtf8(Char* dest, int capacity, int* length, const char* src, int srcLength, ErrorCode* error); + static bool toUtf8WithSub(char* dest, int capacity, int* length, const Char* src, int srcLength, + Char32 substitution, int* substitutions, ErrorCode* error); + +private: + IcuLoader(); + IcuLoader(const IcuLoader&); + IcuLoader& operator=(const IcuLoader&); +}; + +#endif diff --git a/Dependencies/ICU/ICU/IcuSupport.h b/Dependencies/ICU/ICU/IcuSupport.h new file mode 100644 index 00000000000..24c3ce8fd91 --- /dev/null +++ b/Dependencies/ICU/ICU/IcuSupport.h @@ -0,0 +1,59 @@ +/* +** Command & Conquer Generals Zero Hour(tm) +** Copyright 2026 TheSuperHackers +** +** This program is free software: you can redistribute it and/or modify +** it under the terms of the GNU General Public License as published by +** the Free Software Foundation, either version 3 of the License, or +** (at your option) any later version. +** +** This program is distributed in the hope that it will be useful, +** but WITHOUT ANY WARRANTY; without even the implied warranty of +** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +** GNU General Public License for more details. +** +** You should have received a copy of the GNU General Public License +** along with this program. If not, see . +*/ + +#pragma once + +// Engine entry point for ICU4C. +// +// RTS_HAS_ICU - ICU C API is linked; include this header and call ICU functions. +// RTS_HAS_ICU_CXX - ICU C++ API (icu::UnicodeString, icu::Locale, ...). +// RTS_HAS_ICU_I18N - Collation, break iteration, converters, and related i18n APIs. +// RTS_HAS_ICU_WINSDK - Windows SDK merged C API via (no C++ API). +// RTS_ICU_DYNAMIC - VC6 uses IcuLoader and runtime UTF conversion exports from icu.dll. +// The full ICU headers/C++ API are not available in this mode. + +#if defined(RTS_HAS_ICU_WINSDK) + +#include + +#elif defined(RTS_HAS_ICU) + +#include +#include +#include +#include +#include + +#if defined(RTS_HAS_ICU_I18N) +#include +#include +#include +#include +#include +#endif + +#if defined(RTS_HAS_ICU_CXX) +#include +#include +#include +#if defined(RTS_HAS_ICU_I18N) +#include +#endif +#endif + +#endif diff --git a/Dependencies/ICU/ICU/utf8.cpp b/Dependencies/ICU/ICU/utf8.cpp new file mode 100644 index 00000000000..94d1763e668 --- /dev/null +++ b/Dependencies/ICU/ICU/utf8.cpp @@ -0,0 +1,573 @@ +/* +** Command & Conquer Generals Zero Hour(tm) +** Copyright 2026 TheSuperHackers +** +** This program is free software: you can redistribute it and/or modify +** it under the terms of the GNU General Public License as published by +** the Free Software Foundation, either version 3 of the License, or +** (at your option) any later version. +** +** This program is distributed in the hope that it will be useful, +** but WITHOUT ANY WARRANTY; without even the implied warranty of +** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +** GNU General Public License for more details. +** +** You should have received a copy of the GNU General Public License +** along with this program. If not, see . +*/ + +#include "ICU/utf8.h" +#include "ICU/IcuSupport.h" +#include "ICU/IcuLoader.h" + +#include + +#include +#include + +#ifdef _WIN32 +#include +#endif + +#if defined(RTS_HAS_ICU) && !defined(RTS_HAS_ICU_WINSDK) +#include +#endif + +namespace +{ + +bool FitsInt(size_t length) +{ + return length <= static_cast(INT_MAX); +} + +#ifdef _WIN32 + +size_t WindowsWideToUtf8Len(const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + assert(false); + return 0; + } + + const int outputLength = WideCharToMultiByte(CP_UTF8, 0, src, static_cast(srcLen), nullptr, 0, nullptr, nullptr); + if (outputLength == 0 && srcLen != 0) + { + assert(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t WindowsWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + assert(false); + return 0; + } + + const int outputLength = WideCharToMultiByte(CP_UTF8, 0, src, static_cast(srcLen), + dest, static_cast(destLen), nullptr, nullptr); + if (outputLength == 0 && srcLen != 0) + { + assert(false); + return 0; + } + + if (static_cast(outputLength) < destLen) + { + dest[outputLength] = '\0'; + } + + return static_cast(outputLength); +} + +size_t WindowsUtf8ToWideLen(const char* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + return UTF8_INVALID; + } + + const int outputLength = MultiByteToWideChar(CP_UTF8, MB_ERR_INVALID_CHARS, src, + static_cast(srcLen), nullptr, 0); + if (outputLength == 0 && srcLen != 0) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +size_t WindowsUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + const int outputLength = MultiByteToWideChar(CP_UTF8, MB_ERR_INVALID_CHARS, src, + static_cast(srcLen), dest, static_cast(destLen)); + if (outputLength == 0 && srcLen != 0) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + if (static_cast(outputLength) < destLen) + { + dest[outputLength] = L'\0'; + } + + return static_cast(outputLength); +} + +#endif + +#if defined(RTS_HAS_ICU) && !defined(RTS_HAS_ICU_WINSDK) + +bool IcuPreflightSucceeded(UErrorCode error) +{ + return U_SUCCESS(error) || error == U_BUFFER_OVERFLOW_ERROR; +} + +bool IcuConversionSucceeded(UErrorCode error) +{ + return U_SUCCESS(error); +} + +#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) + +size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + assert(false); + return 0; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF8WithSub(dest, static_cast(destLen), &outputLength, + reinterpret_cast(src), static_cast(srcLen), 0xFFFD, nullptr, &error); + if (!IcuConversionSucceeded(error)) + { + assert(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + assert(false); + return 0; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF8WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + static_cast(srcLen), 0xFFFD, nullptr, &error); + if (!IcuPreflightSucceeded(error)) + { + assert(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strFromUTF8(reinterpret_cast(dest), static_cast(destLen), &outputLength, + src, static_cast(srcLen), &error); + if (!IcuConversionSucceeded(error)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + return UTF8_INVALID; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strFromUTF8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + if (!IcuPreflightSucceeded(error)) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +#else + +bool WideToUtf16(std::vector& utf16, const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + return false; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strFromUTF32WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + static_cast(srcLen), 0xFFFD, nullptr, &error); + if (!IcuPreflightSucceeded(error)) + { + return false; + } + + utf16.resize(static_cast(outputLength) + 1); + error = U_ZERO_ERROR; + u_strFromUTF32WithSub(&utf16[0], static_cast(utf16.size()), &outputLength, + reinterpret_cast(src), static_cast(srcLen), 0xFFFD, nullptr, &error); + return U_SUCCESS(error); +} + +bool Utf8ToUtf16(std::vector& utf16, const char* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + return false; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strFromUTF8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + if (!IcuPreflightSucceeded(error)) + { + return false; + } + + utf16.resize(static_cast(outputLength) + 1); + error = U_ZERO_ERROR; + u_strFromUTF8(&utf16[0], static_cast(utf16.size()), &outputLength, + src, static_cast(srcLen), &error); + return U_SUCCESS(error); +} + +size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) +{ + std::vector utf16; + if (!WideToUtf16(utf16, src, srcLen)) + { + assert(false); + return 0; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF8WithSub(nullptr, 0, &outputLength, utf16.empty() ? nullptr : &utf16[0], + static_cast(utf16.empty() ? 0 : utf16.size() - 1), 0xFFFD, nullptr, &error); + if (!IcuPreflightSucceeded(error)) + { + assert(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) +{ + std::vector utf16; + if (!Utf8ToUtf16(utf16, src, srcLen)) + { + return UTF8_INVALID; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF32(nullptr, 0, &outputLength, utf16.empty() ? nullptr : &utf16[0], + static_cast(utf16.empty() ? 0 : utf16.size() - 1), &error); + if (!IcuPreflightSucceeded(error)) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(destLen)) + { + assert(false); + return 0; + } + + std::vector utf16; + if (!WideToUtf16(utf16, src, srcLen)) + { + assert(false); + return 0; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF8WithSub(dest, static_cast(destLen), &outputLength, + utf16.empty() ? nullptr : &utf16[0], static_cast(utf16.empty() ? 0 : utf16.size() - 1), + 0xFFFD, nullptr, &error); + if (U_FAILURE(error)) + { + assert(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (!FitsInt(destLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + std::vector utf16; + if (!Utf8ToUtf16(utf16, src, srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF32(reinterpret_cast(dest), static_cast(destLen), &outputLength, + utf16.empty() ? nullptr : &utf16[0], static_cast(utf16.empty() ? 0 : utf16.size() - 1), &error); + if (U_FAILURE(error)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +#endif + +#elif defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + +typedef IcuLoader::Char IcuChar; +typedef IcuLoader::ErrorCode IcuErrorCode; + +enum +{ + ICU_ZERO_ERROR = 0, + ICU_BUFFER_OVERFLOW_ERROR = 15, + ICU_REPLACEMENT_CHARACTER = 0xFFFD +}; + +bool IcuPreflightSucceeded(IcuErrorCode error) +{ + return error <= ICU_ZERO_ERROR || error == ICU_BUFFER_OVERFLOW_ERROR; +} + +bool IcuConversionSucceeded(IcuErrorCode error) +{ + return error <= ICU_ZERO_ERROR; +} + +size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + assert(false); + return 0; + } + + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + if (!IcuLoader::toUtf8WithSub(dest, static_cast(destLen), &outputLength, + reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error)) + { + return WindowsWideToUtf8(dest, destLen, src, srcLen); + } + + if (!IcuConversionSucceeded(error)) + { + assert(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + assert(false); + return 0; + } + + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + if (!IcuLoader::toUtf8WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error)) + { + return WindowsWideToUtf8Len(src, srcLen); + } + + if (!IcuPreflightSucceeded(error)) + { + assert(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + if (!IcuLoader::fromUtf8(reinterpret_cast(dest), static_cast(destLen), &outputLength, + src, static_cast(srcLen), &error)) + { + return WindowsUtf8ToWide(dest, destLen, src, srcLen); + } + + if (!IcuConversionSucceeded(error)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + return UTF8_INVALID; + } + + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + if (!IcuLoader::fromUtf8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error)) + { + return WindowsUtf8ToWideLen(src, srcLen); + } + + if (!IcuPreflightSucceeded(error)) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +#endif + +} // namespace + +size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) +{ + return IcuWideToUtf8Len(src, srcLen); +} + +size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) +{ + return IcuUtf8ToWideLen(src, srcLen); +} + +size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + return IcuWideToUtf8(dest, destLen, src, srcLen); +} + +size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + return IcuUtf8ToWide(dest, destLen, src, srcLen); +} + +// A UTF-8 continuation byte matches 10xxxxxx, so it can never start a sequence. +static bool Utf8_Is_Continuation_Byte(char c) +{ + return (static_cast(c) & 0xC0) == 0x80; +} + +size_t Utf8_Truncate_Len(const char* src, size_t srcLen, size_t maxLen) +{ + if (srcLen <= maxLen) + { + return srcLen; + } + + size_t len = maxLen; + while (len > 0 && Utf8_Is_Continuation_Byte(src[len])) + { + --len; + } + + return len; +} diff --git a/Dependencies/ICU/ICU/utf8.h b/Dependencies/ICU/ICU/utf8.h new file mode 100644 index 00000000000..bf409bda34c --- /dev/null +++ b/Dependencies/ICU/ICU/utf8.h @@ -0,0 +1,47 @@ +/* +** Command & Conquer Generals Zero Hour(tm) +** Copyright 2026 TheSuperHackers +** +** This program is free software: you can redistribute it and/or modify +** it under the terms of the GNU General Public License as published by +** the Free Software Foundation, either version 3 of the License, or +** (at your option) any later version. +** +** This program is distributed in the hope that it will be useful, +** but WITHOUT ANY WARRANTY; without even the implied warranty of +** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +** GNU General Public License for more details. +** +** You should have received a copy of the GNU General Public License +** along with this program. If not, see . +*/ + +#pragma once + +#include +#include + +// UTF-8 <-> wide-character conversion backed by ICU4C. +// Full ICU packages use linked C APIs. Windows SDK and VC6 builds use IcuLoader +// to load icu.dll lazily and cache its UTF conversion exports until unload. +// Windows builds fall back to Win32 CP_UTF8 if the DLL or required exports are missing. +// Include ICU/IcuSupport.h to use the rest of the linked ICU suite from engine code. + +// Returned when UTF-8 input is malformed. Zero is reserved for a successful empty conversion. +const size_t UTF8_INVALID = (size_t)-1; + +// Return the required destination length without counting a null terminator. +size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen); +size_t Utf8_To_Wide_Len(const char* src, size_t srcLen); + +// Convert exactly srcLen source units. The destination is null-terminated when it has spare +// capacity. Wide input that cannot be represented as Unicode is replaced with U+FFFD. Malformed +// UTF-8 returns UTF8_INVALID and clears dest when destLen is nonzero. +size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen); +size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen); + +// Returns the largest length not greater than maxLen at which the srcLen bytes of the UTF-8 string +// src can be cut without splitting a multibyte sequence, by backing off the continuation bytes at +// the cut point. Returns srcLen when the string already fits in maxLen. Returns 0 when no whole +// sequence fits, which is also what malformed UTF-8 yields once it has no lead byte to back off to. +size_t Utf8_Truncate_Len(const char* src, size_t srcLen, size_t maxLen); diff --git a/Dependencies/ICU/README.md b/Dependencies/ICU/README.md new file mode 100644 index 00000000000..c4373d99164 --- /dev/null +++ b/Dependencies/ICU/README.md @@ -0,0 +1,22 @@ +# ICU integration + +`core_icu` provides UTF conversions and the ICU headers available to the selected toolchain. +Include `ICU/utf8.h` for conversions and `ICU/IcuSupport.h` for linked ICU APIs. + +| Build configuration | ICU access | Fallback | +| --- | --- | --- | +| Full ICU package, including Windows | Linked ICU C APIs and available C++ APIs | None needed | +| Windows SDK ICU | `IcuLoader` resolves the conversion exports; other C APIs remain available through delay imports | Win32 `CP_UTF8` when the DLL or exports are unavailable | +| VC6 and Windows builds without an import library | `IcuLoader` loads `icu.dll` and resolves `u_strFromUTF8` and `u_strToUTF8WithSub` | Win32 `CP_UTF8` when the DLL or exports are unavailable | + +VC6 does use ICU when these exports are available. It does not compile against modern ICU headers or expose the full ICU C++ API. + +`IcuLoader` loads on the first conversion and caches both successful and failed attempts. Later calls reuse the result, including in WorldBuilder and other tools without an enclosing application scope. `unload()` releases the DLL, clears the exports, and permits a later call to retry. Both games explicitly unload after engine teardown; tools also have cleanup at normal module shutdown. + +Loading, conversion calls, and unloading use a Windows `CRITICAL_SECTION`, the same primitive used by the engine's critical-section classes. It is available on VC6 and does not require a dependency on WWLib. Calls are serialized, and unload waits for any active conversion before freeing the DLL. Export pointers remain private to the loader. Conversion workers must stop before static destruction begins. + +The critical section initializes on first use, including conversions triggered by startup logging before global constructors have run. An interlocked initialization guard also supports concurrent first calls on VC6; conversion calls use the native critical section. Cleanup is registered when the lock is initialized. + +Conversions in Windows SDK builds use the resolved function pointers, avoiding an extra delay-loader reference. If a caller uses other SDK ICU APIs directly, the SDK delay loader owns its reference independently. + +The loader tries absolute paths in the executable directory and then the Windows system directory. It supports app-local ICU and Unicode installation paths without searching the working directory or `PATH`. A missing DLL or conversion export selects the Win32 fallback; a DLL with missing exports is released immediately. diff --git a/Generals/Code/GameEngine/Source/Common/GameMain.cpp b/Generals/Code/GameEngine/Source/Common/GameMain.cpp index c9dd0a94427..ed8ebb709a3 100644 --- a/Generals/Code/GameEngine/Source/Common/GameMain.cpp +++ b/Generals/Code/GameEngine/Source/Common/GameMain.cpp @@ -31,6 +31,7 @@ #include "Common/FramePacer.h" #include "Common/GameEngine.h" #include "Common/ReplaySimulation.h" +#include "ICU/IcuLoader.h" /** @@ -61,6 +62,11 @@ Int GameMain() delete TheGameEngine; TheGameEngine = nullptr; +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + // Release cached ICU after the engine has stopped its conversion workers. + IcuLoader::unload(); +#endif + return exitcode; } diff --git a/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp b/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp index ed94ec7bf54..fd67cf8b9af 100644 --- a/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp +++ b/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp @@ -31,6 +31,7 @@ #include "Common/FramePacer.h" #include "Common/GameEngine.h" #include "Common/ReplaySimulation.h" +#include "ICU/IcuLoader.h" /** @@ -61,6 +62,11 @@ Int GameMain() delete TheGameEngine; TheGameEngine = nullptr; +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + // Release cached ICU after the engine has stopped its conversion workers. + IcuLoader::unload(); +#endif + return exitcode; } diff --git a/cmake/icu.cmake b/cmake/icu.cmake new file mode 100644 index 00000000000..fa114191dd4 --- /dev/null +++ b/cmake/icu.cmake @@ -0,0 +1,75 @@ +# ICU4C for the engine. +# +# Preference order: +# 1. find_package(ICU) from vcpkg or the system (C + C++ APIs: uc, i18n, data) +# 2. Windows SDK icu.lib for modern MSVC (C API: common + i18n via ) +# 3. Runtime LoadLibrary of OS icu.dll (VC6 and other Windows toolchains without an import lib) +# +# VC6 uses ICU UTF conversion exports through IcuLoader, without ICU headers or import libraries. +# If the DLL or either export is missing, conversion falls back to Win32 CP_UTF8. + +add_library(core_icu STATIC) + +set(RTS_ICU_LINKED FALSE) +set(RTS_ICU_CXX FALSE) +set(RTS_ICU_I18N FALSE) +set(RTS_ICU_WINSDK FALSE) +set(RTS_ICU_DYNAMIC FALSE) + +if(NOT IS_VS6_BUILD) + find_package(ICU QUIET COMPONENTS uc i18n data) + if(NOT ICU_FOUND) + find_package(ICU QUIET COMPONENTS uc i18n) + endif() + if(NOT ICU_FOUND) + find_package(ICU QUIET COMPONENTS uc data) + endif() + if(NOT ICU_FOUND) + find_package(ICU QUIET COMPONENTS uc) + endif() +endif() + +if(ICU_FOUND AND NOT IS_VS6_BUILD) + set(RTS_ICU_LINKED TRUE) + set(RTS_ICU_CXX TRUE) + target_link_libraries(core_icu PUBLIC ICU::uc) + if(TARGET ICU::i18n) + set(RTS_ICU_I18N TRUE) + target_link_libraries(core_icu PUBLIC ICU::i18n) + endif() + if(TARGET ICU::data) + target_link_libraries(core_icu PUBLIC ICU::data) + endif() + target_compile_definitions(core_icu PUBLIC RTS_HAS_ICU RTS_HAS_ICU_CXX) + if(RTS_ICU_I18N) + target_compile_definitions(core_icu PUBLIC RTS_HAS_ICU_I18N) + endif() + message(STATUS "ICU4C: linked via find_package (C++ API enabled)") +elseif(WIN32 AND NOT IS_VS6_BUILD AND NOT MINGW) + find_library(RTS_ICU_WINSDK_LIB NAMES icu) + if(RTS_ICU_WINSDK_LIB) + set(RTS_ICU_LINKED TRUE) + set(RTS_ICU_I18N TRUE) + set(RTS_ICU_WINSDK TRUE) + target_link_libraries(core_icu PUBLIC ${RTS_ICU_WINSDK_LIB} delayimp) + target_link_options(core_icu PUBLIC "/DELAYLOAD:icu.dll") + target_compile_definitions(core_icu PUBLIC RTS_HAS_ICU RTS_HAS_ICU_WINSDK RTS_HAS_ICU_I18N) + message(STATUS "ICU4C: linked via Windows SDK (${RTS_ICU_WINSDK_LIB})") + endif() +endif() + +if(NOT RTS_ICU_LINKED) + if(WIN32) + set(RTS_ICU_DYNAMIC TRUE) + target_compile_definitions(core_icu PUBLIC RTS_ICU_DYNAMIC) + message(STATUS "ICU4C: IcuLoader resolves UTF conversion exports from icu.dll; Win32 fallback if unavailable") + else() + message(FATAL_ERROR "ICU4C is required on non-Windows platforms. Install libicu or enable vcpkg.") + endif() +endif() + +add_feature_info(IcuLinked RTS_ICU_LINKED "Link ICU4C into the engine") +add_feature_info(IcuCxx RTS_ICU_CXX "ICU C++ API available") +add_feature_info(IcuI18n RTS_ICU_I18N "ICU i18n API available") +add_feature_info(IcuWinSdk RTS_ICU_WINSDK "Windows SDK ICU C API") +add_feature_info(IcuDynamic RTS_ICU_DYNAMIC "Load OS icu.dll at runtime") diff --git a/vcpkg.json b/vcpkg.json index 22db4fd2b7a..7b01ee2913f 100644 --- a/vcpkg.json +++ b/vcpkg.json @@ -3,6 +3,7 @@ "builtin-baseline": "9e593bb18ea69cc5095e012465dcd675a822ed0d", "dependencies": [ "zlib", + "icu", "stb" ], "features": {