From 71f27e974a8de0a1ec351380a3dabab1c6333328 Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Wed, 5 Aug 2026 08:55:16 -0600 Subject: [PATCH 01/11] feat(os): Add UTF8 support using ICU --- .../Source/Common/System/AsciiString.cpp | 20 +- .../Source/Common/System/UnicodeString.cpp | 34 +- .../GameSpy/Thread/ThreadUtils.cpp | 53 +- .../Source/WWVegas/WWLib/CMakeLists.txt | 5 + Core/Libraries/Source/WWVegas/WWLib/utf8.cpp | 701 ++++++++++++------ Core/Libraries/Source/WWVegas/WWLib/utf8.h | 34 +- vcpkg.json | 4 + 7 files changed, 550 insertions(+), 301 deletions(-) diff --git a/Core/GameEngine/Source/Common/System/AsciiString.cpp b/Core/GameEngine/Source/Common/System/AsciiString.cpp index 6bf644428ee..31b51000a73 100644 --- a/Core/GameEngine/Source/Common/System/AsciiString.cpp +++ b/Core/GameEngine/Source/Common/System/AsciiString.cpp @@ -305,19 +305,27 @@ char* AsciiString::getBufferForRead(Int len) void AsciiString::translate(const UnicodeString& stringSrc) { validate(); - // TheSuperHackers @fix bobtista 02/04/2026 Implement UTF-8 conversion replacing 7-bit ASCII only implementation + // TheSuperHackers @bugfix CryoTheRenegade 04/08/2026 Convert wide text to UTF-8 with ICU4C. const WideChar* src = stringSrc.str(); - const size_t srcLen = wcslen(src); - const size_t dstLen = Wide_To_Utf8_Len(src, srcLen); - if (dstLen == 0) + const size_t srcLen = stringSrc.getLength(); + const size_t len = Wide_To_Utf8_Len(src, srcLen); + if (len == 0) { clear(); } + else if (len >= static_cast(MAX_LEN)) + { + DEBUG_ASSERTCRASH(false, + ("AsciiString::translate exceeds max string length %d with required UTF-8 length %u", + MAX_LEN, static_cast(len))); + clear(); + } else { - ensureUniqueBufferOfSize((Int)dstLen + 1, false, nullptr, nullptr); - Wide_To_Utf8(peek(), dstLen + 1, src, srcLen); + ensureUniqueBufferOfSize(static_cast(len) + 1, false, nullptr, nullptr); + Wide_To_Utf8(peek(), len + 1, src, srcLen); } + validate(); } diff --git a/Core/GameEngine/Source/Common/System/UnicodeString.cpp b/Core/GameEngine/Source/Common/System/UnicodeString.cpp index c8e12a9b185..afe56a73980 100644 --- a/Core/GameEngine/Source/Common/System/UnicodeString.cpp +++ b/Core/GameEngine/Source/Common/System/UnicodeString.cpp @@ -219,34 +219,32 @@ WideChar* UnicodeString::getBufferForRead(Int len) void UnicodeString::translate(const AsciiString& stringSrc) { validate(); - // TheSuperHackers @fix bobtista 02/04/2026 Convert UTF-8 to wide, replacing the 7-bit ASCII only - // implementation. Data that is not valid UTF-8 (e.g. legacy CP1252) falls back to a 1:1 byte cast - // to preserve the original characters instead of producing replacement characters. + // TheSuperHackers @bugfix CryoTheRenegade 04/08/2026 Convert UTF-8 to wide text with ICU4C. const char* src = stringSrc.str(); - const size_t srcLen = strlen(src); - const size_t dstLen = Utf8_To_Wide_Len(src, srcLen); - if (dstLen != UTF8_INVALID) + const size_t srcLen = stringSrc.getLength(); + const size_t len = Utf8_To_Wide_Len(src, srcLen); + if (srcLen == 0) { - if (dstLen == 0) - { - clear(); - } - else - { - ensureUniqueBufferOfSize((Int)dstLen + 1, false, nullptr, nullptr); - Utf8_To_Wide(peek(), dstLen + 1, src, srcLen); - } + clear(); } - else + else if (len == UTF8_INVALID) { - ensureUniqueBufferOfSize((Int)srcLen + 1, false, nullptr, nullptr); + // Preserve legacy non-UTF-8 data with the original one-byte-to-one-wide-unit behavior. + ensureUniqueBufferOfSize(static_cast(srcLen) + 1, false, nullptr, nullptr); WideChar* buf = peek(); for (size_t i = 0; i < srcLen; ++i) { - buf[i] = (WideChar)(unsigned char)src[i]; + buf[i] = static_cast(static_cast(src[i])); } + buf[srcLen] = 0; } + else + { + ensureUniqueBufferOfSize(static_cast(len) + 1, false, nullptr, nullptr); + Utf8_To_Wide(peek(), len + 1, src, srcLen); + } + validate(); } diff --git a/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp b/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp index 84ce0c9b19a..07920c2075f 100644 --- a/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp +++ b/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp @@ -32,61 +32,54 @@ //------------------------------------------------------------------------- -// TheSuperHackers @refactor bobtista 02/04/2026 Use WWLib UTF-8 functions instead of raw Win32 API calls -std::wstring MultiByteToWideCharSingleLine( const char *orig ) +// TheSuperHackers @refactor CryoTheRenegade 04/08/2026 Use the shared ICU4C UTF conversion functions. +std::wstring MultiByteToWideCharSingleLine( const char* orig ) { const size_t srcLen = strlen(orig); - const size_t dstLen = Utf8_To_Wide_Len(orig, srcLen); - if (dstLen == 0) + const size_t len = Utf8_To_Wide_Len(orig, srcLen); + if (len == 0) + { return std::wstring(); + } + std::wstring ret; - if (dstLen == UTF8_INVALID) + if (len == UTF8_INVALID) { - // Not UTF-8. Fall back to a 1:1 byte cast so legacy data keeps its characters, matching - // UnicodeString::translate. ret.resize(srcLen); for (size_t i = 0; i < srcLen; ++i) { - ret[i] = (WideChar)(unsigned char)orig[i]; + ret[i] = static_cast(static_cast(orig[i])); } } else { - ret.resize(dstLen); - Utf8_To_Wide(&ret[0], dstLen, orig, srcLen); + ret.resize(len); + Utf8_To_Wide(&ret[0], len, orig, srcLen); } - WideChar *c = nullptr; - do - { - c = wcschr(&ret[0], L'\n'); - if (c) - { - *c = L' '; - } - } - while ( c != nullptr ); - do + + for (size_t i = 0; i < ret.size(); ++i) { - c = wcschr(&ret[0], L'\r'); - if (c) + if (ret[i] == L'\n' || ret[i] == L'\x0D') { - *c = L' '; + ret[i] = L' '; } } - while ( c != nullptr ); return ret; } -std::string WideCharStringToMultiByte( const WideChar *orig ) +std::string WideCharStringToMultiByte( const WideChar* orig ) { const size_t srcLen = wcslen(orig); - const size_t dstLen = Wide_To_Utf8_Len(orig, srcLen); - if (dstLen == 0) + const size_t len = Wide_To_Utf8_Len(orig, srcLen); + if (len == 0) + { return std::string(); + } + std::string ret; - ret.resize(dstLen); - Wide_To_Utf8(&ret[0], dstLen, orig, srcLen); + ret.resize(len); + Wide_To_Utf8(&ret[0], len, orig, srcLen); return ret; } diff --git a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt index 23d32b1dc66..a92b2e9307d 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt +++ b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt @@ -180,3 +180,8 @@ target_link_libraries(core_wwlib PRIVATE core_wwcommon corei_always ) + +if(NOT WIN32) + find_package(ICU REQUIRED COMPONENTS uc data) + target_link_libraries(core_wwlib PRIVATE ICU::uc ICU::data) +endif() diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp index 578299fbe3d..61e4ef31ca9 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp +++ b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp @@ -19,269 +19,528 @@ #include "always.h" #include "utf8.h" -// wchar_t is a 16-bit UTF-16 code unit on Windows and a 32-bit UTF-32 codepoint on most other -// platforms. WCHAR_MAX lets us distinguish the two at compile time so the surrogate-pair paths -// are excluded entirely (not just constant-folded) where wchar_t is wide enough to hold a codepoint. -#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) -#define UTF8_WCHAR_IS_UTF16 1 -#else -#define UTF8_WCHAR_IS_UTF16 0 -#endif +#include +#include -static const unsigned int UTF8_CODEPOINT_MAX = 0x10FFFF; -static const unsigned int UTF8_SURROGATE_MIN = 0xD800; -static const unsigned int UTF8_SURROGATE_MAX = 0xDFFF; -static const unsigned int UTF8_REPLACEMENT_CHAR = 0xFFFD; +#ifdef _WIN32 -// Number of UTF-8 bytes required to encode a codepoint. -static size_t Utf8_Encoded_Length(unsigned int cp) +#include + +namespace { - if (cp < 0x80) - { - return 1; - } - if (cp < 0x800) - { - return 2; - } - if (cp < 0x10000) - { - return 3; - } - return 4; + +typedef unsigned short IcuChar; +typedef int IcuChar32; +typedef int IcuErrorCode; +typedef IcuChar* (__cdecl* IcuStrFromUtf8)( + IcuChar*, int, int*, const char*, int, IcuErrorCode*); +typedef char* (__cdecl* IcuStrToUtf8WithSub)( + char*, int, int*, const IcuChar*, int, IcuChar32, int*, IcuErrorCode*); + +enum +{ + ICU_ZERO_ERROR = 0, + ICU_BUFFER_OVERFLOW_ERROR = 15, + ICU_REPLACEMENT_CHARACTER = 0xFFFD +}; + +class WindowsIcuFunctions +{ +public: + WindowsIcuFunctions() : m_fromUtf8(nullptr), m_toUtf8WithSub(nullptr), m_module(nullptr) + { + char systemIcuPath[MAX_PATH]; + const UINT systemDirectoryLength = GetSystemDirectoryA(systemIcuPath, MAX_PATH); + if (systemDirectoryLength > 0 && systemDirectoryLength <= MAX_PATH - sizeof("\\icu.dll")) + { + memcpy(systemIcuPath + systemDirectoryLength, "\\icu.dll", sizeof("\\icu.dll")); + m_module = LoadLibraryA(systemIcuPath); + } + + if (m_module != nullptr) + { + m_fromUtf8 = reinterpret_cast(GetProcAddress(m_module, "u_strFromUTF8")); + m_toUtf8WithSub = reinterpret_cast(GetProcAddress(m_module, "u_strToUTF8WithSub")); + } + + if (m_fromUtf8 == nullptr || m_toUtf8WithSub == nullptr) + { + if (m_module != nullptr) + { + FreeLibrary(m_module); + } + + m_module = nullptr; + m_fromUtf8 = nullptr; + m_toUtf8WithSub = nullptr; + } + } + + bool isAvailable() const + { + return m_module != nullptr; + } + + IcuStrFromUtf8 m_fromUtf8; + IcuStrToUtf8WithSub m_toUtf8WithSub; + +private: + HMODULE m_module; +}; + +// Keep ICU loaded for the process lifetime so conversions remain safe during global destruction. +WindowsIcuFunctions g_icu; + +bool FitsInt(size_t length) +{ + return length <= static_cast(INT_MAX); } -// Encode a codepoint to dest, which is assumed to have room. Returns the number of bytes written. -static size_t Utf8_Encode(char* dest, unsigned int cp) +bool IcuPreflightSucceeded(IcuErrorCode error) { - if (cp < 0x80) - { - dest[0] = (char)cp; - return 1; - } - if (cp < 0x800) - { - dest[0] = (char)(0xC0 | (cp >> 6)); - dest[1] = (char)(0x80 | (cp & 0x3F)); - return 2; - } - if (cp < 0x10000) - { - dest[0] = (char)(0xE0 | (cp >> 12)); - dest[1] = (char)(0x80 | ((cp >> 6) & 0x3F)); - dest[2] = (char)(0x80 | (cp & 0x3F)); - return 3; - } - dest[0] = (char)(0xF0 | (cp >> 18)); - dest[1] = (char)(0x80 | ((cp >> 12) & 0x3F)); - dest[2] = (char)(0x80 | ((cp >> 6) & 0x3F)); - dest[3] = (char)(0x80 | (cp & 0x3F)); - return 4; + return error <= ICU_ZERO_ERROR || error == ICU_BUFFER_OVERFLOW_ERROR; } -// Decode one UTF-8 sequence at src, with srcLen bytes remaining. On success returns the number of -// bytes consumed (1-4) and sets cp. Returns 0 on any malformed, overlong, out-of-range or surrogate -// encoding. -static size_t Utf8_Decode(const char* src, size_t srcLen, unsigned int& cp) +bool IcuConversionSucceeded(IcuErrorCode error) { - const unsigned char lead = (unsigned char)src[0]; - if (lead < 0x80) - { - cp = lead; - return 1; - } + return error <= ICU_ZERO_ERROR; +} - size_t count; - unsigned int lowerBound; - if ((lead & 0xE0) == 0xC0) - { - count = 2; - cp = lead & 0x1F; - lowerBound = 0x80; - } - else if ((lead & 0xF0) == 0xE0) - { - count = 3; - cp = lead & 0x0F; - lowerBound = 0x800; - } - else if ((lead & 0xF8) == 0xF0) - { - count = 4; - cp = lead & 0x07; - lowerBound = 0x10000; - } - else - { - return 0; // a continuation byte or a 5/6-byte form cannot start a sequence - } +size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + WWASSERT(false); + return 0; + } - if (srcLen < count) - { - return 0; // truncated sequence - } - for (size_t i = 1; i < count; ++i) - { - const unsigned char trail = (unsigned char)src[i]; - if ((trail & 0xC0) != 0x80) - { - return 0; // not a continuation byte - } - cp = (cp << 6) | (trail & 0x3F); - } + WWASSERT(sizeof(wchar_t) == sizeof(IcuChar)); + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + g_icu.m_toUtf8WithSub(dest, static_cast(destLen), &outputLength, + reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); + if (!IcuConversionSucceeded(error)) + { + WWASSERT(false); + return 0; + } - if (cp < lowerBound || cp > UTF8_CODEPOINT_MAX || (cp >= UTF8_SURROGATE_MIN && cp <= UTF8_SURROGATE_MAX)) - { - return 0; // overlong, out of range, or a surrogate codepoint - } - return count; + return static_cast(outputLength); } -// Read one codepoint at src, with srcLen wide characters remaining. Returns the number of wide -// characters consumed (1-2) and sets cp. Combines UTF-16 surrogate pairs where wchar_t is 16-bit; -// treats each element as a whole codepoint where wchar_t is 32-bit. Wide data that has no UTF-8 -// representation is reported as U+FFFD, so the encoder never emits a sequence that the decoder -// would reject. -static size_t Wide_Read(const wchar_t* src, size_t srcLen, unsigned int& cp) +size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) { - size_t consumed = 1; -#if UTF8_WCHAR_IS_UTF16 - cp = (unsigned int)src[0] & 0xFFFF; - if (cp >= UTF8_SURROGATE_MIN && cp <= 0xDBFF && srcLen > 1) - { - const unsigned int low = (unsigned int)src[1] & 0xFFFF; - if (low >= 0xDC00 && low <= UTF8_SURROGATE_MAX) - { - cp = 0x10000 + ((cp - UTF8_SURROGATE_MIN) << 10) + (low - 0xDC00); - consumed = 2; - } - } + if (!FitsInt(srcLen)) + { + WWASSERT(false); + return 0; + } + + WWASSERT(sizeof(wchar_t) == sizeof(IcuChar)); + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + g_icu.m_toUtf8WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); + if (!IcuPreflightSucceeded(error)) + { + WWASSERT(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + WWASSERT(sizeof(wchar_t) == sizeof(IcuChar)); + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + g_icu.m_fromUtf8(reinterpret_cast(dest), static_cast(destLen), &outputLength, + src, static_cast(srcLen), &error); + if (!IcuConversionSucceeded(error)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + return UTF8_INVALID; + } + + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + g_icu.m_fromUtf8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + if (!IcuPreflightSucceeded(error)) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +size_t WindowsWideToUtf8Len(const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + WWASSERT(false); + return 0; + } + + const int outputLength = WideCharToMultiByte(CP_UTF8, 0, src, static_cast(srcLen), nullptr, 0, nullptr, nullptr); + if (outputLength == 0 && srcLen != 0) + { + WWASSERT(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t WindowsWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + WWASSERT(false); + return 0; + } + + const int outputLength = WideCharToMultiByte(CP_UTF8, 0, src, static_cast(srcLen), + dest, static_cast(destLen), nullptr, nullptr); + if (outputLength == 0 && srcLen != 0) + { + WWASSERT(false); + return 0; + } + + if (static_cast(outputLength) < destLen) + { + dest[outputLength] = '\0'; + } + + return static_cast(outputLength); +} + +size_t WindowsUtf8ToWideLen(const char* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + return UTF8_INVALID; + } + + const int outputLength = MultiByteToWideChar(CP_UTF8, MB_ERR_INVALID_CHARS, src, + static_cast(srcLen), nullptr, 0); + if (outputLength == 0 && srcLen != 0) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +size_t WindowsUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + const int outputLength = MultiByteToWideChar(CP_UTF8, MB_ERR_INVALID_CHARS, src, + static_cast(srcLen), dest, static_cast(destLen)); + if (outputLength == 0 && srcLen != 0) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + if (static_cast(outputLength) < destLen) + { + dest[outputLength] = L'\0'; + } + + return static_cast(outputLength); +} + +} // namespace + +size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) +{ + if (g_icu.isAvailable()) + { + return IcuWideToUtf8Len(src, srcLen); + } + + return WindowsWideToUtf8Len(src, srcLen); +} + +size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) +{ + if (g_icu.isAvailable()) + { + return IcuUtf8ToWideLen(src, srcLen); + } + + return WindowsUtf8ToWideLen(src, srcLen); +} + +size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (g_icu.isAvailable()) + { + return IcuWideToUtf8(dest, destLen, src, srcLen); + } + + return WindowsWideToUtf8(dest, destLen, src, srcLen); +} + +size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (g_icu.isAvailable()) + { + return IcuUtf8ToWide(dest, destLen, src, srcLen); + } + + return WindowsUtf8ToWide(dest, destLen, src, srcLen); +} + #else - (void)srcLen; - cp = (unsigned int)src[0]; -#endif - if (cp > UTF8_CODEPOINT_MAX || (cp >= UTF8_SURROGATE_MIN && cp <= UTF8_SURROGATE_MAX)) - { - cp = UTF8_REPLACEMENT_CHAR; - } - return consumed; + +#include + +#include + +namespace +{ + +bool FitsIcuLength(size_t length) +{ + return length <= static_cast(INT32_MAX); +} + +bool IcuPreflightSucceeded(UErrorCode error) +{ + return U_SUCCESS(error) || error == U_BUFFER_OVERFLOW_ERROR; } -// Number of wide characters required to store a codepoint. -static size_t Wide_Encoded_Length(unsigned int cp) +bool WideToUtf16(std::vector& utf16, const wchar_t* src, size_t srcLen) { -#if UTF8_WCHAR_IS_UTF16 - return (cp >= 0x10000) ? 2 : 1; + if (!FitsIcuLength(srcLen)) + { + return false; + } + +#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) + utf16.assign(reinterpret_cast(src), reinterpret_cast(src) + srcLen); + utf16.push_back(0); + return true; #else - (void)cp; - return 1; + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strFromUTF32WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + static_cast(srcLen), 0xFFFD, nullptr, &error); + if (!IcuPreflightSucceeded(error)) + { + return false; + } + + utf16.resize(static_cast(outputLength) + 1); + error = U_ZERO_ERROR; + u_strFromUTF32WithSub(&utf16[0], static_cast(utf16.size()), &outputLength, + reinterpret_cast(src), static_cast(srcLen), 0xFFFD, nullptr, &error); + return U_SUCCESS(error); #endif } -// Write one codepoint to a wide buffer, which is assumed to have room. Returns wide characters written. -static size_t Wide_Write(wchar_t* dest, unsigned int cp) +bool Utf8ToUtf16(std::vector& utf16, const char* src, size_t srcLen) { -#if UTF8_WCHAR_IS_UTF16 - if (cp >= 0x10000) - { - cp -= 0x10000; - dest[0] = (wchar_t)(0xD800 + (cp >> 10)); - dest[1] = (wchar_t)(0xDC00 + (cp & 0x3FF)); - return 2; - } -#endif - dest[0] = (wchar_t)cp; - return 1; + if (!FitsIcuLength(srcLen)) + { + return false; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strFromUTF8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + if (!IcuPreflightSucceeded(error)) + { + return false; + } + + utf16.resize(static_cast(outputLength) + 1); + error = U_ZERO_ERROR; + u_strFromUTF8(&utf16[0], static_cast(utf16.size()), &outputLength, + src, static_cast(srcLen), &error); + return U_SUCCESS(error); } +} // namespace + size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) { - size_t needed = 0; - size_t i = 0; - while (i < srcLen) - { - unsigned int cp; - i += Wide_Read(src + i, srcLen - i, cp); - needed += Utf8_Encoded_Length(cp); - } - return needed; + std::vector utf16; + if (!WideToUtf16(utf16, src, srcLen)) + { + WWASSERT(false); + return 0; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF8WithSub(nullptr, 0, &outputLength, utf16.empty() ? nullptr : &utf16[0], + static_cast(utf16.empty() ? 0 : utf16.size() - 1), 0xFFFD, nullptr, &error); + if (!IcuPreflightSucceeded(error)) + { + WWASSERT(false); + return 0; + } + + return static_cast(outputLength); } size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) { - size_t needed = 0; - size_t i = 0; - while (i < srcLen) - { - unsigned int cp; - const size_t consumed = Utf8_Decode(src + i, srcLen - i, cp); - if (consumed == 0) - { - return UTF8_INVALID; - } - i += consumed; - needed += Wide_Encoded_Length(cp); - } - return needed; + std::vector utf16; + if (!Utf8ToUtf16(utf16, src, srcLen)) + { + return UTF8_INVALID; + } + +#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) + return utf16.empty() ? 0 : utf16.size() - 1; +#else + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF32(nullptr, 0, &outputLength, utf16.empty() ? nullptr : &utf16[0], + static_cast(utf16.empty() ? 0 : utf16.size() - 1), &error); + if (!IcuPreflightSucceeded(error)) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); +#endif } size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) { - size_t needed = 0; - size_t out = 0; - size_t i = 0; - while (i < srcLen) - { - unsigned int cp; - i += Wide_Read(src + i, srcLen - i, cp); - const size_t need = Utf8_Encoded_Length(cp); - // Stop writing at the first codepoint that does not fit, but keep counting for the caller. - if (needed == out && out + need <= destLen) - { - out += Utf8_Encode(dest + out, cp); - } - needed += need; - } - if (out < destLen) - { - dest[out] = '\0'; - } - return needed; + if (!FitsIcuLength(destLen)) + { + WWASSERT(false); + return 0; + } + + std::vector utf16; + if (!WideToUtf16(utf16, src, srcLen)) + { + WWASSERT(false); + return 0; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF8WithSub(dest, static_cast(destLen), &outputLength, + utf16.empty() ? nullptr : &utf16[0], static_cast(utf16.empty() ? 0 : utf16.size() - 1), + 0xFFFD, nullptr, &error); + if (U_FAILURE(error)) + { + WWASSERT(false); + return 0; + } + + return static_cast(outputLength); } size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) { - size_t needed = 0; - size_t out = 0; - size_t i = 0; - while (i < srcLen) - { - unsigned int cp; - const size_t consumed = Utf8_Decode(src + i, srcLen - i, cp); - if (consumed == 0) - { - if (destLen > 0) - { - dest[0] = L'\0'; - } - return UTF8_INVALID; - } - i += consumed; - const size_t need = Wide_Encoded_Length(cp); - // Stop writing at the first codepoint that does not fit, but keep counting for the caller. - if (needed == out && out + need <= destLen) - { - out += Wide_Write(dest + out, cp); - } - needed += need; - } - if (out < destLen) - { - dest[out] = L'\0'; - } - return needed; + if (!FitsIcuLength(destLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + std::vector utf16; + if (!Utf8ToUtf16(utf16, src, srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + +#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) + const size_t outputLength = utf16.empty() ? 0 : utf16.size() - 1; + if (outputLength > destLen) + { + WWASSERT(false); + return UTF8_INVALID; + } + + for (size_t i = 0; i < outputLength; ++i) + { + dest[i] = static_cast(utf16[i]); + } + + if (outputLength < destLen) + { + dest[outputLength] = L'\0'; + } + + return outputLength; +#else + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF32(reinterpret_cast(dest), static_cast(destLen), &outputLength, + utf16.empty() ? nullptr : &utf16[0], static_cast(utf16.empty() ? 0 : utf16.size() - 1), &error); + if (U_FAILURE(error)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + return static_cast(outputLength); +#endif } +#endif + // A UTF-8 continuation byte matches 10xxxxxx, so it can never start a sequence. static bool Utf8_Is_Continuation_Byte(char c) { diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.h b/Core/Libraries/Source/WWVegas/WWLib/utf8.h index 0a4f92f95ee..d8d332b5815 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.h +++ b/Core/Libraries/Source/WWVegas/WWLib/utf8.h @@ -21,39 +21,21 @@ #include #include -// UTF-8 <-> wide-character transcoding, hand-rolled per RFC 3629, using no platform text APIs. -// The wide side is wchar_t, whose width is platform-dependent: on Windows it is a 16-bit UTF-16 -// code unit (astral codepoints use surrogate pairs); on most other platforms it is a 32-bit -// UTF-32 codepoint. Both are handled transparently based on the width of wchar_t. +// UTF-8 <-> wide-character conversion backed by ICU4C. Windows dynamically uses the ICU4C +// implementation shipped with the operating system and falls back to the native text APIs when +// ICU is unavailable. Other platforms link ICU4C's common library. -// Returned by the decoding functions when the source is not well-formed UTF-8. A return of 0 means -// an empty result, which is a success and must not be confused with a decoding failure. +// Returned when UTF-8 input is malformed. Zero is reserved for a successful empty conversion. const size_t UTF8_INVALID = (size_t)-1; -// Returns the number of UTF-8 bytes needed for the UTF-8 representation of srcLen wide characters -// from src, not counting a null terminator. Returns 0 if srcLen is 0. Wide values that have no -// UTF-8 representation are counted as U+FFFD. +// Return the required destination length without counting a null terminator. size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen); - -// Returns the number of wide characters needed for the wide representation of srcLen bytes from the -// UTF-8 string src, not counting a null terminator. Returns 0 if srcLen is 0, or UTF8_INVALID if -// src is not well-formed UTF-8. size_t Utf8_To_Wide_Len(const char* src, size_t srcLen); -// Converts srcLen wide characters from src to UTF-8. destLen is the destination buffer capacity in -// bytes. Writes a null terminator if room remains, otherwise not. Wide values that have no UTF-8 -// representation are written as U+FFFD, so the output always decodes back through Utf8_To_Wide. -// Returns the number of bytes the whole conversion needs, not counting a null terminator. A return -// greater than destLen means the output was truncated on a codepoint boundary; retry with that many -// bytes plus one for the terminator. Pass destLen 0 to measure without writing. +// Convert exactly srcLen source units. The destination is null-terminated when it has spare +// capacity. Wide input that cannot be represented as Unicode is replaced with U+FFFD. Malformed +// UTF-8 returns UTF8_INVALID and clears dest when destLen is nonzero. size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen); - -// Converts srcLen bytes from the UTF-8 string src to wide characters. destLen is the destination -// buffer capacity in wide characters. Writes a null terminator if room remains, otherwise not. -// Returns the number of wide characters the whole conversion needs, not counting a null terminator. -// A return greater than destLen means the output was truncated on a codepoint boundary; retry with -// that many wide characters plus one for the terminator. Pass destLen 0 to measure without writing. -// Returns UTF8_INVALID if src is not well-formed UTF-8, setting dest[0] to L'\0' if destLen > 0. size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen); // Returns the largest length not greater than maxLen at which the srcLen bytes of the UTF-8 string diff --git a/vcpkg.json b/vcpkg.json index 22db4fd2b7a..e07a95526c3 100644 --- a/vcpkg.json +++ b/vcpkg.json @@ -3,6 +3,10 @@ "builtin-baseline": "9e593bb18ea69cc5095e012465dcd675a822ed0d", "dependencies": [ "zlib", + { + "name": "icu", + "platform": "!windows" + }, "stb" ], "features": { From 7966466a4bfc2d5642b24c7b9258dc7007a9628d Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Thu, 20 Aug 2026 17:33:45 -0600 Subject: [PATCH 02/11] feat(icu): Link the full ICU4C suite across toolchains Use vcpkg or the Windows SDK C API on modern builds, keep VC6 on runtime LoadLibrary, and only probe system icu.dll for the delay-loaded SDK path. --- CMakeLists.txt | 1 + Core/GameEngine/CMakeLists.txt | 1 + .../Source/WWVegas/WWLib/CMakeLists.txt | 8 +- .../Source/WWVegas/WWLib/IcuSupport.h | 58 ++ Core/Libraries/Source/WWVegas/WWLib/utf8.cpp | 559 +++++++++++------- Core/Libraries/Source/WWVegas/WWLib/utf8.h | 7 +- cmake/icu.cmake | 74 +++ vcpkg.json | 5 +- 8 files changed, 481 insertions(+), 232 deletions(-) create mode 100644 Core/Libraries/Source/WWVegas/WWLib/IcuSupport.h create mode 100644 cmake/icu.cmake diff --git a/CMakeLists.txt b/CMakeLists.txt index d838f64fde8..9e9c4c5d09f 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -65,6 +65,7 @@ else() endif() include(cmake/config.cmake) +include(cmake/icu.cmake) include(cmake/gamespy.cmake) include(cmake/lzhl.cmake) include(cmake/stb.cmake) diff --git a/Core/GameEngine/CMakeLists.txt b/Core/GameEngine/CMakeLists.txt index ba6c63ac21e..28af20fbde9 100644 --- a/Core/GameEngine/CMakeLists.txt +++ b/Core/GameEngine/CMakeLists.txt @@ -1212,6 +1212,7 @@ target_link_libraries(corei_gameengine_public INTERFACE core_browserdispatch dbghelploader #core_wwvegas + core_icu d3d8lib gamespy::gamespy stlport diff --git a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt index a92b2e9307d..f042507c773 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt +++ b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt @@ -43,6 +43,7 @@ set(WWLIB_SRC #global.h hash.cpp hash.h + IcuSupport.h hashcalc.h HASHLIST.h #hashtab.h @@ -181,7 +182,6 @@ target_link_libraries(core_wwlib PRIVATE corei_always ) -if(NOT WIN32) - find_package(ICU REQUIRED COMPONENTS uc data) - target_link_libraries(core_wwlib PRIVATE ICU::uc ICU::data) -endif() +target_link_libraries(core_wwlib PUBLIC + core_icu +) diff --git a/Core/Libraries/Source/WWVegas/WWLib/IcuSupport.h b/Core/Libraries/Source/WWVegas/WWLib/IcuSupport.h new file mode 100644 index 00000000000..6ce7ff720f2 --- /dev/null +++ b/Core/Libraries/Source/WWVegas/WWLib/IcuSupport.h @@ -0,0 +1,58 @@ +/* +** Command & Conquer Generals Zero Hour(tm) +** Copyright 2026 TheSuperHackers +** +** This program is free software: you can redistribute it and/or modify +** it under the terms of the GNU General Public License as published by +** the Free Software Foundation, either version 3 of the License, or +** (at your option) any later version. +** +** This program is distributed in the hope that it will be useful, +** but WITHOUT ANY WARRANTY; without even the implied warranty of +** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +** GNU General Public License for more details. +** +** You should have received a copy of the GNU General Public License +** along with this program. If not, see . +*/ + +#pragma once + +// Engine entry point for ICU4C. +// +// RTS_HAS_ICU - ICU C API is linked; include this header and call ICU functions. +// RTS_HAS_ICU_CXX - ICU C++ API (icu::UnicodeString, icu::Locale, ...). +// RTS_HAS_ICU_I18N - Collation, break iteration, converters, and related i18n APIs. +// RTS_HAS_ICU_WINSDK - Windows SDK merged C API via (no C++ API). +// RTS_ICU_DYNAMIC - No import library; utf8.cpp LoadLibrary's OS icu.dll (VC6). + +#if defined(RTS_HAS_ICU_WINSDK) + +#include + +#elif defined(RTS_HAS_ICU) + +#include +#include +#include +#include +#include + +#if defined(RTS_HAS_ICU_I18N) +#include +#include +#include +#include +#include +#endif + +#if defined(RTS_HAS_ICU_CXX) +#include +#include +#include +#if defined(RTS_HAS_ICU_I18N) +#include +#endif +#endif + +#endif diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp index 61e4ef31ca9..0030718f278 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp +++ b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp @@ -22,180 +22,49 @@ #include #include +#if defined(RTS_HAS_ICU_WINSDK) +#include +#include +#elif defined(RTS_HAS_ICU) +#include #ifdef _WIN32 - #include +#endif +#include +#elif defined(_WIN32) +#include +#endif namespace { -typedef unsigned short IcuChar; -typedef int IcuChar32; -typedef int IcuErrorCode; -typedef IcuChar* (__cdecl* IcuStrFromUtf8)( - IcuChar*, int, int*, const char*, int, IcuErrorCode*); -typedef char* (__cdecl* IcuStrToUtf8WithSub)( - char*, int, int*, const IcuChar*, int, IcuChar32, int*, IcuErrorCode*); - -enum -{ - ICU_ZERO_ERROR = 0, - ICU_BUFFER_OVERFLOW_ERROR = 15, - ICU_REPLACEMENT_CHARACTER = 0xFFFD -}; - -class WindowsIcuFunctions -{ -public: - WindowsIcuFunctions() : m_fromUtf8(nullptr), m_toUtf8WithSub(nullptr), m_module(nullptr) - { - char systemIcuPath[MAX_PATH]; - const UINT systemDirectoryLength = GetSystemDirectoryA(systemIcuPath, MAX_PATH); - if (systemDirectoryLength > 0 && systemDirectoryLength <= MAX_PATH - sizeof("\\icu.dll")) - { - memcpy(systemIcuPath + systemDirectoryLength, "\\icu.dll", sizeof("\\icu.dll")); - m_module = LoadLibraryA(systemIcuPath); - } - - if (m_module != nullptr) - { - m_fromUtf8 = reinterpret_cast(GetProcAddress(m_module, "u_strFromUTF8")); - m_toUtf8WithSub = reinterpret_cast(GetProcAddress(m_module, "u_strToUTF8WithSub")); - } - - if (m_fromUtf8 == nullptr || m_toUtf8WithSub == nullptr) - { - if (m_module != nullptr) - { - FreeLibrary(m_module); - } - - m_module = nullptr; - m_fromUtf8 = nullptr; - m_toUtf8WithSub = nullptr; - } - } - - bool isAvailable() const - { - return m_module != nullptr; - } - - IcuStrFromUtf8 m_fromUtf8; - IcuStrToUtf8WithSub m_toUtf8WithSub; - -private: - HMODULE m_module; -}; - -// Keep ICU loaded for the process lifetime so conversions remain safe during global destruction. -WindowsIcuFunctions g_icu; - bool FitsInt(size_t length) { return length <= static_cast(INT_MAX); } -bool IcuPreflightSucceeded(IcuErrorCode error) -{ - return error <= ICU_ZERO_ERROR || error == ICU_BUFFER_OVERFLOW_ERROR; -} - -bool IcuConversionSucceeded(IcuErrorCode error) -{ - return error <= ICU_ZERO_ERROR; -} - -size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) -{ - if (!FitsInt(destLen) || !FitsInt(srcLen)) - { - WWASSERT(false); - return 0; - } - - WWASSERT(sizeof(wchar_t) == sizeof(IcuChar)); - IcuErrorCode error = ICU_ZERO_ERROR; - int outputLength = 0; - g_icu.m_toUtf8WithSub(dest, static_cast(destLen), &outputLength, - reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); - if (!IcuConversionSucceeded(error)) - { - WWASSERT(false); - return 0; - } - - return static_cast(outputLength); -} - -size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) -{ - if (!FitsInt(srcLen)) - { - WWASSERT(false); - return 0; - } - - WWASSERT(sizeof(wchar_t) == sizeof(IcuChar)); - IcuErrorCode error = ICU_ZERO_ERROR; - int outputLength = 0; - g_icu.m_toUtf8WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), - static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); - if (!IcuPreflightSucceeded(error)) - { - WWASSERT(false); - return 0; - } - - return static_cast(outputLength); -} +#ifdef _WIN32 -size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +bool LoadSystemIcu() { - if (!FitsInt(destLen) || !FitsInt(srcLen)) - { - if (destLen > 0) - { - dest[0] = L'\0'; - } - - return UTF8_INVALID; - } - - WWASSERT(sizeof(wchar_t) == sizeof(IcuChar)); - IcuErrorCode error = ICU_ZERO_ERROR; - int outputLength = 0; - g_icu.m_fromUtf8(reinterpret_cast(dest), static_cast(destLen), &outputLength, - src, static_cast(srcLen), &error); - if (!IcuConversionSucceeded(error)) + static bool attempted = false; + static bool available = false; + if (attempted) { - if (destLen > 0) - { - dest[0] = L'\0'; - } - - return UTF8_INVALID; + return available; } - return static_cast(outputLength); -} - -size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) -{ - if (!FitsInt(srcLen)) - { - return UTF8_INVALID; - } + attempted = true; - IcuErrorCode error = ICU_ZERO_ERROR; - int outputLength = 0; - g_icu.m_fromUtf8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); - if (!IcuPreflightSucceeded(error)) + char systemIcuPath[MAX_PATH]; + const UINT systemDirectoryLength = GetSystemDirectoryA(systemIcuPath, MAX_PATH); + if (systemDirectoryLength > 0 && systemDirectoryLength <= MAX_PATH - sizeof("\\icu.dll")) { - return UTF8_INVALID; + memcpy(systemIcuPath + systemDirectoryLength, "\\icu.dll", sizeof("\\icu.dll")); + available = LoadLibraryA(systemIcuPath) != nullptr; } - return static_cast(outputLength); + return available; } size_t WindowsWideToUtf8Len(const wchar_t* src, size_t srcLen) @@ -289,79 +158,120 @@ size_t WindowsUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t return static_cast(outputLength); } -} // namespace +#endif -size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) +#if defined(RTS_HAS_ICU) + +bool IcuPreflightSucceeded(UErrorCode error) { - if (g_icu.isAvailable()) - { - return IcuWideToUtf8Len(src, srcLen); - } + return U_SUCCESS(error) || error == U_BUFFER_OVERFLOW_ERROR; +} - return WindowsWideToUtf8Len(src, srcLen); +bool IcuConversionSucceeded(UErrorCode error) +{ + return U_SUCCESS(error); } -size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) +#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) + +size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) { - if (g_icu.isAvailable()) + if (!FitsInt(destLen) || !FitsInt(srcLen)) { - return IcuUtf8ToWideLen(src, srcLen); + WWASSERT(false); + return 0; } - return WindowsUtf8ToWideLen(src, srcLen); -} - -size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) -{ - if (g_icu.isAvailable()) + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF8WithSub(dest, static_cast(destLen), &outputLength, + reinterpret_cast(src), static_cast(srcLen), 0xFFFD, nullptr, &error); + if (!IcuConversionSucceeded(error)) { - return IcuWideToUtf8(dest, destLen, src, srcLen); + WWASSERT(false); + return 0; } - return WindowsWideToUtf8(dest, destLen, src, srcLen); + return static_cast(outputLength); } -size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) { - if (g_icu.isAvailable()) + if (!FitsInt(srcLen)) { - return IcuUtf8ToWide(dest, destLen, src, srcLen); + WWASSERT(false); + return 0; } - return WindowsUtf8ToWide(dest, destLen, src, srcLen); + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF8WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + static_cast(srcLen), 0xFFFD, nullptr, &error); + if (!IcuPreflightSucceeded(error)) + { + WWASSERT(false); + return 0; + } + + return static_cast(outputLength); } -#else +size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } -#include + return UTF8_INVALID; + } -#include + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strFromUTF8(reinterpret_cast(dest), static_cast(destLen), &outputLength, + src, static_cast(srcLen), &error); + if (!IcuConversionSucceeded(error)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } -namespace -{ + return UTF8_INVALID; + } -bool FitsIcuLength(size_t length) -{ - return length <= static_cast(INT32_MAX); + return static_cast(outputLength); } -bool IcuPreflightSucceeded(UErrorCode error) +size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) { - return U_SUCCESS(error) || error == U_BUFFER_OVERFLOW_ERROR; + if (!FitsInt(srcLen)) + { + return UTF8_INVALID; + } + + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strFromUTF8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + if (!IcuPreflightSucceeded(error)) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); } +#else + bool WideToUtf16(std::vector& utf16, const wchar_t* src, size_t srcLen) { - if (!FitsIcuLength(srcLen)) + if (!FitsInt(srcLen)) { return false; } -#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) - utf16.assign(reinterpret_cast(src), reinterpret_cast(src) + srcLen); - utf16.push_back(0); - return true; -#else UErrorCode error = U_ZERO_ERROR; int32_t outputLength = 0; u_strFromUTF32WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), @@ -376,12 +286,11 @@ bool WideToUtf16(std::vector& utf16, const wchar_t* src, size_t srcLen) u_strFromUTF32WithSub(&utf16[0], static_cast(utf16.size()), &outputLength, reinterpret_cast(src), static_cast(srcLen), 0xFFFD, nullptr, &error); return U_SUCCESS(error); -#endif } bool Utf8ToUtf16(std::vector& utf16, const char* src, size_t srcLen) { - if (!FitsIcuLength(srcLen)) + if (!FitsInt(srcLen)) { return false; } @@ -401,9 +310,7 @@ bool Utf8ToUtf16(std::vector& utf16, const char* src, size_t srcLen) return U_SUCCESS(error); } -} // namespace - -size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) +size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) { std::vector utf16; if (!WideToUtf16(utf16, src, srcLen)) @@ -425,7 +332,7 @@ size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) return static_cast(outputLength); } -size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) +size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) { std::vector utf16; if (!Utf8ToUtf16(utf16, src, srcLen)) @@ -433,9 +340,6 @@ size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) return UTF8_INVALID; } -#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) - return utf16.empty() ? 0 : utf16.size() - 1; -#else UErrorCode error = U_ZERO_ERROR; int32_t outputLength = 0; u_strToUTF32(nullptr, 0, &outputLength, utf16.empty() ? nullptr : &utf16[0], @@ -446,12 +350,11 @@ size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) } return static_cast(outputLength); -#endif } -size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) { - if (!FitsIcuLength(destLen)) + if (!FitsInt(destLen)) { WWASSERT(false); return 0; @@ -478,9 +381,9 @@ size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLe return static_cast(outputLength); } -size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) { - if (!FitsIcuLength(destLen)) + if (!FitsInt(destLen)) { if (destLen > 0) { @@ -501,31 +404,150 @@ size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLe return UTF8_INVALID; } -#if defined(WCHAR_MAX) && (WCHAR_MAX <= 0xFFFF) - const size_t outputLength = utf16.empty() ? 0 : utf16.size() - 1; - if (outputLength > destLen) + UErrorCode error = U_ZERO_ERROR; + int32_t outputLength = 0; + u_strToUTF32(reinterpret_cast(dest), static_cast(destLen), &outputLength, + utf16.empty() ? nullptr : &utf16[0], static_cast(utf16.empty() ? 0 : utf16.size() - 1), &error); + if (U_FAILURE(error)) { - WWASSERT(false); + if (destLen > 0) + { + dest[0] = L'\0'; + } + return UTF8_INVALID; } - for (size_t i = 0; i < outputLength; ++i) + return static_cast(outputLength); +} + +#endif + +#elif defined(RTS_ICU_DYNAMIC) + +typedef unsigned short IcuChar; +typedef int IcuChar32; +typedef int IcuErrorCode; +typedef IcuChar* (__cdecl* IcuStrFromUtf8)( + IcuChar*, int, int*, const char*, int, IcuErrorCode*); +typedef char* (__cdecl* IcuStrToUtf8WithSub)( + char*, int, int*, const IcuChar*, int, IcuChar32, int*, IcuErrorCode*); + +enum +{ + ICU_ZERO_ERROR = 0, + ICU_BUFFER_OVERFLOW_ERROR = 15, + ICU_REPLACEMENT_CHARACTER = 0xFFFD +}; + +class WindowsIcuFunctions +{ +public: + WindowsIcuFunctions() : m_fromUtf8(nullptr), m_toUtf8WithSub(nullptr), m_module(nullptr) + { + if (!LoadSystemIcu()) + { + return; + } + + m_module = GetModuleHandleA("icu.dll"); + if (m_module != nullptr) + { + m_fromUtf8 = reinterpret_cast(GetProcAddress(m_module, "u_strFromUTF8")); + m_toUtf8WithSub = reinterpret_cast(GetProcAddress(m_module, "u_strToUTF8WithSub")); + } + + if (m_fromUtf8 == nullptr || m_toUtf8WithSub == nullptr) + { + m_module = nullptr; + m_fromUtf8 = nullptr; + m_toUtf8WithSub = nullptr; + } + } + + bool isAvailable() const { - dest[i] = static_cast(utf16[i]); + return m_fromUtf8 != nullptr && m_toUtf8WithSub != nullptr; } - if (outputLength < destLen) + IcuStrFromUtf8 m_fromUtf8; + IcuStrToUtf8WithSub m_toUtf8WithSub; + +private: + HMODULE m_module; +}; + +WindowsIcuFunctions g_icu; + +bool IcuPreflightSucceeded(IcuErrorCode error) +{ + return error <= ICU_ZERO_ERROR || error == ICU_BUFFER_OVERFLOW_ERROR; +} + +bool IcuConversionSucceeded(IcuErrorCode error) +{ + return error <= ICU_ZERO_ERROR; +} + +size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) { - dest[outputLength] = L'\0'; + WWASSERT(false); + return 0; } - return outputLength; -#else - UErrorCode error = U_ZERO_ERROR; - int32_t outputLength = 0; - u_strToUTF32(reinterpret_cast(dest), static_cast(destLen), &outputLength, - utf16.empty() ? nullptr : &utf16[0], static_cast(utf16.empty() ? 0 : utf16.size() - 1), &error); - if (U_FAILURE(error)) + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + g_icu.m_toUtf8WithSub(dest, static_cast(destLen), &outputLength, + reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); + if (!IcuConversionSucceeded(error)) + { + WWASSERT(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + WWASSERT(false); + return 0; + } + + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + g_icu.m_toUtf8WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); + if (!IcuPreflightSucceeded(error)) + { + WWASSERT(false); + return 0; + } + + return static_cast(outputLength); +} + +size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (!FitsInt(destLen) || !FitsInt(srcLen)) + { + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; + } + + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + g_icu.m_fromUtf8(reinterpret_cast(dest), static_cast(destLen), &outputLength, + src, static_cast(srcLen), &error); + if (!IcuConversionSucceeded(error)) { if (destLen > 0) { @@ -536,10 +558,105 @@ size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLe } return static_cast(outputLength); +} + +size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) +{ + if (!FitsInt(srcLen)) + { + return UTF8_INVALID; + } + + IcuErrorCode error = ICU_ZERO_ERROR; + int outputLength = 0; + g_icu.m_fromUtf8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + if (!IcuPreflightSucceeded(error)) + { + return UTF8_INVALID; + } + + return static_cast(outputLength); +} + +#endif + +bool IcuIsAvailable() +{ +#if defined(RTS_HAS_ICU_WINSDK) + return LoadSystemIcu(); +#elif defined(RTS_HAS_ICU) + return true; +#elif defined(RTS_ICU_DYNAMIC) + return g_icu.isAvailable(); +#else + return false; #endif } +} // namespace + +size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) +{ + if (IcuIsAvailable()) + { + return IcuWideToUtf8Len(src, srcLen); + } + +#ifdef _WIN32 + return WindowsWideToUtf8Len(src, srcLen); +#else + WWASSERT(false); + return 0; #endif +} + +size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) +{ + if (IcuIsAvailable()) + { + return IcuUtf8ToWideLen(src, srcLen); + } + +#ifdef _WIN32 + return WindowsUtf8ToWideLen(src, srcLen); +#else + return UTF8_INVALID; +#endif +} + +size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) +{ + if (IcuIsAvailable()) + { + return IcuWideToUtf8(dest, destLen, src, srcLen); + } + +#ifdef _WIN32 + return WindowsWideToUtf8(dest, destLen, src, srcLen); +#else + WWASSERT(false); + return 0; +#endif +} + +size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) +{ + if (IcuIsAvailable()) + { + return IcuUtf8ToWide(dest, destLen, src, srcLen); + } + +#ifdef _WIN32 + return WindowsUtf8ToWide(dest, destLen, src, srcLen); +#else + if (destLen > 0) + { + dest[0] = L'\0'; + } + + return UTF8_INVALID; +#endif +} // A UTF-8 continuation byte matches 10xxxxxx, so it can never start a sequence. static bool Utf8_Is_Continuation_Byte(char c) diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.h b/Core/Libraries/Source/WWVegas/WWLib/utf8.h index d8d332b5815..88216fa2e76 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.h +++ b/Core/Libraries/Source/WWVegas/WWLib/utf8.h @@ -21,9 +21,10 @@ #include #include -// UTF-8 <-> wide-character conversion backed by ICU4C. Windows dynamically uses the ICU4C -// implementation shipped with the operating system and falls back to the native text APIs when -// ICU is unavailable. Other platforms link ICU4C's common library. +// UTF-8 <-> wide-character conversion backed by ICU4C. +// Modern toolchains link the ICU C API (Windows SDK or vcpkg). VC6 LoadLibrary's OS icu.dll +// and falls back to Win32 CP_UTF8 when it is missing. Include WWLib/IcuSupport.h to use the +// rest of the linked ICU suite from engine code. // Returned when UTF-8 input is malformed. Zero is reserved for a successful empty conversion. const size_t UTF8_INVALID = (size_t)-1; diff --git a/cmake/icu.cmake b/cmake/icu.cmake new file mode 100644 index 00000000000..6c8e99ec752 --- /dev/null +++ b/cmake/icu.cmake @@ -0,0 +1,74 @@ +# ICU4C for the engine. +# +# Preference order: +# 1. find_package(ICU) from vcpkg or the system (C + C++ APIs: uc, i18n, data) +# 2. Windows SDK icu.lib for modern MSVC (C API: common + i18n via ) +# 3. Runtime LoadLibrary of OS icu.dll (VC6 and other Windows toolchains without an import lib) +# +# VC6 cannot consume modern ICU headers or import libraries. It keeps the dynamic loader. + +add_library(core_icu INTERFACE) + +set(RTS_ICU_LINKED FALSE) +set(RTS_ICU_CXX FALSE) +set(RTS_ICU_I18N FALSE) +set(RTS_ICU_WINSDK FALSE) +set(RTS_ICU_DYNAMIC FALSE) + +if(NOT IS_VS6_BUILD) + find_package(ICU QUIET COMPONENTS uc i18n data) + if(NOT ICU_FOUND) + find_package(ICU QUIET COMPONENTS uc i18n) + endif() + if(NOT ICU_FOUND) + find_package(ICU QUIET COMPONENTS uc data) + endif() + if(NOT ICU_FOUND) + find_package(ICU QUIET COMPONENTS uc) + endif() +endif() + +if(ICU_FOUND AND NOT IS_VS6_BUILD) + set(RTS_ICU_LINKED TRUE) + set(RTS_ICU_CXX TRUE) + target_link_libraries(core_icu INTERFACE ICU::uc) + if(TARGET ICU::i18n) + set(RTS_ICU_I18N TRUE) + target_link_libraries(core_icu INTERFACE ICU::i18n) + endif() + if(TARGET ICU::data) + target_link_libraries(core_icu INTERFACE ICU::data) + endif() + target_compile_definitions(core_icu INTERFACE RTS_HAS_ICU RTS_HAS_ICU_CXX) + if(RTS_ICU_I18N) + target_compile_definitions(core_icu INTERFACE RTS_HAS_ICU_I18N) + endif() + message(STATUS "ICU4C: linked via find_package (C++ API enabled)") +elseif(WIN32 AND NOT IS_VS6_BUILD AND NOT MINGW) + find_library(RTS_ICU_WINSDK_LIB NAMES icu) + if(RTS_ICU_WINSDK_LIB) + set(RTS_ICU_LINKED TRUE) + set(RTS_ICU_I18N TRUE) + set(RTS_ICU_WINSDK TRUE) + target_link_libraries(core_icu INTERFACE icu delayimp) + target_link_options(core_icu INTERFACE "/DELAYLOAD:icu.dll") + target_compile_definitions(core_icu INTERFACE RTS_HAS_ICU RTS_HAS_ICU_WINSDK RTS_HAS_ICU_I18N) + message(STATUS "ICU4C: linked via Windows SDK (${RTS_ICU_WINSDK_LIB})") + endif() +endif() + +if(NOT RTS_ICU_LINKED) + if(WIN32) + set(RTS_ICU_DYNAMIC TRUE) + target_compile_definitions(core_icu INTERFACE RTS_ICU_DYNAMIC) + message(STATUS "ICU4C: runtime load of system icu.dll") + else() + message(FATAL_ERROR "ICU4C is required on non-Windows platforms. Install libicu or enable vcpkg.") + endif() +endif() + +add_feature_info(IcuLinked RTS_ICU_LINKED "Link ICU4C into the engine") +add_feature_info(IcuCxx RTS_ICU_CXX "ICU C++ API available") +add_feature_info(IcuI18n RTS_ICU_I18N "ICU i18n API available") +add_feature_info(IcuWinSdk RTS_ICU_WINSDK "Windows SDK ICU C API") +add_feature_info(IcuDynamic RTS_ICU_DYNAMIC "Load OS icu.dll at runtime") diff --git a/vcpkg.json b/vcpkg.json index e07a95526c3..7b01ee2913f 100644 --- a/vcpkg.json +++ b/vcpkg.json @@ -3,10 +3,7 @@ "builtin-baseline": "9e593bb18ea69cc5095e012465dcd675a822ed0d", "dependencies": [ "zlib", - { - "name": "icu", - "platform": "!windows" - }, + "icu", "stb" ], "features": { From 85dde71db280c1b3c3f5d82d803a663834f9756c Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Fri, 21 Aug 2026 10:19:12 -0600 Subject: [PATCH 03/11] docs(icu): Keep prior UTF-8 comments alongside the ICU notes Preserve bobtista's original change comments and append the ICU conversion notes instead of replacing them. --- Core/GameEngine/Source/Common/System/AsciiString.cpp | 1 + Core/GameEngine/Source/Common/System/UnicodeString.cpp | 3 +++ .../Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp | 1 + 3 files changed, 5 insertions(+) diff --git a/Core/GameEngine/Source/Common/System/AsciiString.cpp b/Core/GameEngine/Source/Common/System/AsciiString.cpp index 31b51000a73..4859aceae92 100644 --- a/Core/GameEngine/Source/Common/System/AsciiString.cpp +++ b/Core/GameEngine/Source/Common/System/AsciiString.cpp @@ -305,6 +305,7 @@ char* AsciiString::getBufferForRead(Int len) void AsciiString::translate(const UnicodeString& stringSrc) { validate(); + // TheSuperHackers @fix bobtista 02/04/2026 Implement UTF-8 conversion replacing 7-bit ASCII only implementation // TheSuperHackers @bugfix CryoTheRenegade 04/08/2026 Convert wide text to UTF-8 with ICU4C. const WideChar* src = stringSrc.str(); const size_t srcLen = stringSrc.getLength(); diff --git a/Core/GameEngine/Source/Common/System/UnicodeString.cpp b/Core/GameEngine/Source/Common/System/UnicodeString.cpp index afe56a73980..ebaaa022443 100644 --- a/Core/GameEngine/Source/Common/System/UnicodeString.cpp +++ b/Core/GameEngine/Source/Common/System/UnicodeString.cpp @@ -219,6 +219,9 @@ WideChar* UnicodeString::getBufferForRead(Int len) void UnicodeString::translate(const AsciiString& stringSrc) { validate(); + // TheSuperHackers @fix bobtista 02/04/2026 Convert UTF-8 to wide, replacing the 7-bit ASCII only + // implementation. Data that is not valid UTF-8 (e.g. legacy CP1252) falls back to a 1:1 byte cast + // to preserve the original characters instead of producing replacement characters. // TheSuperHackers @bugfix CryoTheRenegade 04/08/2026 Convert UTF-8 to wide text with ICU4C. const char* src = stringSrc.str(); const size_t srcLen = stringSrc.getLength(); diff --git a/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp b/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp index 07920c2075f..9b7c16b7a65 100644 --- a/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp +++ b/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp @@ -32,6 +32,7 @@ //------------------------------------------------------------------------- +// TheSuperHackers @refactor bobtista 02/04/2026 Use WWLib UTF-8 functions instead of raw Win32 API calls // TheSuperHackers @refactor CryoTheRenegade 04/08/2026 Use the shared ICU4C UTF conversion functions. std::wstring MultiByteToWideCharSingleLine( const char* orig ) { From 92df643a68facbdae3787c01a4d7852ec39aafd1 Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Sun, 23 Aug 2026 16:14:27 -0600 Subject: [PATCH 04/11] fix(icu): Probe icu.dll once and link the found SDK import lib Publish the availability result with InterlockedCompareExchange, search the normal DLL path instead of System32 only, and pass the CMake-found icu.lib into the link line. --- Core/Libraries/Source/WWVegas/WWLib/utf8.cpp | 32 ++++++++++++-------- cmake/icu.cmake | 2 +- 2 files changed, 21 insertions(+), 13 deletions(-) diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp index 0030718f278..d4a2b5a073a 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp +++ b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp @@ -45,26 +45,34 @@ bool FitsInt(size_t length) #ifdef _WIN32 +#include "Utility/interlocked_adapter.h" + +enum +{ + IcuProbeUnknown = 0, + IcuProbeMissing = 1, + IcuProbeLoaded = 2 +}; + +// TheSuperHackers @fix CryoTheRenegade 23/08/2026 Probe icu.dll once through the normal DLL search order. bool LoadSystemIcu() { - static bool attempted = false; - static bool available = false; - if (attempted) + static volatile LONG cached = IcuProbeUnknown; + + const LONG existing = cached; + if (existing != IcuProbeUnknown) { - return available; + return existing == IcuProbeLoaded; } - attempted = true; - - char systemIcuPath[MAX_PATH]; - const UINT systemDirectoryLength = GetSystemDirectoryA(systemIcuPath, MAX_PATH); - if (systemDirectoryLength > 0 && systemDirectoryLength <= MAX_PATH - sizeof("\\icu.dll")) + const LONG result = LoadLibraryA("icu.dll") != nullptr ? IcuProbeLoaded : IcuProbeMissing; + const LONG previous = InterlockedCompareExchange(&cached, result, IcuProbeUnknown); + if (previous != IcuProbeUnknown) { - memcpy(systemIcuPath + systemDirectoryLength, "\\icu.dll", sizeof("\\icu.dll")); - available = LoadLibraryA(systemIcuPath) != nullptr; + return previous == IcuProbeLoaded; } - return available; + return result == IcuProbeLoaded; } size_t WindowsWideToUtf8Len(const wchar_t* src, size_t srcLen) diff --git a/cmake/icu.cmake b/cmake/icu.cmake index 6c8e99ec752..15555a0af8a 100644 --- a/cmake/icu.cmake +++ b/cmake/icu.cmake @@ -50,7 +50,7 @@ elseif(WIN32 AND NOT IS_VS6_BUILD AND NOT MINGW) set(RTS_ICU_LINKED TRUE) set(RTS_ICU_I18N TRUE) set(RTS_ICU_WINSDK TRUE) - target_link_libraries(core_icu INTERFACE icu delayimp) + target_link_libraries(core_icu INTERFACE ${RTS_ICU_WINSDK_LIB} delayimp) target_link_options(core_icu INTERFACE "/DELAYLOAD:icu.dll") target_compile_definitions(core_icu INTERFACE RTS_HAS_ICU RTS_HAS_ICU_WINSDK RTS_HAS_ICU_I18N) message(STATUS "ICU4C: linked via Windows SDK (${RTS_ICU_WINSDK_LIB})") From 36f7e78f1725725c1bce869818da59465a49ad40 Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Sun, 23 Aug 2026 16:29:39 -0600 Subject: [PATCH 05/11] fix(icu): Use InterlockedIncrement for the VC6 icu.dll probe Avoid InterlockedCompareExchange, whose VC6 and later SDK signatures disagree, so utf8.cpp compiles on both toolchains. --- Core/Libraries/Source/WWVegas/WWLib/utf8.cpp | 30 +++++++++++++------- 1 file changed, 20 insertions(+), 10 deletions(-) diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp index d4a2b5a073a..f64f1fe09ef 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp +++ b/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp @@ -45,8 +45,6 @@ bool FitsInt(size_t length) #ifdef _WIN32 -#include "Utility/interlocked_adapter.h" - enum { IcuProbeUnknown = 0, @@ -58,21 +56,33 @@ enum bool LoadSystemIcu() { static volatile LONG cached = IcuProbeUnknown; + static volatile LONG initGate = 0; - const LONG existing = cached; - if (existing != IcuProbeUnknown) + if (cached != IcuProbeUnknown) { - return existing == IcuProbeLoaded; + return cached == IcuProbeLoaded; } - const LONG result = LoadLibraryA("icu.dll") != nullptr ? IcuProbeLoaded : IcuProbeMissing; - const LONG previous = InterlockedCompareExchange(&cached, result, IcuProbeUnknown); - if (previous != IcuProbeUnknown) + // InterlockedIncrement is LONG* on every supported SDK. InterlockedCompareExchange is not: + // VC6 winbase.h takes PVOID*, while later SDKs take LONG*. +#if defined(_MSC_VER) && _MSC_VER < 1300 + const LONG gate = InterlockedIncrement(const_cast(&initGate)); +#else + const LONG gate = InterlockedIncrement(&initGate); +#endif + if (gate == 1) { - return previous == IcuProbeLoaded; + cached = LoadLibraryA("icu.dll") != nullptr ? IcuProbeLoaded : IcuProbeMissing; + } + else + { + while (cached == IcuProbeUnknown) + { + Sleep(0); + } } - return result == IcuProbeLoaded; + return cached == IcuProbeLoaded; } size_t WindowsWideToUtf8Len(const wchar_t* src, size_t srcLen) From 8e7569b8e363688d837bef776bd3fbb431f26e76 Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Thu, 17 Sep 2026 18:35:37 -0600 Subject: [PATCH 06/11] refactor(icu): Isolate ICU loading and own the DLL handle --- CMakeLists.txt | 1 + .../Source/Common/System/AsciiString.cpp | 14 +- .../Source/Common/System/UnicodeString.cpp | 10 +- .../GameSpy/Thread/ThreadUtils.cpp | 20 +-- .../Source/WWVegas/WWLib/CMakeLists.txt | 3 - Dependencies/ICU/CMakeLists.txt | 11 ++ Dependencies/ICU/ICU/IcuLoader.cpp | 74 +++++++++ Dependencies/ICU/ICU/IcuLoader.h | 64 ++++++++ .../ICU/ICU}/IcuSupport.h | 3 +- .../WWLib => Dependencies/ICU/ICU}/utf8.cpp | 154 ++++-------------- .../WWLib => Dependencies/ICU/ICU}/utf8.h | 7 +- Dependencies/ICU/README.md | 16 ++ cmake/icu.cmake | 25 +-- 13 files changed, 241 insertions(+), 161 deletions(-) create mode 100644 Dependencies/ICU/CMakeLists.txt create mode 100644 Dependencies/ICU/ICU/IcuLoader.cpp create mode 100644 Dependencies/ICU/ICU/IcuLoader.h rename {Core/Libraries/Source/WWVegas/WWLib => Dependencies/ICU/ICU}/IcuSupport.h (90%) rename {Core/Libraries/Source/WWVegas/WWLib => Dependencies/ICU/ICU}/utf8.cpp (79%) rename {Core/Libraries/Source/WWVegas/WWLib => Dependencies/ICU/ICU}/utf8.h (85%) create mode 100644 Dependencies/ICU/README.md diff --git a/CMakeLists.txt b/CMakeLists.txt index 9e9c4c5d09f..3764a2f5ba3 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -78,6 +78,7 @@ endif() add_subdirectory(Dependencies/Precompiled) add_subdirectory(Dependencies/Utility) +add_subdirectory(Dependencies/ICU) add_subdirectory(Dependencies/Bink) add_subdirectory(Dependencies/Miles) if (WIN32) diff --git a/Core/GameEngine/Source/Common/System/AsciiString.cpp b/Core/GameEngine/Source/Common/System/AsciiString.cpp index 4859aceae92..653b4a6396c 100644 --- a/Core/GameEngine/Source/Common/System/AsciiString.cpp +++ b/Core/GameEngine/Source/Common/System/AsciiString.cpp @@ -45,7 +45,7 @@ #include "PreRTS.h" // This must go first in EVERY cpp file in the GameEngine #include "Common/CriticalSection.h" -#include "WWLib/utf8.h" +#include "ICU/utf8.h" // ----------------------------------------------------- @@ -309,22 +309,22 @@ void AsciiString::translate(const UnicodeString& stringSrc) // TheSuperHackers @bugfix CryoTheRenegade 04/08/2026 Convert wide text to UTF-8 with ICU4C. const WideChar* src = stringSrc.str(); const size_t srcLen = stringSrc.getLength(); - const size_t len = Wide_To_Utf8_Len(src, srcLen); - if (len == 0) + const size_t dstLen = Wide_To_Utf8_Len(src, srcLen); + if (dstLen == 0) { clear(); } - else if (len >= static_cast(MAX_LEN)) + else if (dstLen >= static_cast(MAX_LEN)) { DEBUG_ASSERTCRASH(false, ("AsciiString::translate exceeds max string length %d with required UTF-8 length %u", - MAX_LEN, static_cast(len))); + MAX_LEN, static_cast(dstLen))); clear(); } else { - ensureUniqueBufferOfSize(static_cast(len) + 1, false, nullptr, nullptr); - Wide_To_Utf8(peek(), len + 1, src, srcLen); + ensureUniqueBufferOfSize(static_cast(dstLen) + 1, false, nullptr, nullptr); + Wide_To_Utf8(peek(), dstLen + 1, src, srcLen); } validate(); diff --git a/Core/GameEngine/Source/Common/System/UnicodeString.cpp b/Core/GameEngine/Source/Common/System/UnicodeString.cpp index ebaaa022443..859d6d48642 100644 --- a/Core/GameEngine/Source/Common/System/UnicodeString.cpp +++ b/Core/GameEngine/Source/Common/System/UnicodeString.cpp @@ -45,7 +45,7 @@ #include "PreRTS.h" // This must go first in EVERY cpp file in the GameEngine #include "Common/CriticalSection.h" -#include "WWLib/utf8.h" +#include "ICU/utf8.h" // ----------------------------------------------------- @@ -225,12 +225,12 @@ void UnicodeString::translate(const AsciiString& stringSrc) // TheSuperHackers @bugfix CryoTheRenegade 04/08/2026 Convert UTF-8 to wide text with ICU4C. const char* src = stringSrc.str(); const size_t srcLen = stringSrc.getLength(); - const size_t len = Utf8_To_Wide_Len(src, srcLen); + const size_t dstLen = Utf8_To_Wide_Len(src, srcLen); if (srcLen == 0) { clear(); } - else if (len == UTF8_INVALID) + else if (dstLen == UTF8_INVALID) { // Preserve legacy non-UTF-8 data with the original one-byte-to-one-wide-unit behavior. ensureUniqueBufferOfSize(static_cast(srcLen) + 1, false, nullptr, nullptr); @@ -244,8 +244,8 @@ void UnicodeString::translate(const AsciiString& stringSrc) } else { - ensureUniqueBufferOfSize(static_cast(len) + 1, false, nullptr, nullptr); - Utf8_To_Wide(peek(), len + 1, src, srcLen); + ensureUniqueBufferOfSize(static_cast(dstLen) + 1, false, nullptr, nullptr); + Utf8_To_Wide(peek(), dstLen + 1, src, srcLen); } validate(); diff --git a/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp b/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp index 9b7c16b7a65..9699a427d71 100644 --- a/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp +++ b/Core/GameEngine/Source/GameNetwork/GameSpy/Thread/ThreadUtils.cpp @@ -28,7 +28,7 @@ #include "PreRTS.h" // This must go first in EVERY cpp file in the GameEngine -#include "WWLib/utf8.h" +#include "ICU/utf8.h" //------------------------------------------------------------------------- @@ -37,14 +37,14 @@ std::wstring MultiByteToWideCharSingleLine( const char* orig ) { const size_t srcLen = strlen(orig); - const size_t len = Utf8_To_Wide_Len(orig, srcLen); - if (len == 0) + const size_t dstLen = Utf8_To_Wide_Len(orig, srcLen); + if (dstLen == 0) { return std::wstring(); } std::wstring ret; - if (len == UTF8_INVALID) + if (dstLen == UTF8_INVALID) { ret.resize(srcLen); for (size_t i = 0; i < srcLen; ++i) @@ -54,8 +54,8 @@ std::wstring MultiByteToWideCharSingleLine( const char* orig ) } else { - ret.resize(len); - Utf8_To_Wide(&ret[0], len, orig, srcLen); + ret.resize(dstLen); + Utf8_To_Wide(&ret[0], dstLen, orig, srcLen); } for (size_t i = 0; i < ret.size(); ++i) @@ -72,15 +72,15 @@ std::wstring MultiByteToWideCharSingleLine( const char* orig ) std::string WideCharStringToMultiByte( const WideChar* orig ) { const size_t srcLen = wcslen(orig); - const size_t len = Wide_To_Utf8_Len(orig, srcLen); - if (len == 0) + const size_t dstLen = Wide_To_Utf8_Len(orig, srcLen); + if (dstLen == 0) { return std::string(); } std::string ret; - ret.resize(len); - Wide_To_Utf8(&ret[0], len, orig, srcLen); + ret.resize(dstLen); + Wide_To_Utf8(&ret[0], dstLen, orig, srcLen); return ret; } diff --git a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt index f042507c773..cbca9c63baa 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt +++ b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt @@ -43,7 +43,6 @@ set(WWLIB_SRC #global.h hash.cpp hash.h - IcuSupport.h hashcalc.h HASHLIST.h #hashtab.h @@ -127,8 +126,6 @@ set(WWLIB_SRC trim.cpp trim.h uarray.h - utf8.cpp - utf8.h vector.cpp Vector.h visualc.h diff --git a/Dependencies/ICU/CMakeLists.txt b/Dependencies/ICU/CMakeLists.txt new file mode 100644 index 00000000000..245a5ba357a --- /dev/null +++ b/Dependencies/ICU/CMakeLists.txt @@ -0,0 +1,11 @@ +# ICU selection and public link settings are configured in cmake/icu.cmake. +target_sources(core_icu PRIVATE + ICU/IcuLoader.cpp + ICU/IcuLoader.h + ICU/IcuSupport.h + ICU/utf8.cpp + ICU/utf8.h +) + +target_include_directories(core_icu PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}) +target_link_libraries(core_icu PRIVATE core_config core_utility stlport) diff --git a/Dependencies/ICU/ICU/IcuLoader.cpp b/Dependencies/ICU/ICU/IcuLoader.cpp new file mode 100644 index 00000000000..5ea1578f8cb --- /dev/null +++ b/Dependencies/ICU/ICU/IcuLoader.cpp @@ -0,0 +1,74 @@ +/* +** Command & Conquer Generals Zero Hour(tm) +** Copyright 2026 TheSuperHackers +** +** This program is free software: you can redistribute it and/or modify +** it under the terms of the GNU General Public License as published by +** the Free Software Foundation, either version 3 of the License, or +** (at your option) any later version. +** +** This program is distributed in the hope that it will be useful, +** but WITHOUT ANY WARRANTY; without even the implied warranty of +** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +** GNU General Public License for more details. +** +** You should have received a copy of the GNU General Public License +** along with this program. If not, see . +*/ + +#include "ICU/IcuLoader.h" + +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + +#include + +const IcuLoader& IcuLoader::get() +{ + // VC6 does not synchronize function-local static construction. Serialize it + // on every toolchain, including publication of the resolved function pointers. + // LONG* works with both VC6 and current SDK InterlockedExchange signatures. + static LONG gate = 0; + while (InterlockedCompareExchange(&gate, 1, 0) != 0) + { + Sleep(0); + } + + static IcuLoader loader; + InterlockedExchange(&gate, 0); + return loader; +} + +IcuLoader::IcuLoader() : m_module(nullptr), m_fromUtf8(nullptr), m_toUtf8WithSub(nullptr) +{ + // Match the Windows SDK delay loader's DLL search order, including app-local ICU. + m_module = LoadLibraryA("icu.dll"); + if (m_module == nullptr) + { + return; + } + + m_fromUtf8 = reinterpret_cast(GetProcAddress(m_module, "u_strFromUTF8")); + m_toUtf8WithSub = reinterpret_cast(GetProcAddress(m_module, "u_strToUTF8WithSub")); + if (m_fromUtf8 == nullptr || m_toUtf8WithSub == nullptr) + { + FreeLibrary(m_module); + m_module = nullptr; + m_fromUtf8 = nullptr; + m_toUtf8WithSub = nullptr; + } +} + +IcuLoader::~IcuLoader() +{ + if (m_module != nullptr) + { + FreeLibrary(m_module); + } +} + +bool IcuLoader::isAvailable() const +{ + return m_module != nullptr; +} + +#endif diff --git a/Dependencies/ICU/ICU/IcuLoader.h b/Dependencies/ICU/ICU/IcuLoader.h new file mode 100644 index 00000000000..4da0e3c2dd9 --- /dev/null +++ b/Dependencies/ICU/ICU/IcuLoader.h @@ -0,0 +1,64 @@ +/* +** Command & Conquer Generals Zero Hour(tm) +** Copyright 2026 TheSuperHackers +** +** This program is free software: you can redistribute it and/or modify +** it under the terms of the GNU General Public License as published by +** the Free Software Foundation, either version 3 of the License, or +** (at your option) any later version. +** +** This program is distributed in the hope that it will be useful, +** but WITHOUT ANY WARRANTY; without even the implied warranty of +** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +** GNU General Public License for more details. +** +** You should have received a copy of the GNU General Public License +** along with this program. If not, see . +*/ + +#pragma once + +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + +#include + +// Owns one reference to icu.dll until normal process shutdown. The Windows SDK +// uses it to check availability before a delay-loaded call; VC6 calls the two +// resolved functions directly without ICU headers or an import library. +class IcuLoader +{ +public: + // These types match the Windows ICU C ABI, including its 32-bit error enum. + typedef unsigned short Char; + typedef int Char32; + typedef int ErrorCode; + typedef Char* (__cdecl* StrFromUtf8)(Char*, int, int*, const char*, int, ErrorCode*); + typedef char* (__cdecl* StrToUtf8WithSub)(char*, int, int*, const Char*, int, Char32, int*, ErrorCode*); + + // VC6 requires access from its generated static-destruction helper. + ~IcuLoader(); + + static const IcuLoader& get(); + bool isAvailable() const; + + StrFromUtf8 fromUtf8() const + { + return m_fromUtf8; + } + + StrToUtf8WithSub toUtf8WithSub() const + { + return m_toUtf8WithSub; + } + +private: + IcuLoader(); + IcuLoader(const IcuLoader&); + IcuLoader& operator=(const IcuLoader&); + + HMODULE m_module; + StrFromUtf8 m_fromUtf8; + StrToUtf8WithSub m_toUtf8WithSub; +}; + +#endif diff --git a/Core/Libraries/Source/WWVegas/WWLib/IcuSupport.h b/Dependencies/ICU/ICU/IcuSupport.h similarity index 90% rename from Core/Libraries/Source/WWVegas/WWLib/IcuSupport.h rename to Dependencies/ICU/ICU/IcuSupport.h index 6ce7ff720f2..24c3ce8fd91 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/IcuSupport.h +++ b/Dependencies/ICU/ICU/IcuSupport.h @@ -24,7 +24,8 @@ // RTS_HAS_ICU_CXX - ICU C++ API (icu::UnicodeString, icu::Locale, ...). // RTS_HAS_ICU_I18N - Collation, break iteration, converters, and related i18n APIs. // RTS_HAS_ICU_WINSDK - Windows SDK merged C API via (no C++ API). -// RTS_ICU_DYNAMIC - No import library; utf8.cpp LoadLibrary's OS icu.dll (VC6). +// RTS_ICU_DYNAMIC - VC6 uses IcuLoader and runtime UTF conversion exports from icu.dll. +// The full ICU headers/C++ API are not available in this mode. #if defined(RTS_HAS_ICU_WINSDK) diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp b/Dependencies/ICU/ICU/utf8.cpp similarity index 79% rename from Core/Libraries/Source/WWVegas/WWLib/utf8.cpp rename to Dependencies/ICU/ICU/utf8.cpp index f64f1fe09ef..3daa02fe54e 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.cpp +++ b/Dependencies/ICU/ICU/utf8.cpp @@ -16,23 +16,21 @@ ** along with this program. If not, see . */ -#include "always.h" -#include "utf8.h" +#include "ICU/utf8.h" +#include "ICU/IcuSupport.h" +#include "ICU/IcuLoader.h" + +#include #include #include -#if defined(RTS_HAS_ICU_WINSDK) -#include -#include -#elif defined(RTS_HAS_ICU) -#include #ifdef _WIN32 #include #endif + +#if defined(RTS_HAS_ICU) && !defined(RTS_HAS_ICU_WINSDK) #include -#elif defined(_WIN32) -#include #endif namespace @@ -45,58 +43,18 @@ bool FitsInt(size_t length) #ifdef _WIN32 -enum -{ - IcuProbeUnknown = 0, - IcuProbeMissing = 1, - IcuProbeLoaded = 2 -}; - -// TheSuperHackers @fix CryoTheRenegade 23/08/2026 Probe icu.dll once through the normal DLL search order. -bool LoadSystemIcu() -{ - static volatile LONG cached = IcuProbeUnknown; - static volatile LONG initGate = 0; - - if (cached != IcuProbeUnknown) - { - return cached == IcuProbeLoaded; - } - - // InterlockedIncrement is LONG* on every supported SDK. InterlockedCompareExchange is not: - // VC6 winbase.h takes PVOID*, while later SDKs take LONG*. -#if defined(_MSC_VER) && _MSC_VER < 1300 - const LONG gate = InterlockedIncrement(const_cast(&initGate)); -#else - const LONG gate = InterlockedIncrement(&initGate); -#endif - if (gate == 1) - { - cached = LoadLibraryA("icu.dll") != nullptr ? IcuProbeLoaded : IcuProbeMissing; - } - else - { - while (cached == IcuProbeUnknown) - { - Sleep(0); - } - } - - return cached == IcuProbeLoaded; -} - size_t WindowsWideToUtf8Len(const wchar_t* src, size_t srcLen) { if (!FitsInt(srcLen)) { - WWASSERT(false); + assert(false); return 0; } const int outputLength = WideCharToMultiByte(CP_UTF8, 0, src, static_cast(srcLen), nullptr, 0, nullptr, nullptr); if (outputLength == 0 && srcLen != 0) { - WWASSERT(false); + assert(false); return 0; } @@ -107,7 +65,7 @@ size_t WindowsWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t { if (!FitsInt(destLen) || !FitsInt(srcLen)) { - WWASSERT(false); + assert(false); return 0; } @@ -115,7 +73,7 @@ size_t WindowsWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t dest, static_cast(destLen), nullptr, nullptr); if (outputLength == 0 && srcLen != 0) { - WWASSERT(false); + assert(false); return 0; } @@ -196,7 +154,7 @@ size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcL { if (!FitsInt(destLen) || !FitsInt(srcLen)) { - WWASSERT(false); + assert(false); return 0; } @@ -206,7 +164,7 @@ size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcL reinterpret_cast(src), static_cast(srcLen), 0xFFFD, nullptr, &error); if (!IcuConversionSucceeded(error)) { - WWASSERT(false); + assert(false); return 0; } @@ -217,7 +175,7 @@ size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) { if (!FitsInt(srcLen)) { - WWASSERT(false); + assert(false); return 0; } @@ -227,7 +185,7 @@ size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) static_cast(srcLen), 0xFFFD, nullptr, &error); if (!IcuPreflightSucceeded(error)) { - WWASSERT(false); + assert(false); return 0; } @@ -333,7 +291,7 @@ size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) std::vector utf16; if (!WideToUtf16(utf16, src, srcLen)) { - WWASSERT(false); + assert(false); return 0; } @@ -343,7 +301,7 @@ size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) static_cast(utf16.empty() ? 0 : utf16.size() - 1), 0xFFFD, nullptr, &error); if (!IcuPreflightSucceeded(error)) { - WWASSERT(false); + assert(false); return 0; } @@ -374,14 +332,14 @@ size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcL { if (!FitsInt(destLen)) { - WWASSERT(false); + assert(false); return 0; } std::vector utf16; if (!WideToUtf16(utf16, src, srcLen)) { - WWASSERT(false); + assert(false); return 0; } @@ -392,7 +350,7 @@ size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcL 0xFFFD, nullptr, &error); if (U_FAILURE(error)) { - WWASSERT(false); + assert(false); return 0; } @@ -443,13 +401,8 @@ size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcL #elif defined(RTS_ICU_DYNAMIC) -typedef unsigned short IcuChar; -typedef int IcuChar32; -typedef int IcuErrorCode; -typedef IcuChar* (__cdecl* IcuStrFromUtf8)( - IcuChar*, int, int*, const char*, int, IcuErrorCode*); -typedef char* (__cdecl* IcuStrToUtf8WithSub)( - char*, int, int*, const IcuChar*, int, IcuChar32, int*, IcuErrorCode*); +typedef IcuLoader::Char IcuChar; +typedef IcuLoader::ErrorCode IcuErrorCode; enum { @@ -458,45 +411,6 @@ enum ICU_REPLACEMENT_CHARACTER = 0xFFFD }; -class WindowsIcuFunctions -{ -public: - WindowsIcuFunctions() : m_fromUtf8(nullptr), m_toUtf8WithSub(nullptr), m_module(nullptr) - { - if (!LoadSystemIcu()) - { - return; - } - - m_module = GetModuleHandleA("icu.dll"); - if (m_module != nullptr) - { - m_fromUtf8 = reinterpret_cast(GetProcAddress(m_module, "u_strFromUTF8")); - m_toUtf8WithSub = reinterpret_cast(GetProcAddress(m_module, "u_strToUTF8WithSub")); - } - - if (m_fromUtf8 == nullptr || m_toUtf8WithSub == nullptr) - { - m_module = nullptr; - m_fromUtf8 = nullptr; - m_toUtf8WithSub = nullptr; - } - } - - bool isAvailable() const - { - return m_fromUtf8 != nullptr && m_toUtf8WithSub != nullptr; - } - - IcuStrFromUtf8 m_fromUtf8; - IcuStrToUtf8WithSub m_toUtf8WithSub; - -private: - HMODULE m_module; -}; - -WindowsIcuFunctions g_icu; - bool IcuPreflightSucceeded(IcuErrorCode error) { return error <= ICU_ZERO_ERROR || error == ICU_BUFFER_OVERFLOW_ERROR; @@ -511,17 +425,17 @@ size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcL { if (!FitsInt(destLen) || !FitsInt(srcLen)) { - WWASSERT(false); + assert(false); return 0; } IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - g_icu.m_toUtf8WithSub(dest, static_cast(destLen), &outputLength, + IcuLoader::get().toUtf8WithSub()(dest, static_cast(destLen), &outputLength, reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); if (!IcuConversionSucceeded(error)) { - WWASSERT(false); + assert(false); return 0; } @@ -532,17 +446,17 @@ size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) { if (!FitsInt(srcLen)) { - WWASSERT(false); + assert(false); return 0; } IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - g_icu.m_toUtf8WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + IcuLoader::get().toUtf8WithSub()(nullptr, 0, &outputLength, reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); if (!IcuPreflightSucceeded(error)) { - WWASSERT(false); + assert(false); return 0; } @@ -563,7 +477,7 @@ size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcL IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - g_icu.m_fromUtf8(reinterpret_cast(dest), static_cast(destLen), &outputLength, + IcuLoader::get().fromUtf8()(reinterpret_cast(dest), static_cast(destLen), &outputLength, src, static_cast(srcLen), &error); if (!IcuConversionSucceeded(error)) { @@ -587,7 +501,7 @@ size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - g_icu.m_fromUtf8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + IcuLoader::get().fromUtf8()(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); if (!IcuPreflightSucceeded(error)) { return UTF8_INVALID; @@ -601,11 +515,11 @@ size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) bool IcuIsAvailable() { #if defined(RTS_HAS_ICU_WINSDK) - return LoadSystemIcu(); + return IcuLoader::get().isAvailable(); #elif defined(RTS_HAS_ICU) return true; #elif defined(RTS_ICU_DYNAMIC) - return g_icu.isAvailable(); + return IcuLoader::get().isAvailable(); #else return false; #endif @@ -623,7 +537,7 @@ size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) #ifdef _WIN32 return WindowsWideToUtf8Len(src, srcLen); #else - WWASSERT(false); + assert(false); return 0; #endif } @@ -652,7 +566,7 @@ size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLe #ifdef _WIN32 return WindowsWideToUtf8(dest, destLen, src, srcLen); #else - WWASSERT(false); + assert(false); return 0; #endif } diff --git a/Core/Libraries/Source/WWVegas/WWLib/utf8.h b/Dependencies/ICU/ICU/utf8.h similarity index 85% rename from Core/Libraries/Source/WWVegas/WWLib/utf8.h rename to Dependencies/ICU/ICU/utf8.h index 88216fa2e76..d6f9f3675a9 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/utf8.h +++ b/Dependencies/ICU/ICU/utf8.h @@ -22,9 +22,10 @@ #include // UTF-8 <-> wide-character conversion backed by ICU4C. -// Modern toolchains link the ICU C API (Windows SDK or vcpkg). VC6 LoadLibrary's OS icu.dll -// and falls back to Win32 CP_UTF8 when it is missing. Include WWLib/IcuSupport.h to use the -// rest of the linked ICU suite from engine code. +// Modern toolchains link the ICU C API through the Windows SDK or a full ICU package. +// VC6 uses IcuLoader to load icu.dll and call its UTF conversion exports at runtime. +// Windows builds fall back to Win32 CP_UTF8 if the DLL or required exports are missing. +// Include ICU/IcuSupport.h to use the rest of the linked ICU suite from engine code. // Returned when UTF-8 input is malformed. Zero is reserved for a successful empty conversion. const size_t UTF8_INVALID = (size_t)-1; diff --git a/Dependencies/ICU/README.md b/Dependencies/ICU/README.md new file mode 100644 index 00000000000..79f0ee6d8fa --- /dev/null +++ b/Dependencies/ICU/README.md @@ -0,0 +1,16 @@ +# ICU integration + +`core_icu` provides UTF conversions and the ICU headers available to the selected toolchain. +Include `ICU/utf8.h` for conversions and `ICU/IcuSupport.h` for linked ICU APIs. + +| Build configuration | ICU access | Fallback | +| --- | --- | --- | +| Full ICU package, including Windows | Linked ICU C APIs and available C++ APIs | None needed | +| Windows SDK ICU | Linked, delay-loaded C APIs; `IcuLoader` checks the conversion exports first | Win32 `CP_UTF8` when the DLL or exports are unavailable | +| VC6 and Windows builds without an import library | `IcuLoader` loads `icu.dll` and resolves `u_strFromUTF8` and `u_strToUTF8WithSub` | Win32 `CP_UTF8` when the DLL or exports are unavailable | + +VC6 does use ICU when these exports are available. It does not compile against modern ICU headers or expose the full ICU C++ API. + +`IcuLoader` initializes once, with synchronization that also works on VC6. It owns the DLL reference returned by `LoadLibraryA`, releases it immediately if an export is missing, and otherwise releases it at normal process shutdown. The SDK delay loader owns any additional reference it acquires independently. Conversion workers must finish before static destruction begins. + +The loader uses the normal Windows DLL search order, matching SDK delay loading and allowing an app-local `icu.dll`. There is no separate probe that discards a DLL handle. diff --git a/cmake/icu.cmake b/cmake/icu.cmake index 15555a0af8a..fa114191dd4 100644 --- a/cmake/icu.cmake +++ b/cmake/icu.cmake @@ -5,9 +5,10 @@ # 2. Windows SDK icu.lib for modern MSVC (C API: common + i18n via ) # 3. Runtime LoadLibrary of OS icu.dll (VC6 and other Windows toolchains without an import lib) # -# VC6 cannot consume modern ICU headers or import libraries. It keeps the dynamic loader. +# VC6 uses ICU UTF conversion exports through IcuLoader, without ICU headers or import libraries. +# If the DLL or either export is missing, conversion falls back to Win32 CP_UTF8. -add_library(core_icu INTERFACE) +add_library(core_icu STATIC) set(RTS_ICU_LINKED FALSE) set(RTS_ICU_CXX FALSE) @@ -31,17 +32,17 @@ endif() if(ICU_FOUND AND NOT IS_VS6_BUILD) set(RTS_ICU_LINKED TRUE) set(RTS_ICU_CXX TRUE) - target_link_libraries(core_icu INTERFACE ICU::uc) + target_link_libraries(core_icu PUBLIC ICU::uc) if(TARGET ICU::i18n) set(RTS_ICU_I18N TRUE) - target_link_libraries(core_icu INTERFACE ICU::i18n) + target_link_libraries(core_icu PUBLIC ICU::i18n) endif() if(TARGET ICU::data) - target_link_libraries(core_icu INTERFACE ICU::data) + target_link_libraries(core_icu PUBLIC ICU::data) endif() - target_compile_definitions(core_icu INTERFACE RTS_HAS_ICU RTS_HAS_ICU_CXX) + target_compile_definitions(core_icu PUBLIC RTS_HAS_ICU RTS_HAS_ICU_CXX) if(RTS_ICU_I18N) - target_compile_definitions(core_icu INTERFACE RTS_HAS_ICU_I18N) + target_compile_definitions(core_icu PUBLIC RTS_HAS_ICU_I18N) endif() message(STATUS "ICU4C: linked via find_package (C++ API enabled)") elseif(WIN32 AND NOT IS_VS6_BUILD AND NOT MINGW) @@ -50,9 +51,9 @@ elseif(WIN32 AND NOT IS_VS6_BUILD AND NOT MINGW) set(RTS_ICU_LINKED TRUE) set(RTS_ICU_I18N TRUE) set(RTS_ICU_WINSDK TRUE) - target_link_libraries(core_icu INTERFACE ${RTS_ICU_WINSDK_LIB} delayimp) - target_link_options(core_icu INTERFACE "/DELAYLOAD:icu.dll") - target_compile_definitions(core_icu INTERFACE RTS_HAS_ICU RTS_HAS_ICU_WINSDK RTS_HAS_ICU_I18N) + target_link_libraries(core_icu PUBLIC ${RTS_ICU_WINSDK_LIB} delayimp) + target_link_options(core_icu PUBLIC "/DELAYLOAD:icu.dll") + target_compile_definitions(core_icu PUBLIC RTS_HAS_ICU RTS_HAS_ICU_WINSDK RTS_HAS_ICU_I18N) message(STATUS "ICU4C: linked via Windows SDK (${RTS_ICU_WINSDK_LIB})") endif() endif() @@ -60,8 +61,8 @@ endif() if(NOT RTS_ICU_LINKED) if(WIN32) set(RTS_ICU_DYNAMIC TRUE) - target_compile_definitions(core_icu INTERFACE RTS_ICU_DYNAMIC) - message(STATUS "ICU4C: runtime load of system icu.dll") + target_compile_definitions(core_icu PUBLIC RTS_ICU_DYNAMIC) + message(STATUS "ICU4C: IcuLoader resolves UTF conversion exports from icu.dll; Win32 fallback if unavailable") else() message(FATAL_ERROR "ICU4C is required on non-Windows platforms. Install libicu or enable vcpkg.") endif() From b6e0a07f2091671120a517d14fe8a7ea769f0e15 Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Sat, 19 Sep 2026 10:06:56 -0600 Subject: [PATCH 07/11] refactor(icu): Add explicit loader lifetime and restrict DLL lookup --- Dependencies/ICU/ICU/IcuLoader.cpp | 158 +++++++++++++++--- Dependencies/ICU/ICU/IcuLoader.h | 50 +++--- Dependencies/ICU/ICU/utf8.cpp | 37 ++-- Dependencies/ICU/ICU/utf8.h | 4 +- Dependencies/ICU/README.md | 10 +- .../GameEngine/Source/Common/GameMain.cpp | 3 + .../GameEngine/Source/Common/GameMain.cpp | 3 + 7 files changed, 189 insertions(+), 76 deletions(-) diff --git a/Dependencies/ICU/ICU/IcuLoader.cpp b/Dependencies/ICU/ICU/IcuLoader.cpp index 5ea1578f8cb..b117c162ff6 100644 --- a/Dependencies/ICU/ICU/IcuLoader.cpp +++ b/Dependencies/ICU/ICU/IcuLoader.cpp @@ -21,54 +21,158 @@ #if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) #include +#include +#include -const IcuLoader& IcuLoader::get() +namespace { - // VC6 does not synchronize function-local static construction. Serialize it - // on every toolchain, including publication of the resolved function pointers. - // LONG* works with both VC6 and current SDK InterlockedExchange signatures. - static LONG gate = 0; - while (InterlockedCompareExchange(&gate, 1, 0) != 0) + +LONG Gate = 0; +unsigned int ReferenceCount = 0; +HMODULE Module = nullptr; +IcuLoader::StrFromUtf8 FromUtf8 = nullptr; +IcuLoader::StrToUtf8WithSub ToUtf8WithSub = nullptr; + +class LoaderLock +{ +public: + LoaderLock() + { + // LONG* works with both VC6 and current SDK interlocked signatures. + while (InterlockedCompareExchange(&Gate, 1, 0) != 0) + { + Sleep(0); + } + } + + ~LoaderLock() { - Sleep(0); + InterlockedExchange(&Gate, 0); } - static IcuLoader loader; - InterlockedExchange(&gate, 0); - return loader; +private: + LoaderLock(const LoaderLock&); + LoaderLock& operator=(const LoaderLock&); +}; + +HMODULE loadModule() +{ + // Use absolute paths so -setCwd and PATH cannot supply an unexpected DLL. + // Wide paths also allow app-local ICU in non-ASCII installation directories. + wchar_t path[MAX_PATH]; + const wchar_t name[] = L"icu.dll"; + const DWORD length = GetModuleFileNameW(nullptr, path, MAX_PATH); + if (length > 0 && length < MAX_PATH) + { + wchar_t* separator = wcsrchr(path, L'\\'); + if (separator != nullptr && separator + 1 - path + sizeof(name) / sizeof(name[0]) <= MAX_PATH) + { + memcpy(separator + 1, name, sizeof(name)); + HMODULE module = LoadLibraryW(path); + if (module != nullptr) + { + return module; + } + } + } + + const UINT systemLength = GetSystemDirectoryW(path, MAX_PATH); + if (systemLength > 0 && systemLength + 1 + sizeof(name) / sizeof(name[0]) <= MAX_PATH) + { + path[systemLength] = L'\\'; + memcpy(path + systemLength + 1, name, sizeof(name)); + return LoadLibraryW(path); + } + + return nullptr; } -IcuLoader::IcuLoader() : m_module(nullptr), m_fromUtf8(nullptr), m_toUtf8WithSub(nullptr) +void freeResources() { - // Match the Windows SDK delay loader's DLL search order, including app-local ICU. - m_module = LoadLibraryA("icu.dll"); - if (m_module == nullptr) + if (Module != nullptr) { - return; + FreeLibrary(Module); + Module = nullptr; + } + + FromUtf8 = nullptr; + ToUtf8WithSub = nullptr; +} + +} // namespace + +bool IcuLoader::load() +{ + LoaderLock lock; + // Failed attempts also retain a reference so overlapping callers share the + // result. Once all callers unload, a later load can retry. + if (++ReferenceCount > 1) + { + return isLoaded(); + } + + Module = loadModule(); + if (Module == nullptr) + { + return false; } - m_fromUtf8 = reinterpret_cast(GetProcAddress(m_module, "u_strFromUTF8")); - m_toUtf8WithSub = reinterpret_cast(GetProcAddress(m_module, "u_strToUTF8WithSub")); - if (m_fromUtf8 == nullptr || m_toUtf8WithSub == nullptr) + FromUtf8 = reinterpret_cast(GetProcAddress(Module, "u_strFromUTF8")); + ToUtf8WithSub = reinterpret_cast(GetProcAddress(Module, "u_strToUTF8WithSub")); + if (FromUtf8 == nullptr || ToUtf8WithSub == nullptr) { - FreeLibrary(m_module); - m_module = nullptr; - m_fromUtf8 = nullptr; - m_toUtf8WithSub = nullptr; + freeResources(); + return false; } + + return true; } -IcuLoader::~IcuLoader() +void IcuLoader::unload() { - if (m_module != nullptr) + LoaderLock lock; + if (ReferenceCount == 0) + { + return; + } + + if (--ReferenceCount == 0) { - FreeLibrary(m_module); + freeResources(); } } -bool IcuLoader::isAvailable() const +bool IcuLoader::isLoaded() +{ + return Module != nullptr; +} + +IcuLoader::StrFromUtf8 IcuLoader::fromUtf8() { - return m_module != nullptr; + return FromUtf8; } +IcuLoader::StrToUtf8WithSub IcuLoader::toUtf8WithSub() +{ + return ToUtf8WithSub; +} + +#endif + +IcuScope::IcuScope() +{ +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + m_available = IcuLoader::load(); +#elif defined(RTS_HAS_ICU) + m_available = true; +#else + m_available = false; #endif +} + +IcuScope::~IcuScope() +{ +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + IcuLoader::unload(); +#endif +} diff --git a/Dependencies/ICU/ICU/IcuLoader.h b/Dependencies/ICU/ICU/IcuLoader.h index 4da0e3c2dd9..5af065a025e 100644 --- a/Dependencies/ICU/ICU/IcuLoader.h +++ b/Dependencies/ICU/ICU/IcuLoader.h @@ -22,9 +22,10 @@ #include -// Owns one reference to icu.dll until normal process shutdown. The Windows SDK -// uses it to check availability before a delay-loaded call; VC6 calls the two -// resolved functions directly without ICU headers or an import library. +// Loads and unloads icu.dll with a shared reference count, like BinkLoader and +// MilesLoader. Every load needs a paired unload, even when loading fails. +// Load/unload are synchronized for conversion workers, including on VC6. +// Hold a reference while checking availability or using the resolved functions. class IcuLoader { public: @@ -35,30 +36,37 @@ class IcuLoader typedef Char* (__cdecl* StrFromUtf8)(Char*, int, int*, const char*, int, ErrorCode*); typedef char* (__cdecl* StrToUtf8WithSub)(char*, int, int*, const Char*, int, Char32, int*, ErrorCode*); - // VC6 requires access from its generated static-destruction helper. - ~IcuLoader(); + static bool load(); + static void unload(); + static bool isLoaded(); + static StrFromUtf8 fromUtf8(); + static StrToUtf8WithSub toUtf8WithSub(); - static const IcuLoader& get(); - bool isAvailable() const; +private: + IcuLoader(); + IcuLoader(const IcuLoader&); + IcuLoader& operator=(const IcuLoader&); +}; - StrFromUtf8 fromUtf8() const - { - return m_fromUtf8; - } +#endif - StrToUtf8WithSub toUtf8WithSub() const +// Keeps ICU available throughout a conversion or a longer application lifetime. +// Linked ICU needs no explicit DLL ownership; unavailable Windows ICU uses the +// Win32 conversion fallback. Scopes may overlap on different threads. +class IcuScope +{ +public: + IcuScope(); + ~IcuScope(); + + bool isAvailable() const { - return m_toUtf8WithSub; + return m_available; } private: - IcuLoader(); - IcuLoader(const IcuLoader&); - IcuLoader& operator=(const IcuLoader&); + IcuScope(const IcuScope&); + IcuScope& operator=(const IcuScope&); - HMODULE m_module; - StrFromUtf8 m_fromUtf8; - StrToUtf8WithSub m_toUtf8WithSub; + bool m_available; }; - -#endif diff --git a/Dependencies/ICU/ICU/utf8.cpp b/Dependencies/ICU/ICU/utf8.cpp index 3daa02fe54e..c612c9526ff 100644 --- a/Dependencies/ICU/ICU/utf8.cpp +++ b/Dependencies/ICU/ICU/utf8.cpp @@ -136,7 +136,7 @@ size_t WindowsUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t #endif -#if defined(RTS_HAS_ICU) +#if defined(RTS_HAS_ICU) && !defined(RTS_HAS_ICU_WINSDK) bool IcuPreflightSucceeded(UErrorCode error) { @@ -399,7 +399,7 @@ size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcL #endif -#elif defined(RTS_ICU_DYNAMIC) +#elif defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) typedef IcuLoader::Char IcuChar; typedef IcuLoader::ErrorCode IcuErrorCode; @@ -431,7 +431,7 @@ size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcL IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - IcuLoader::get().toUtf8WithSub()(dest, static_cast(destLen), &outputLength, + IcuLoader::toUtf8WithSub()(dest, static_cast(destLen), &outputLength, reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); if (!IcuConversionSucceeded(error)) { @@ -452,7 +452,7 @@ size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - IcuLoader::get().toUtf8WithSub()(nullptr, 0, &outputLength, reinterpret_cast(src), + IcuLoader::toUtf8WithSub()(nullptr, 0, &outputLength, reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); if (!IcuPreflightSucceeded(error)) { @@ -477,7 +477,7 @@ size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcL IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - IcuLoader::get().fromUtf8()(reinterpret_cast(dest), static_cast(destLen), &outputLength, + IcuLoader::fromUtf8()(reinterpret_cast(dest), static_cast(destLen), &outputLength, src, static_cast(srcLen), &error); if (!IcuConversionSucceeded(error)) { @@ -501,7 +501,7 @@ size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - IcuLoader::get().fromUtf8()(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + IcuLoader::fromUtf8()(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); if (!IcuPreflightSucceeded(error)) { return UTF8_INVALID; @@ -512,24 +512,12 @@ size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) #endif -bool IcuIsAvailable() -{ -#if defined(RTS_HAS_ICU_WINSDK) - return IcuLoader::get().isAvailable(); -#elif defined(RTS_HAS_ICU) - return true; -#elif defined(RTS_ICU_DYNAMIC) - return IcuLoader::get().isAvailable(); -#else - return false; -#endif -} - } // namespace size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) { - if (IcuIsAvailable()) + IcuScope icu; + if (icu.isAvailable()) { return IcuWideToUtf8Len(src, srcLen); } @@ -544,7 +532,8 @@ size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) { - if (IcuIsAvailable()) + IcuScope icu; + if (icu.isAvailable()) { return IcuUtf8ToWideLen(src, srcLen); } @@ -558,7 +547,8 @@ size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) { - if (IcuIsAvailable()) + IcuScope icu; + if (icu.isAvailable()) { return IcuWideToUtf8(dest, destLen, src, srcLen); } @@ -573,7 +563,8 @@ size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLe size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) { - if (IcuIsAvailable()) + IcuScope icu; + if (icu.isAvailable()) { return IcuUtf8ToWide(dest, destLen, src, srcLen); } diff --git a/Dependencies/ICU/ICU/utf8.h b/Dependencies/ICU/ICU/utf8.h index d6f9f3675a9..1e3ed8e84ee 100644 --- a/Dependencies/ICU/ICU/utf8.h +++ b/Dependencies/ICU/ICU/utf8.h @@ -22,8 +22,8 @@ #include // UTF-8 <-> wide-character conversion backed by ICU4C. -// Modern toolchains link the ICU C API through the Windows SDK or a full ICU package. -// VC6 uses IcuLoader to load icu.dll and call its UTF conversion exports at runtime. +// Full ICU packages use linked C APIs. Windows SDK and VC6 builds use IcuLoader +// to load icu.dll and retain its UTF conversion exports for each call. // Windows builds fall back to Win32 CP_UTF8 if the DLL or required exports are missing. // Include ICU/IcuSupport.h to use the rest of the linked ICU suite from engine code. diff --git a/Dependencies/ICU/README.md b/Dependencies/ICU/README.md index 79f0ee6d8fa..e6818578f8b 100644 --- a/Dependencies/ICU/README.md +++ b/Dependencies/ICU/README.md @@ -6,11 +6,15 @@ Include `ICU/utf8.h` for conversions and `ICU/IcuSupport.h` for linked ICU APIs. | Build configuration | ICU access | Fallback | | --- | --- | --- | | Full ICU package, including Windows | Linked ICU C APIs and available C++ APIs | None needed | -| Windows SDK ICU | Linked, delay-loaded C APIs; `IcuLoader` checks the conversion exports first | Win32 `CP_UTF8` when the DLL or exports are unavailable | +| Windows SDK ICU | `IcuLoader` resolves the conversion exports; other C APIs remain available through delay imports | Win32 `CP_UTF8` when the DLL or exports are unavailable | | VC6 and Windows builds without an import library | `IcuLoader` loads `icu.dll` and resolves `u_strFromUTF8` and `u_strToUTF8WithSub` | Win32 `CP_UTF8` when the DLL or exports are unavailable | VC6 does use ICU when these exports are available. It does not compile against modern ICU headers or expose the full ICU C++ API. -`IcuLoader` initializes once, with synchronization that also works on VC6. It owns the DLL reference returned by `LoadLibraryA`, releases it immediately if an export is missing, and otherwise releases it at normal process shutdown. The SDK delay loader owns any additional reference it acquires independently. Conversion workers must finish before static destruction begins. +`IcuLoader` has paired `load()` and `unload()` calls, following `BinkLoader` and `MilesLoader`. Every load needs an unload, including failed loads. Overlapping callers share the result, and the last unload releases the DLL and clears the function pointers. A later load can retry. Reference changes are synchronized, including on VC6; callers must retain a reference while using the resolved functions. -The loader uses the normal Windows DLL search order, matching SDK delay loading and allowing an app-local `icu.dll`. There is no separate probe that discards a DLL handle. +`IcuScope` pairs these calls automatically. Each conversion holds a scope so another thread cannot unload its functions while they are running. Both games retain an additional scope through engine teardown to avoid repeated loading. Tools can retain a scope around batches of conversions too. Full linked ICU does not need explicit ownership. + +Conversions in Windows SDK builds use the resolved function pointers, avoiding an extra delay-loader reference. If a caller uses other SDK ICU APIs directly, the SDK delay loader owns its reference independently. + +The loader tries absolute paths in the executable directory and then the Windows system directory. It supports app-local ICU and Unicode installation paths without searching the working directory or `PATH`. A missing DLL or conversion export selects the Win32 fallback; a DLL with missing exports is released immediately. diff --git a/Generals/Code/GameEngine/Source/Common/GameMain.cpp b/Generals/Code/GameEngine/Source/Common/GameMain.cpp index c9dd0a94427..a8da5ab64c8 100644 --- a/Generals/Code/GameEngine/Source/Common/GameMain.cpp +++ b/Generals/Code/GameEngine/Source/Common/GameMain.cpp @@ -31,6 +31,7 @@ #include "Common/FramePacer.h" #include "Common/GameEngine.h" #include "Common/ReplaySimulation.h" +#include "ICU/IcuLoader.h" /** @@ -38,6 +39,8 @@ */ Int GameMain() { + // Retain ICU through engine teardown; worker conversions hold their own references. + IcuScope icu; int exitcode = 0; // initialize the game engine using factory function TheFramePacer = new FramePacer(); diff --git a/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp b/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp index ed94ec7bf54..606c7e0ff5f 100644 --- a/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp +++ b/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp @@ -31,6 +31,7 @@ #include "Common/FramePacer.h" #include "Common/GameEngine.h" #include "Common/ReplaySimulation.h" +#include "ICU/IcuLoader.h" /** @@ -38,6 +39,8 @@ */ Int GameMain() { + // Retain ICU through engine teardown; worker conversions hold their own references. + IcuScope icu; int exitcode = 0; // initialize the game engine using factory function TheFramePacer = new FramePacer(); From c3428aefbe983851e570b7834f4a32833f45f85d Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Sun, 20 Sep 2026 09:20:55 -0600 Subject: [PATCH 08/11] refactor(icu): Cache lazy loading and synchronize conversion calls --- Dependencies/ICU/ICU/IcuLoader.cpp | 119 ++++++++++-------- Dependencies/ICU/ICU/IcuLoader.h | 42 ++----- Dependencies/ICU/ICU/utf8.cpp | 85 ++++--------- Dependencies/ICU/ICU/utf8.h | 2 +- Dependencies/ICU/README.md | 4 +- .../GameEngine/Source/Common/GameMain.cpp | 7 +- .../GameEngine/Source/Common/GameMain.cpp | 7 +- 7 files changed, 118 insertions(+), 148 deletions(-) diff --git a/Dependencies/ICU/ICU/IcuLoader.cpp b/Dependencies/ICU/ICU/IcuLoader.cpp index b117c162ff6..7aa084b1e51 100644 --- a/Dependencies/ICU/ICU/IcuLoader.cpp +++ b/Dependencies/ICU/ICU/IcuLoader.cpp @@ -20,34 +20,59 @@ #if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) -#include #include #include namespace { -LONG Gate = 0; -unsigned int ReferenceCount = 0; +typedef IcuLoader::Char* (__cdecl* StrFromUtf8)(IcuLoader::Char*, int, int*, const char*, int, IcuLoader::ErrorCode*); +typedef char* (__cdecl* StrToUtf8WithSub)(char*, int, int*, const IcuLoader::Char*, int, + IcuLoader::Char32, int*, IcuLoader::ErrorCode*); + +bool LoadAttempted = false; HMODULE Module = nullptr; -IcuLoader::StrFromUtf8 FromUtf8 = nullptr; -IcuLoader::StrToUtf8WithSub ToUtf8WithSub = nullptr; +StrFromUtf8 FromUtf8 = nullptr; +StrToUtf8WithSub ToUtf8WithSub = nullptr; + +void freeResources(); + +class LoaderCriticalSection +{ +public: + LoaderCriticalSection() + { + InitializeCriticalSection(§ion); + } + + ~LoaderCriticalSection() + { + // Also release ICU for tools that do not explicitly unload at shutdown. + // Conversion workers must have stopped before static destruction. + freeResources(); + DeleteCriticalSection(§ion); + } + + CRITICAL_SECTION section; + +private: + LoaderCriticalSection(const LoaderCriticalSection&); + LoaderCriticalSection& operator=(const LoaderCriticalSection&); +}; + +LoaderCriticalSection CriticalSection; class LoaderLock { public: LoaderLock() { - // LONG* works with both VC6 and current SDK interlocked signatures. - while (InterlockedCompareExchange(&Gate, 1, 0) != 0) - { - Sleep(0); - } + EnterCriticalSection(&CriticalSection.section); } ~LoaderLock() { - InterlockedExchange(&Gate, 0); + LeaveCriticalSection(&CriticalSection.section); } private: @@ -99,18 +124,15 @@ void freeResources() ToUtf8WithSub = nullptr; } -} // namespace - -bool IcuLoader::load() +// The caller holds the critical section across loading and the ICU call. +bool load() { - LoaderLock lock; - // Failed attempts also retain a reference so overlapping callers share the - // result. Once all callers unload, a later load can retry. - if (++ReferenceCount > 1) + if (LoadAttempted) { - return isLoaded(); + return Module != nullptr; } + LoadAttempted = true; Module = loadModule(); if (Module == nullptr) { @@ -128,51 +150,44 @@ bool IcuLoader::load() return true; } -void IcuLoader::unload() +} // namespace + +bool IcuLoader::isAvailable() { LoaderLock lock; - if (ReferenceCount == 0) - { - return; - } - - if (--ReferenceCount == 0) - { - freeResources(); - } + return load(); } -bool IcuLoader::isLoaded() +void IcuLoader::unload() { - return Module != nullptr; + LoaderLock lock; + freeResources(); + LoadAttempted = false; } -IcuLoader::StrFromUtf8 IcuLoader::fromUtf8() +bool IcuLoader::fromUtf8(Char* dest, int capacity, int* length, const char* src, int srcLength, ErrorCode* error) { - return FromUtf8; -} + LoaderLock lock; + if (!load()) + { + return false; + } -IcuLoader::StrToUtf8WithSub IcuLoader::toUtf8WithSub() -{ - return ToUtf8WithSub; + FromUtf8(dest, capacity, length, src, srcLength, error); + return true; } -#endif - -IcuScope::IcuScope() +bool IcuLoader::toUtf8WithSub(char* dest, int capacity, int* length, const Char* src, int srcLength, + Char32 substitution, int* substitutions, ErrorCode* error) { -#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) - m_available = IcuLoader::load(); -#elif defined(RTS_HAS_ICU) - m_available = true; -#else - m_available = false; -#endif + LoaderLock lock; + if (!load()) + { + return false; + } + + ToUtf8WithSub(dest, capacity, length, src, srcLength, substitution, substitutions, error); + return true; } -IcuScope::~IcuScope() -{ -#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) - IcuLoader::unload(); #endif -} diff --git a/Dependencies/ICU/ICU/IcuLoader.h b/Dependencies/ICU/ICU/IcuLoader.h index 5af065a025e..b3b383806eb 100644 --- a/Dependencies/ICU/ICU/IcuLoader.h +++ b/Dependencies/ICU/ICU/IcuLoader.h @@ -22,10 +22,10 @@ #include -// Loads and unloads icu.dll with a shared reference count, like BinkLoader and -// MilesLoader. Every load needs a paired unload, even when loading fails. -// Load/unload are synchronized for conversion workers, including on VC6. -// Hold a reference while checking availability or using the resolved functions. +// Loads icu.dll on the first conversion and caches success or failure until unload. +// Conversion calls and unload are serialized with a Windows critical section, +// including on VC6. Function pointers stay private and cannot outlive the DLL. +// An explicit unload releases the DLL and allows the next conversion to retry. class IcuLoader { public: @@ -33,14 +33,15 @@ class IcuLoader typedef unsigned short Char; typedef int Char32; typedef int ErrorCode; - typedef Char* (__cdecl* StrFromUtf8)(Char*, int, int*, const char*, int, ErrorCode*); - typedef char* (__cdecl* StrToUtf8WithSub)(char*, int, int*, const Char*, int, Char32, int*, ErrorCode*); - static bool load(); + static bool isAvailable(); static void unload(); - static bool isLoaded(); - static StrFromUtf8 fromUtf8(); - static StrToUtf8WithSub toUtf8WithSub(); + + // Return false when ICU is unavailable; otherwise call ICU and return true. + // The caller checks error for the conversion result, including preflight. + static bool fromUtf8(Char* dest, int capacity, int* length, const char* src, int srcLength, ErrorCode* error); + static bool toUtf8WithSub(char* dest, int capacity, int* length, const Char* src, int srcLength, + Char32 substitution, int* substitutions, ErrorCode* error); private: IcuLoader(); @@ -49,24 +50,3 @@ class IcuLoader }; #endif - -// Keeps ICU available throughout a conversion or a longer application lifetime. -// Linked ICU needs no explicit DLL ownership; unavailable Windows ICU uses the -// Win32 conversion fallback. Scopes may overlap on different threads. -class IcuScope -{ -public: - IcuScope(); - ~IcuScope(); - - bool isAvailable() const - { - return m_available; - } - -private: - IcuScope(const IcuScope&); - IcuScope& operator=(const IcuScope&); - - bool m_available; -}; diff --git a/Dependencies/ICU/ICU/utf8.cpp b/Dependencies/ICU/ICU/utf8.cpp index c612c9526ff..2bf284a4220 100644 --- a/Dependencies/ICU/ICU/utf8.cpp +++ b/Dependencies/ICU/ICU/utf8.cpp @@ -431,8 +431,12 @@ size_t IcuWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcL IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - IcuLoader::toUtf8WithSub()(dest, static_cast(destLen), &outputLength, - reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); + if (!IcuLoader::toUtf8WithSub(dest, static_cast(destLen), &outputLength, + reinterpret_cast(src), static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error)) + { + return WindowsWideToUtf8(dest, destLen, src, srcLen); + } + if (!IcuConversionSucceeded(error)) { assert(false); @@ -452,8 +456,12 @@ size_t IcuWideToUtf8Len(const wchar_t* src, size_t srcLen) IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - IcuLoader::toUtf8WithSub()(nullptr, 0, &outputLength, reinterpret_cast(src), - static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error); + if (!IcuLoader::toUtf8WithSub(nullptr, 0, &outputLength, reinterpret_cast(src), + static_cast(srcLen), ICU_REPLACEMENT_CHARACTER, nullptr, &error)) + { + return WindowsWideToUtf8Len(src, srcLen); + } + if (!IcuPreflightSucceeded(error)) { assert(false); @@ -477,8 +485,12 @@ size_t IcuUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcL IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - IcuLoader::fromUtf8()(reinterpret_cast(dest), static_cast(destLen), &outputLength, - src, static_cast(srcLen), &error); + if (!IcuLoader::fromUtf8(reinterpret_cast(dest), static_cast(destLen), &outputLength, + src, static_cast(srcLen), &error)) + { + return WindowsUtf8ToWide(dest, destLen, src, srcLen); + } + if (!IcuConversionSucceeded(error)) { if (destLen > 0) @@ -501,7 +513,11 @@ size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) IcuErrorCode error = ICU_ZERO_ERROR; int outputLength = 0; - IcuLoader::fromUtf8()(nullptr, 0, &outputLength, src, static_cast(srcLen), &error); + if (!IcuLoader::fromUtf8(nullptr, 0, &outputLength, src, static_cast(srcLen), &error)) + { + return WindowsUtf8ToWideLen(src, srcLen); + } + if (!IcuPreflightSucceeded(error)) { return UTF8_INVALID; @@ -516,69 +532,22 @@ size_t IcuUtf8ToWideLen(const char* src, size_t srcLen) size_t Wide_To_Utf8_Len(const wchar_t* src, size_t srcLen) { - IcuScope icu; - if (icu.isAvailable()) - { - return IcuWideToUtf8Len(src, srcLen); - } - -#ifdef _WIN32 - return WindowsWideToUtf8Len(src, srcLen); -#else - assert(false); - return 0; -#endif + return IcuWideToUtf8Len(src, srcLen); } size_t Utf8_To_Wide_Len(const char* src, size_t srcLen) { - IcuScope icu; - if (icu.isAvailable()) - { - return IcuUtf8ToWideLen(src, srcLen); - } - -#ifdef _WIN32 - return WindowsUtf8ToWideLen(src, srcLen); -#else - return UTF8_INVALID; -#endif + return IcuUtf8ToWideLen(src, srcLen); } size_t Wide_To_Utf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) { - IcuScope icu; - if (icu.isAvailable()) - { - return IcuWideToUtf8(dest, destLen, src, srcLen); - } - -#ifdef _WIN32 - return WindowsWideToUtf8(dest, destLen, src, srcLen); -#else - assert(false); - return 0; -#endif + return IcuWideToUtf8(dest, destLen, src, srcLen); } size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) { - IcuScope icu; - if (icu.isAvailable()) - { - return IcuUtf8ToWide(dest, destLen, src, srcLen); - } - -#ifdef _WIN32 - return WindowsUtf8ToWide(dest, destLen, src, srcLen); -#else - if (destLen > 0) - { - dest[0] = L'\0'; - } - - return UTF8_INVALID; -#endif + return IcuUtf8ToWide(dest, destLen, src, srcLen); } // A UTF-8 continuation byte matches 10xxxxxx, so it can never start a sequence. diff --git a/Dependencies/ICU/ICU/utf8.h b/Dependencies/ICU/ICU/utf8.h index 1e3ed8e84ee..bf409bda34c 100644 --- a/Dependencies/ICU/ICU/utf8.h +++ b/Dependencies/ICU/ICU/utf8.h @@ -23,7 +23,7 @@ // UTF-8 <-> wide-character conversion backed by ICU4C. // Full ICU packages use linked C APIs. Windows SDK and VC6 builds use IcuLoader -// to load icu.dll and retain its UTF conversion exports for each call. +// to load icu.dll lazily and cache its UTF conversion exports until unload. // Windows builds fall back to Win32 CP_UTF8 if the DLL or required exports are missing. // Include ICU/IcuSupport.h to use the rest of the linked ICU suite from engine code. diff --git a/Dependencies/ICU/README.md b/Dependencies/ICU/README.md index e6818578f8b..c2488d82d1b 100644 --- a/Dependencies/ICU/README.md +++ b/Dependencies/ICU/README.md @@ -11,9 +11,9 @@ Include `ICU/utf8.h` for conversions and `ICU/IcuSupport.h` for linked ICU APIs. VC6 does use ICU when these exports are available. It does not compile against modern ICU headers or expose the full ICU C++ API. -`IcuLoader` has paired `load()` and `unload()` calls, following `BinkLoader` and `MilesLoader`. Every load needs an unload, including failed loads. Overlapping callers share the result, and the last unload releases the DLL and clears the function pointers. A later load can retry. Reference changes are synchronized, including on VC6; callers must retain a reference while using the resolved functions. +`IcuLoader` loads on the first conversion and caches both successful and failed attempts. Later calls reuse the result, including in WorldBuilder and other tools without an enclosing application scope. `unload()` releases the DLL, clears the exports, and permits a later call to retry. Both games explicitly unload after engine teardown; tools also have cleanup at normal module shutdown. -`IcuScope` pairs these calls automatically. Each conversion holds a scope so another thread cannot unload its functions while they are running. Both games retain an additional scope through engine teardown to avoid repeated loading. Tools can retain a scope around batches of conversions too. Full linked ICU does not need explicit ownership. +Loading, conversion calls, and unloading use a Windows `CRITICAL_SECTION`, the same primitive used by the engine's critical-section classes. It is available on VC6 and does not require a dependency on WWLib. Calls are serialized, and unload waits for any active conversion before freeing the DLL. Export pointers remain private to the loader. Conversion workers must stop before static destruction begins. Conversions in Windows SDK builds use the resolved function pointers, avoiding an extra delay-loader reference. If a caller uses other SDK ICU APIs directly, the SDK delay loader owns its reference independently. diff --git a/Generals/Code/GameEngine/Source/Common/GameMain.cpp b/Generals/Code/GameEngine/Source/Common/GameMain.cpp index a8da5ab64c8..ed8ebb709a3 100644 --- a/Generals/Code/GameEngine/Source/Common/GameMain.cpp +++ b/Generals/Code/GameEngine/Source/Common/GameMain.cpp @@ -39,8 +39,6 @@ */ Int GameMain() { - // Retain ICU through engine teardown; worker conversions hold their own references. - IcuScope icu; int exitcode = 0; // initialize the game engine using factory function TheFramePacer = new FramePacer(); @@ -64,6 +62,11 @@ Int GameMain() delete TheGameEngine; TheGameEngine = nullptr; +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + // Release cached ICU after the engine has stopped its conversion workers. + IcuLoader::unload(); +#endif + return exitcode; } diff --git a/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp b/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp index 606c7e0ff5f..fd67cf8b9af 100644 --- a/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp +++ b/GeneralsMD/Code/GameEngine/Source/Common/GameMain.cpp @@ -39,8 +39,6 @@ */ Int GameMain() { - // Retain ICU through engine teardown; worker conversions hold their own references. - IcuScope icu; int exitcode = 0; // initialize the game engine using factory function TheFramePacer = new FramePacer(); @@ -64,6 +62,11 @@ Int GameMain() delete TheGameEngine; TheGameEngine = nullptr; +#if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) + // Release cached ICU after the engine has stopped its conversion workers. + IcuLoader::unload(); +#endif + return exitcode; } From 5a6871e825da42b54c82e1dfba46d6bb8163d476 Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Sun, 20 Sep 2026 14:32:31 -0600 Subject: [PATCH 09/11] fix(icu): Initialize loader lock before startup conversions --- Dependencies/ICU/ICU/IcuLoader.cpp | 53 +++++++++++++++++------------- Dependencies/ICU/README.md | 2 ++ 2 files changed, 33 insertions(+), 22 deletions(-) diff --git a/Dependencies/ICU/ICU/IcuLoader.cpp b/Dependencies/ICU/ICU/IcuLoader.cpp index 7aa084b1e51..5a50cf6d0d3 100644 --- a/Dependencies/ICU/ICU/IcuLoader.cpp +++ b/Dependencies/ICU/ICU/IcuLoader.cpp @@ -20,6 +20,8 @@ #if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) +#include +#include #include #include @@ -35,44 +37,51 @@ HMODULE Module = nullptr; StrFromUtf8 FromUtf8 = nullptr; StrToUtf8WithSub ToUtf8WithSub = nullptr; +CRITICAL_SECTION CriticalSection; +LONG CriticalSectionState = 0; + void freeResources(); -class LoaderCriticalSection +void cleanup() { -public: - LoaderCriticalSection() - { - InitializeCriticalSection(§ion); - } + // Conversion workers must have stopped before process shutdown. + freeResources(); + LoadAttempted = false; + DeleteCriticalSection(&CriticalSection); + InterlockedExchange(&CriticalSectionState, 0); +} - ~LoaderCriticalSection() +void initializeCriticalSection() +{ + // Startup logging can convert strings before global constructors have run. + // VC6 also lacks synchronized local statics. Publish the native lock once; + // this loop only waits during initialization, not during ICU conversions. + while (InterlockedCompareExchange(&CriticalSectionState, 2, 2) != 2) { - // Also release ICU for tools that do not explicitly unload at shutdown. - // Conversion workers must have stopped before static destruction. - freeResources(); - DeleteCriticalSection(§ion); - } - - CRITICAL_SECTION section; - -private: - LoaderCriticalSection(const LoaderCriticalSection&); - LoaderCriticalSection& operator=(const LoaderCriticalSection&); -}; + if (InterlockedCompareExchange(&CriticalSectionState, 1, 0) == 0) + { + InitializeCriticalSection(&CriticalSection); + atexit(cleanup); + InterlockedExchange(&CriticalSectionState, 2); + return; + } -LoaderCriticalSection CriticalSection; + Sleep(0); + } +} class LoaderLock { public: LoaderLock() { - EnterCriticalSection(&CriticalSection.section); + initializeCriticalSection(); + EnterCriticalSection(&CriticalSection); } ~LoaderLock() { - LeaveCriticalSection(&CriticalSection.section); + LeaveCriticalSection(&CriticalSection); } private: diff --git a/Dependencies/ICU/README.md b/Dependencies/ICU/README.md index c2488d82d1b..c4373d99164 100644 --- a/Dependencies/ICU/README.md +++ b/Dependencies/ICU/README.md @@ -15,6 +15,8 @@ VC6 does use ICU when these exports are available. It does not compile against m Loading, conversion calls, and unloading use a Windows `CRITICAL_SECTION`, the same primitive used by the engine's critical-section classes. It is available on VC6 and does not require a dependency on WWLib. Calls are serialized, and unload waits for any active conversion before freeing the DLL. Export pointers remain private to the loader. Conversion workers must stop before static destruction begins. +The critical section initializes on first use, including conversions triggered by startup logging before global constructors have run. An interlocked initialization guard also supports concurrent first calls on VC6; conversion calls use the native critical section. Cleanup is registered when the lock is initialized. + Conversions in Windows SDK builds use the resolved function pointers, avoiding an extra delay-loader reference. If a caller uses other SDK ICU APIs directly, the SDK delay loader owns its reference independently. The loader tries absolute paths in the executable directory and then the Windows system directory. It supports app-local ICU and Unicode installation paths without searching the working directory or `PATH`. A missing DLL or conversion export selects the Win32 fallback; a DLL with missing exports is released immediately. From c1ee4e6a6385a2427d1624c38c08741ff8d846db Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Tue, 22 Sep 2026 09:21:19 -0600 Subject: [PATCH 10/11] fix(icu): Include the moved UTF-8 header from GameInfo Main's LAN name truncation includes WWLib/utf8.h. After that header moves into the ICU library, GameInfo has to include ICU/utf8.h. Co-authored-by: Cursor --- Core/GameEngine/Source/GameNetwork/GameInfo.cpp | 2 +- Dependencies/ICU/ICU/utf8.cpp | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/Core/GameEngine/Source/GameNetwork/GameInfo.cpp b/Core/GameEngine/Source/GameNetwork/GameInfo.cpp index d7ec6355124..8dd9b58c961 100644 --- a/Core/GameEngine/Source/GameNetwork/GameInfo.cpp +++ b/Core/GameEngine/Source/GameNetwork/GameInfo.cpp @@ -44,7 +44,7 @@ #include "GameNetwork/LANAPI.h" // for testing packet size #include "GameNetwork/LANAPICallbacks.h" // for testing packet size #include "WWLib/strtok_r.h" -#include "WWLib/utf8.h" +#include "ICU/utf8.h" diff --git a/Dependencies/ICU/ICU/utf8.cpp b/Dependencies/ICU/ICU/utf8.cpp index 2bf284a4220..94d1763e668 100644 --- a/Dependencies/ICU/ICU/utf8.cpp +++ b/Dependencies/ICU/ICU/utf8.cpp @@ -553,7 +553,7 @@ size_t Utf8_To_Wide(wchar_t* dest, size_t destLen, const char* src, size_t srcLe // A UTF-8 continuation byte matches 10xxxxxx, so it can never start a sequence. static bool Utf8_Is_Continuation_Byte(char c) { - return ((unsigned char)c & 0xC0) == 0x80; + return (static_cast(c) & 0xC0) == 0x80; } size_t Utf8_Truncate_Len(const char* src, size_t srcLen, size_t maxLen) @@ -568,5 +568,6 @@ size_t Utf8_Truncate_Len(const char* src, size_t srcLen, size_t maxLen) { --len; } + return len; } From f1dfa00f1012d6feab3b5a64243108701921371f Mon Sep 17 00:00:00 2001 From: Jacob Ledbetter Date: Tue, 6 Oct 2026 09:18:34 -0600 Subject: [PATCH 11/11] refactor(icu): Reuse lazy initialization and simplify build setup --- .../Source/WWVegas/WWLib/CMakeLists.txt | 4 -- Dependencies/ICU/ICU/IcuLoader.cpp | 57 ++++++------------- Dependencies/ICU/ICU/IcuLoader.h | 1 - Dependencies/ICU/ICU/utf8.cpp | 39 +++---------- Dependencies/ICU/README.md | 2 +- cmake/icu.cmake | 11 +--- 6 files changed, 27 insertions(+), 87 deletions(-) diff --git a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt index cbca9c63baa..1c796f1c72d 100644 --- a/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt +++ b/Core/Libraries/Source/WWVegas/WWLib/CMakeLists.txt @@ -178,7 +178,3 @@ target_link_libraries(core_wwlib PRIVATE core_wwcommon corei_always ) - -target_link_libraries(core_wwlib PUBLIC - core_icu -) diff --git a/Dependencies/ICU/ICU/IcuLoader.cpp b/Dependencies/ICU/ICU/IcuLoader.cpp index 5a50cf6d0d3..af1ed7ea241 100644 --- a/Dependencies/ICU/ICU/IcuLoader.cpp +++ b/Dependencies/ICU/ICU/IcuLoader.cpp @@ -20,7 +20,7 @@ #if defined(RTS_ICU_DYNAMIC) || defined(RTS_HAS_ICU_WINSDK) -#include +#include #include #include #include @@ -37,51 +37,36 @@ HMODULE Module = nullptr; StrFromUtf8 FromUtf8 = nullptr; StrToUtf8WithSub ToUtf8WithSub = nullptr; -CRITICAL_SECTION CriticalSection; -LONG CriticalSectionState = 0; - -void freeResources(); - -void cleanup() +class CriticalSection { - // Conversion workers must have stopped before process shutdown. - freeResources(); - LoadAttempted = false; - DeleteCriticalSection(&CriticalSection); - InterlockedExchange(&CriticalSectionState, 0); -} - -void initializeCriticalSection() -{ - // Startup logging can convert strings before global constructors have run. - // VC6 also lacks synchronized local statics. Publish the native lock once; - // this loop only waits during initialization, not during ICU conversions. - while (InterlockedCompareExchange(&CriticalSectionState, 2, 2) != 2) +public: + CriticalSection() { - if (InterlockedCompareExchange(&CriticalSectionState, 1, 0) == 0) - { - InitializeCriticalSection(&CriticalSection); - atexit(cleanup); - InterlockedExchange(&CriticalSectionState, 2); - return; - } - - Sleep(0); + InitializeCriticalSection(&m_criticalSection); + atexit(IcuLoader::unload); } -} + + void lock() { EnterCriticalSection(&m_criticalSection); } + void unlock() { LeaveCriticalSection(&m_criticalSection); } + +private: + CRITICAL_SECTION m_criticalSection; +}; + +// Usable before global constructors and throughout static destruction, including on VC6. +lazy_static Lock; class LoaderLock { public: LoaderLock() { - initializeCriticalSection(); - EnterCriticalSection(&CriticalSection); + Lock.get().lock(); } ~LoaderLock() { - LeaveCriticalSection(&CriticalSection); + Lock.get().unlock(); } private: @@ -161,12 +146,6 @@ bool load() } // namespace -bool IcuLoader::isAvailable() -{ - LoaderLock lock; - return load(); -} - void IcuLoader::unload() { LoaderLock lock; diff --git a/Dependencies/ICU/ICU/IcuLoader.h b/Dependencies/ICU/ICU/IcuLoader.h index b3b383806eb..7c1276a5d72 100644 --- a/Dependencies/ICU/ICU/IcuLoader.h +++ b/Dependencies/ICU/ICU/IcuLoader.h @@ -34,7 +34,6 @@ class IcuLoader typedef int Char32; typedef int ErrorCode; - static bool isAvailable(); static void unload(); // Return false when ICU is unavailable; otherwise call ICU and return true. diff --git a/Dependencies/ICU/ICU/utf8.cpp b/Dependencies/ICU/ICU/utf8.cpp index 94d1763e668..0e6f08398f4 100644 --- a/Dependencies/ICU/ICU/utf8.cpp +++ b/Dependencies/ICU/ICU/utf8.cpp @@ -43,24 +43,6 @@ bool FitsInt(size_t length) #ifdef _WIN32 -size_t WindowsWideToUtf8Len(const wchar_t* src, size_t srcLen) -{ - if (!FitsInt(srcLen)) - { - assert(false); - return 0; - } - - const int outputLength = WideCharToMultiByte(CP_UTF8, 0, src, static_cast(srcLen), nullptr, 0, nullptr, nullptr); - if (outputLength == 0 && srcLen != 0) - { - assert(false); - return 0; - } - - return static_cast(outputLength); -} - size_t WindowsWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t srcLen) { if (!FitsInt(destLen) || !FitsInt(srcLen)) @@ -85,21 +67,9 @@ size_t WindowsWideToUtf8(char* dest, size_t destLen, const wchar_t* src, size_t return static_cast(outputLength); } -size_t WindowsUtf8ToWideLen(const char* src, size_t srcLen) +size_t WindowsWideToUtf8Len(const wchar_t* src, size_t srcLen) { - if (!FitsInt(srcLen)) - { - return UTF8_INVALID; - } - - const int outputLength = MultiByteToWideChar(CP_UTF8, MB_ERR_INVALID_CHARS, src, - static_cast(srcLen), nullptr, 0); - if (outputLength == 0 && srcLen != 0) - { - return UTF8_INVALID; - } - - return static_cast(outputLength); + return WindowsWideToUtf8(nullptr, 0, src, srcLen); } size_t WindowsUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t srcLen) @@ -134,6 +104,11 @@ size_t WindowsUtf8ToWide(wchar_t* dest, size_t destLen, const char* src, size_t return static_cast(outputLength); } +size_t WindowsUtf8ToWideLen(const char* src, size_t srcLen) +{ + return WindowsUtf8ToWide(nullptr, 0, src, srcLen); +} + #endif #if defined(RTS_HAS_ICU) && !defined(RTS_HAS_ICU_WINSDK) diff --git a/Dependencies/ICU/README.md b/Dependencies/ICU/README.md index c4373d99164..c8becee1142 100644 --- a/Dependencies/ICU/README.md +++ b/Dependencies/ICU/README.md @@ -15,7 +15,7 @@ VC6 does use ICU when these exports are available. It does not compile against m Loading, conversion calls, and unloading use a Windows `CRITICAL_SECTION`, the same primitive used by the engine's critical-section classes. It is available on VC6 and does not require a dependency on WWLib. Calls are serialized, and unload waits for any active conversion before freeing the DLL. Export pointers remain private to the loader. Conversion workers must stop before static destruction begins. -The critical section initializes on first use, including conversions triggered by startup logging before global constructors have run. An interlocked initialization guard also supports concurrent first calls on VC6; conversion calls use the native critical section. Cleanup is registered when the lock is initialized. +The critical section uses `Utility/lazy_static.h` for thread-safe first use on VC6, including startup logging before global constructors have run. The lock remains usable throughout static destruction. DLL cleanup is registered when the lock is initialized. Conversions in Windows SDK builds use the resolved function pointers, avoiding an extra delay-loader reference. If a caller uses other SDK ICU APIs directly, the SDK delay loader owns its reference independently. diff --git a/cmake/icu.cmake b/cmake/icu.cmake index fa114191dd4..88d82e9521d 100644 --- a/cmake/icu.cmake +++ b/cmake/icu.cmake @@ -17,16 +17,7 @@ set(RTS_ICU_WINSDK FALSE) set(RTS_ICU_DYNAMIC FALSE) if(NOT IS_VS6_BUILD) - find_package(ICU QUIET COMPONENTS uc i18n data) - if(NOT ICU_FOUND) - find_package(ICU QUIET COMPONENTS uc i18n) - endif() - if(NOT ICU_FOUND) - find_package(ICU QUIET COMPONENTS uc data) - endif() - if(NOT ICU_FOUND) - find_package(ICU QUIET COMPONENTS uc) - endif() + find_package(ICU QUIET COMPONENTS uc OPTIONAL_COMPONENTS i18n data) endif() if(ICU_FOUND AND NOT IS_VS6_BUILD)