From 4308a4f035791f58ae111f56c39dac598bf476be Mon Sep 17 00:00:00 2001 From: Pascal Date: Tue, 4 Aug 2026 22:24:55 +0200 Subject: [PATCH] server: decode Windows OEM output to UTF-8 in built-in tools (#26597) a child process writes in the OEM code page, which is not UTF-8 on a western Windows install, so accented output reaches the JSON layer as invalid bytes and gets replaced there, silently losing the characters run() spawns without a console, so the child never inherits the console code page and GetOEMCP is the one that applies decode with MB_ERR_INVALID_CHARS so a wrong code page returns the text untouched instead of emitting replacement characters, and pass text that already decodes as UTF-8 through so a child emitting UTF-8 is never decoded twice the check drops an incomplete trailing sequence before validating, since a streamed chunk can end in the middle of a multi-byte character --- tools/server/server-tools.cpp | 54 +++++++++++++++++++++++++++++++++-- 1 file changed, 51 insertions(+), 3 deletions(-) diff --git a/tools/server/server-tools.cpp b/tools/server/server-tools.cpp index e050d03b52..5309de5ef3 100644 --- a/tools/server/server-tools.cpp +++ b/tools/server/server-tools.cpp @@ -17,12 +17,60 @@ #include #include +#if defined(_WIN32) +# ifndef NOMINMAX +# define NOMINMAX +# endif +# include +#endif + namespace fs = std::filesystem; // // internal helpers // +#if defined(_WIN32) +// A chunk can end in the middle of a multi-byte sequence, so the incomplete +// tail is dropped before validating what precedes it. +static bool is_utf8_text(const std::string & text) { + return is_valid_utf8(text.substr(0, validate_utf8(text))); +} + +// A child process writes its output in the OEM code page, which is not UTF-8 +// on a western Windows install, so accented text reaches the JSON layer as +// invalid bytes and is replaced there. Text that already decodes as UTF-8 is +// returned untouched, so a child that emits UTF-8 is never decoded twice. +// run() spawns without a console, so the console code page does not apply. +static std::string console_output_to_utf8(const std::string & text) { + if (text.empty() || is_utf8_text(text)) { + return text; + } + + const UINT cp = GetOEMCP(); + + // fail rather than emit replacement characters when the code page is wrong + const int wide_len = MultiByteToWideChar(cp, MB_ERR_INVALID_CHARS, text.data(), (int) text.size(), nullptr, 0); + if (wide_len <= 0) { + return text; + } + std::wstring wide(wide_len, L'\0'); + MultiByteToWideChar(cp, MB_ERR_INVALID_CHARS, text.data(), (int) text.size(), wide.data(), wide_len); + + const int utf8_len = WideCharToMultiByte(CP_UTF8, 0, wide.data(), wide_len, nullptr, 0, nullptr, nullptr); + if (utf8_len <= 0) { + return text; + } + std::string utf8(utf8_len, '\0'); + WideCharToMultiByte(CP_UTF8, 0, wide.data(), wide_len, utf8.data(), utf8_len, nullptr, nullptr); + return utf8; +} +#else +static std::string console_output_to_utf8(const std::string & text) { + return text; +} +#endif + json server_tool::to_json() const { return { {"display_name", display_name}, @@ -246,14 +294,14 @@ public: size_t len = strlen(buf); if (output.size() + len <= max_output) { output.append(buf, len); - if (on_chunk && !on_chunk(std::string(buf, len))) { + if (on_chunk && !on_chunk(console_output_to_utf8(std::string(buf, len)))) { proc.terminate(); break; } } else { size_t remaining = max_output - output.size(); output.append(buf, remaining); - if (on_chunk && remaining > 0) on_chunk(std::string(buf, remaining)); + if (on_chunk && remaining > 0) on_chunk(console_output_to_utf8(std::string(buf, remaining))); truncated = true; } } @@ -267,7 +315,7 @@ public: res.exit_code = proc.join(); - res.output = output; + res.output = console_output_to_utf8(output); res.timed_out = timed_out.load(); if (truncated) { res.output += "\n[output truncated]";