diff --git a/common/chat.cpp b/common/chat.cpp index 24618d35aee5..4cc4210a36bd 100644 --- a/common/chat.cpp +++ b/common/chat.cpp @@ -8,6 +8,7 @@ #include "json-schema-to-grammar.h" #include "json.h" #include "log.h" +#include "unicode.h" #include "jinja/value.h" #include "jinja/runtime.h" @@ -3845,6 +3846,29 @@ common_chat_msg common_chat_peg_parse(const common_peg_arena & src_pars common_peg_parse_context ctx(effective_input, flags); auto result = parser.parse(ctx); + // Kept alive in outer scope: on a recovered parse its AST text views + // (std::string_view into ctx.input) must stay valid until mapping is done. + common_peg_parse_context retry_ctx(flags); + bool recovered_malformed_utf8 = false; + if (result.fail()) { + // The model can emit malformed UTF-8 (e.g. a truncated multi-byte + // character), which makes any/chars/until parsers fail the whole + // message. Retry once on a copy where malformed sequences are + // replaced by U+FFFD; a trailing truncated sequence is kept so that + // partial input can still complete. + std::string sanitized; + if (common_utf8_sanitize(effective_input, sanitized)) { + retry_ctx.input = std::move(sanitized); + auto retry_result = parser.parse(retry_ctx); + if (!retry_result.fail()) { + result = retry_result; + recovered_malformed_utf8 = true; + } + } + } + + common_peg_parse_context & active_ctx = recovered_malformed_utf8 ? retry_ctx : ctx; + if (result.fail()) { // During partial parsing, return partial results if any AST nodes were captured // This allows streaming to work correctly for formats like FUNC_MARKDOWN_CODE_BLOCK @@ -3884,10 +3908,24 @@ common_chat_msg common_chat_peg_parse(const common_peg_arena & src_pars } else { mapper = std::make_unique(msg); } - mapper->from_ast(ctx.ast, result); + mapper->from_ast(active_ctx.ast, result); + + if (recovered_malformed_utf8) { + // Recovery must not turn a corrupted tool call into an executable one: + // malformed bytes inside a tool call surface as U+FFFD in its fields. + for (const auto & tool_call : msg.tool_calls) { + if (tool_call.id.find("\xef\xbf\xbd") != std::string::npos || + tool_call.name.find("\xef\xbf\xbd") != std::string::npos || + tool_call.arguments.find("\xef\xbf\xbd") != std::string::npos) { + LOG_WRN("%s: malformed UTF-8 inside %s tool call\n", __func__, common_chat_format_name(params.format)); + throw std::runtime_error(std::string("The model produced output that does not match the expected ") + + common_chat_format_name(params.format) + " format"); + } + } + } - if (ctx.is_debug()) { - fprintf(stderr, "\nAST for %s parse:\n%s\n", is_partial ? "partial" : "full", ctx.ast.dump().c_str()); + if (active_ctx.is_debug()) { + fprintf(stderr, "\nAST for %s parse:\n%s\n", is_partial ? "partial" : "full", active_ctx.ast.dump().c_str()); fflush(stderr); } diff --git a/common/unicode.cpp b/common/unicode.cpp index f71fe56783ff..7fdbbd5a6125 100644 --- a/common/unicode.cpp +++ b/common/unicode.cpp @@ -69,6 +69,43 @@ utf8_parse_result common_parse_utf8_codepoint(std::string_view input, size_t off return utf8_parse_result(utf8_parse_result::INVALID); } +bool common_utf8_sanitize(const std::string & input, std::string & output) { + output.clear(); + + size_t pos = 0; + size_t last = 0; // start of the pending run of valid input + while (pos < input.size()) { + auto result = common_parse_utf8_codepoint(input, pos); + if (result.status == utf8_parse_result::SUCCESS) { + pos += result.bytes_consumed; + continue; + } + if (result.status == utf8_parse_result::INCOMPLETE) { + // Only possible at the tail of input; keep it so that a partial + // input can still complete when more data arrives. + break; + } + // INVALID: replace the maximal invalid subpart (lead byte plus any + // valid continuation bytes it carries) with a single U+FFFD. + output.append(input, last, pos - last); + output.append("\xef\xbf\xbd"); + + size_t advance = 1; + size_t expect = common_utf8_sequence_length(static_cast(input[pos])); + while (advance < expect && pos + advance < input.size() && + (static_cast(input[pos + advance]) & 0xc0) == 0x80) { + ++advance; + } + pos += advance; + last = pos; + } + if (last == 0) { + return false; + } + output.append(input, last, std::string::npos); + return true; +} + bool common_utf8_is_complete(const std::string & s) { if (s.empty()) { return true; diff --git a/common/unicode.h b/common/unicode.h index 9b32fa19d62b..7d5c80383b87 100644 --- a/common/unicode.h +++ b/common/unicode.h @@ -26,5 +26,11 @@ bool common_utf8_is_complete(const std::string & s); // Parse a single UTF-8 codepoint from input utf8_parse_result common_parse_utf8_codepoint(std::string_view input, size_t offset); +// Replace each malformed UTF-8 sequence in input with U+FFFD and store the +// result in output. A truncated sequence at the very end of input is copied +// as-is so that a partial input can still complete when more data arrives. +// Returns false (leaving output untouched) when input is already valid UTF-8. +bool common_utf8_sanitize(const std::string & input, std::string & output); + std::string common_unicode_cpts_to_utf8(const std::vector & cps); std::string common_unicode_cpt_to_utf8(uint32_t cpt); diff --git a/tests/test-chat.cpp b/tests/test-chat.cpp index 7918f0ffcf48..b1e9b23a0ca9 100644 --- a/tests/test-chat.cpp +++ b/tests/test-chat.cpp @@ -2092,6 +2092,204 @@ static void test_lfm2_parser(const std::string & template_path, bool detailed_de } + +// Regression for production HTTP 500 on malformed model output: the logged +// "unparsed peg-native output" text is effective_input.substr(result.end), i.e. +// the suffix starting at the failed region, not the full model output. The +// fixture below reconstructs a plausible output around that real suffix: a +// reasoning block closed with , then a content region that carries two +// malformed sequences (EB 82 followed by '.', a truncated 3-byte character). +static const std::string malformed_utf8_suffix = +"바다 위를 나는 갈매기는 아침 햇살을 받아 하얗게 빛났다. 날개가 바람에 실려 천천히, 그러나 단호하게 펴지며 파도 위를 스쳤다. 그 아래로 끝없이 펼쳐진 바다는 아직 어둡고, 그러나 햇살이 닿는 곳마다 은빛 물결이 살아 움직이는 듯했다.\n" +"\n" +"갈매기는 한 번 더 날개를 접었다. 그 순간, 바람이 날개 아래로 밀어 올려주고, 다시 한 번 힘차게 펴졌다. 그 반복 속에서, 마치 바다가 그에게 날아갈 힘을 빌려주는 것 같았다.\n" +"\n" +"연안에 서 있는 낯선한 낯선한 " +"\xeb\x82" +"... 아니, 연안에서 낯선 " +"\xeb\x82" +"...\n" +"\n" +"---\n" +"\n" +"죄송합니다. 반복된 문장 때문에 자연스럽게 이어쓰기 어렵습니다. 원하시는 방향을 알려주시면 다시 작성드리겠습니다. 예를 들어:\n" +"\n" +"- **시적·수상적**으로 이어쓰기 (분위기, 감각 중심)\n" +"- **사건·이야기**로 이어쓰기 (갈매기를 따라가는 인물, 사건 발생)\n" +"- **단순히 한 문장만** 자연스럽게 확장하기\n" +"\n" +"어떤 방향이 좋으신가요?"; + +static const std::string malformed_utf8_suffix_sanitized = +"바다 위를 나는 갈매기는 아침 햇살을 받아 하얗게 빛났다. 날개가 바람에 실려 천천히, 그러나 단호하게 펴지며 파도 위를 스쳤다. 그 아래로 끝없이 펼쳐진 바다는 아직 어둡고, 그러나 햇살이 닿는 곳마다 은빛 물결이 살아 움직이는 듯했다.\n" +"\n" +"갈매기는 한 번 더 날개를 접었다. 그 순간, 바람이 날개 아래로 밀어 올려주고, 다시 한 번 힘차게 펴졌다. 그 반복 속에서, 마치 바다가 그에게 날아갈 힘을 빌려주는 것 같았다.\n" +"\n" +"연안에 서 있는 낯선한 낯선한 �... 아니, 연안에서 낯선 �...\n" +"\n" +"---\n" +"\n" +"죄송합니다. 반복된 문장 때문에 자연스럽게 이어쓰기 어렵습니다. 원하시는 방향을 알려주시면 다시 작성드리겠습니다. 예를 들어:\n" +"\n" +"- **시적·수상적**으로 이어쓰기 (분위기, 감각 중심)\n" +"- **사건·이야기**로 이어쓰기 (갈매기를 따라가는 인물, 사건 발생)\n" +"- **단순히 한 문장만** 자연스럽게 확장하기\n" +"\n" +"어떤 방향이 좋으신가요?"; + +static void test_chat_malformed_utf8(bool detailed_debug) { + LOG_DBG("%s\n", __func__); + + auto tst = peg_tester("models/templates/Qwen3.5-4B.jinja", detailed_debug); + + // malformed UTF-8 inside the reasoning region is recovered as U+FFFD + tst.test("reasoning with bad \xeb\x82 bytes.\n\nDone.") + .reasoning_format(COMMON_REASONING_FORMAT_AUTO) + .enable_thinking(true) + .expect_reasoning("reasoning with bad \xef\xbf\xbd bytes.") + .expect_content("Done.") + .run(); + + // reconstructed real failure: reasoning closed, malformed bytes in content + tst.test("short reasoning.\n\n\n" + malformed_utf8_suffix) + .reasoning_format(COMMON_REASONING_FORMAT_AUTO) + .enable_thinking(true) + .expect_reasoning("short reasoning.") + .expect_content(malformed_utf8_suffix_sanitized) + .run(); + + // reasoning region itself never terminates (no ): the malformed + // bytes must still not abort the parse + tst.test(malformed_utf8_suffix) + .reasoning_format(COMMON_REASONING_FORMAT_AUTO) + .enable_thinking(true) + .expect_reasoning(malformed_utf8_suffix_sanitized) + .run(); + + // content-only path (thinking disabled / no reasoning extraction): the + // empty prefill from the generation prompt stays in content + tst.test("plain \xeb\x82 text") + .reasoning_format(COMMON_REASONING_FORMAT_NONE) + .enable_thinking(false) + .expect_content("\n\n\n\nplain \xef\xbf\xbd text") + .run(); + + auto tmpls = read_templates("models/templates/Qwen3.5-4B.jinja"); + + common_chat_templates_inputs tool_inputs; + tool_inputs.messages = { message_user }; + tool_inputs.add_generation_prompt = true; + tool_inputs.enable_thinking = true; + tool_inputs.reasoning_format = COMMON_REASONING_FORMAT_AUTO; + tool_inputs.tools = { run_in_terminal_tool }; + // REQUIRED so a corrupted tool call cannot be silently dropped by the + // optional tool-call rule + tool_inputs.tool_choice = COMMON_CHAT_TOOL_CHOICE_REQUIRED; + + auto tool_parser = make_peg_parser(tmpls.get(), tool_inputs, detailed_debug); + + // negative: malformed UTF-8 inside a tool call argument must not be + // repaired into an executable call - the parse must still fail + { + bool threw = false; + try { + tool_parser.parse("\n\n\npw\xeb\x82" "d\n\n\n", false); + } catch (const std::exception &) { + threw = true; + } + if (!threw) { + throw std::runtime_error("expected failure for corrupted tool call argument"); + } + } + + // negative: malformed UTF-8 breaking a structural literal must still fail + { + bool threw = false; + try { + tool_parser.parse("\n\n\npwd\n\n\n", false); + } catch (const std::exception &) { + threw = true; + } + if (!threw) { + throw std::runtime_error("expected failure for corrupted tool call structure"); + } + } + + // negative: malformed UTF-8 inside a non-string (JSON) argument + { + common_chat_templates_inputs int_tool_inputs; + int_tool_inputs.messages = { message_user }; + int_tool_inputs.add_generation_prompt = true; + int_tool_inputs.enable_thinking = false; + int_tool_inputs.reasoning_format = COMMON_REASONING_FORMAT_AUTO; + int_tool_inputs.tools = { special_function_tool }; + int_tool_inputs.tool_choice = COMMON_CHAT_TOOL_CHOICE_REQUIRED; + auto int_tool_parser = make_peg_parser(tmpls.get(), int_tool_inputs, detailed_debug); + bool threw = false; + try { + int_tool_parser.parse("\n\n\n1\xeb\x82\n\n\n", false); + } catch (const std::exception &) { + threw = true; + } + if (!threw) { + throw std::runtime_error("expected failure for corrupted JSON tool argument"); + } + } + + // a trailing truncated sequence is preserved, not committed to U+FFFD: + // in partial input more bytes can still complete it, and on a final + // parse it keeps the same NEED_MORE semantics as before (content up to + // the tail is emitted, the incomplete bytes are withheld, no failure) + { + common_chat_templates_inputs partial_inputs; + partial_inputs.messages = { message_user }; + partial_inputs.add_generation_prompt = true; + partial_inputs.enable_thinking = false; + partial_inputs.reasoning_format = COMMON_REASONING_FORMAT_NONE; + auto partial_parser = make_peg_parser(tmpls.get(), partial_inputs, detailed_debug); + auto msg = partial_parser.parse(std::string("plain text \xeb\x82"), /* is_partial = */ true); + assert_not_contains(msg.content, "\xef\xbf\xbd"); + assert_not_contains(msg.content, "\xeb"); + + auto msg_final = partial_parser.parse(std::string("plain text \xeb\x82"), /* is_partial = */ false); + if (msg_final.content.find("plain text") == std::string::npos) { + throw std::runtime_error("final parse dropped content before incomplete tail"); + } + assert_not_contains(msg_final.content, "\xef\xbf\xbd"); + } + + // generic content parser (empty arena -> content(rest) + end): malformed + // bytes in plain input are recovered the same way + { + common_chat_parser_params generic_params; + auto msg = common_chat_parse("plain \xeb\x82 text", false, generic_params); + if (msg.content != "plain \xef\xbf\xbd text") { + throw std::runtime_error("unexpected generic parse content: " + msg.content); + } + } + + // structural mismatch without malformed UTF-8 still fails as before + { + common_chat_templates_inputs plain_inputs; + plain_inputs.messages = { message_user }; + plain_inputs.add_generation_prompt = true; + plain_inputs.enable_thinking = false; + plain_inputs.reasoning_format = COMMON_REASONING_FORMAT_NONE; + plain_inputs.tools = { run_in_terminal_tool }; + plain_inputs.tool_choice = COMMON_CHAT_TOOL_CHOICE_REQUIRED; + auto strict_parser = make_peg_parser(tmpls.get(), plain_inputs, detailed_debug); + bool threw = false; + try { + strict_parser.parse("\n\n\n", false); + } catch (const std::exception &) { + threw = true; + } + if (!threw) { + throw std::runtime_error("expected failure for structural mismatch"); + } + } +} + static void test_template_output_peg_parsers(bool detailed_debug) { LOG_DBG("%s\n", __func__); @@ -7239,6 +7437,7 @@ int main(int argc, char ** argv) { test_reasoning_budget_tokens_per_request(); test_reasoning_budget_message_per_request(); test_template_output_peg_parsers(detailed_debug); + test_chat_malformed_utf8(detailed_debug); std::cout << "\n[chat] All tests passed!" << '\n'; } return 0;