Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
44 changes: 41 additions & 3 deletions common/chat.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
#include "json-schema-to-grammar.h"
#include "json.h"
#include "log.h"
#include "unicode.h"

#include "jinja/value.h"
#include "jinja/runtime.h"
Expand Down Expand Up @@ -3845,6 +3846,29 @@ common_chat_msg common_chat_peg_parse(const common_peg_arena & src_pars
common_peg_parse_context ctx(effective_input, flags);
auto result = parser.parse(ctx);

// Kept alive in outer scope: on a recovered parse its AST text views
// (std::string_view into ctx.input) must stay valid until mapping is done.
common_peg_parse_context retry_ctx(flags);
bool recovered_malformed_utf8 = false;
if (result.fail()) {
// The model can emit malformed UTF-8 (e.g. a truncated multi-byte
// character), which makes any/chars/until parsers fail the whole
// message. Retry once on a copy where malformed sequences are
// replaced by U+FFFD; a trailing truncated sequence is kept so that
// partial input can still complete.
std::string sanitized;
if (common_utf8_sanitize(effective_input, sanitized)) {
retry_ctx.input = std::move(sanitized);
auto retry_result = parser.parse(retry_ctx);
if (!retry_result.fail()) {
result = retry_result;
recovered_malformed_utf8 = true;
}
}
}

common_peg_parse_context & active_ctx = recovered_malformed_utf8 ? retry_ctx : ctx;

if (result.fail()) {
// During partial parsing, return partial results if any AST nodes were captured
// This allows streaming to work correctly for formats like FUNC_MARKDOWN_CODE_BLOCK
Expand Down Expand Up @@ -3884,10 +3908,24 @@ common_chat_msg common_chat_peg_parse(const common_peg_arena & src_pars
} else {
mapper = std::make_unique<common_chat_peg_mapper>(msg);
}
mapper->from_ast(ctx.ast, result);
mapper->from_ast(active_ctx.ast, result);

if (recovered_malformed_utf8) {
// Recovery must not turn a corrupted tool call into an executable one:
// malformed bytes inside a tool call surface as U+FFFD in its fields.
for (const auto & tool_call : msg.tool_calls) {
if (tool_call.id.find("\xef\xbf\xbd") != std::string::npos ||
tool_call.name.find("\xef\xbf\xbd") != std::string::npos ||
tool_call.arguments.find("\xef\xbf\xbd") != std::string::npos) {
LOG_WRN("%s: malformed UTF-8 inside %s tool call\n", __func__, common_chat_format_name(params.format));
throw std::runtime_error(std::string("The model produced output that does not match the expected ") +
common_chat_format_name(params.format) + " format");
}
}
}

if (ctx.is_debug()) {
fprintf(stderr, "\nAST for %s parse:\n%s\n", is_partial ? "partial" : "full", ctx.ast.dump().c_str());
if (active_ctx.is_debug()) {
fprintf(stderr, "\nAST for %s parse:\n%s\n", is_partial ? "partial" : "full", active_ctx.ast.dump().c_str());
fflush(stderr);
}

Expand Down
37 changes: 37 additions & 0 deletions common/unicode.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -69,6 +69,43 @@ utf8_parse_result common_parse_utf8_codepoint(std::string_view input, size_t off
return utf8_parse_result(utf8_parse_result::INVALID);
}

bool common_utf8_sanitize(const std::string & input, std::string & output) {
output.clear();

size_t pos = 0;
size_t last = 0; // start of the pending run of valid input
while (pos < input.size()) {
auto result = common_parse_utf8_codepoint(input, pos);
if (result.status == utf8_parse_result::SUCCESS) {
pos += result.bytes_consumed;
continue;
}
if (result.status == utf8_parse_result::INCOMPLETE) {
// Only possible at the tail of input; keep it so that a partial
// input can still complete when more data arrives.
break;
}
// INVALID: replace the maximal invalid subpart (lead byte plus any
// valid continuation bytes it carries) with a single U+FFFD.
output.append(input, last, pos - last);
output.append("\xef\xbf\xbd");

size_t advance = 1;
size_t expect = common_utf8_sequence_length(static_cast<unsigned char>(input[pos]));
while (advance < expect && pos + advance < input.size() &&
(static_cast<unsigned char>(input[pos + advance]) & 0xc0) == 0x80) {
++advance;
}
pos += advance;
last = pos;
}
if (last == 0) {
return false;
}
output.append(input, last, std::string::npos);
return true;
}

bool common_utf8_is_complete(const std::string & s) {
if (s.empty()) {
return true;
Expand Down
6 changes: 6 additions & 0 deletions common/unicode.h
Original file line number Diff line number Diff line change
Expand Up @@ -26,5 +26,11 @@ bool common_utf8_is_complete(const std::string & s);
// Parse a single UTF-8 codepoint from input
utf8_parse_result common_parse_utf8_codepoint(std::string_view input, size_t offset);

// Replace each malformed UTF-8 sequence in input with U+FFFD and store the
// result in output. A truncated sequence at the very end of input is copied
// as-is so that a partial input can still complete when more data arrives.
// Returns false (leaving output untouched) when input is already valid UTF-8.
bool common_utf8_sanitize(const std::string & input, std::string & output);

std::string common_unicode_cpts_to_utf8(const std::vector<uint32_t> & cps);
std::string common_unicode_cpt_to_utf8(uint32_t cpt);
199 changes: 199 additions & 0 deletions tests/test-chat.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -2092,6 +2092,204 @@ static void test_lfm2_parser(const std::string & template_path, bool detailed_de

}


// Regression for production HTTP 500 on malformed model output: the logged
// "unparsed peg-native output" text is effective_input.substr(result.end), i.e.
// the suffix starting at the failed region, not the full model output. The
// fixture below reconstructs a plausible output around that real suffix: a
// reasoning block closed with </think>, then a content region that carries two
// malformed sequences (EB 82 followed by '.', a truncated 3-byte character).
static const std::string malformed_utf8_suffix =
"바다 위를 나는 갈매기는 아침 햇살을 받아 하얗게 빛났다. 날개가 바람에 실려 천천히, 그러나 단호하게 펴지며 파도 위를 스쳤다. 그 아래로 끝없이 펼쳐진 바다는 아직 어둡고, 그러나 햇살이 닿는 곳마다 은빛 물결이 살아 움직이는 듯했다.\n"
"\n"
"갈매기는 한 번 더 날개를 접었다. 그 순간, 바람이 날개 아래로 밀어 올려주고, 다시 한 번 힘차게 펴졌다. 그 반복 속에서, 마치 바다가 그에게 날아갈 힘을 빌려주는 것 같았다.\n"
"\n"
"연안에 서 있는 낯선한 낯선한 "
"\xeb\x82"
"... 아니, 연안에서 낯선 "
"\xeb\x82"
"...\n"
"\n"
"---\n"
"\n"
"죄송합니다. 반복된 문장 때문에 자연스럽게 이어쓰기 어렵습니다. 원하시는 방향을 알려주시면 다시 작성드리겠습니다. 예를 들어:\n"
"\n"
"- **시적·수상적**으로 이어쓰기 (분위기, 감각 중심)\n"
"- **사건·이야기**로 이어쓰기 (갈매기를 따라가는 인물, 사건 발생)\n"
"- **단순히 한 문장만** 자연스럽게 확장하기\n"
"\n"
"어떤 방향이 좋으신가요?";

static const std::string malformed_utf8_suffix_sanitized =
"바다 위를 나는 갈매기는 아침 햇살을 받아 하얗게 빛났다. 날개가 바람에 실려 천천히, 그러나 단호하게 펴지며 파도 위를 스쳤다. 그 아래로 끝없이 펼쳐진 바다는 아직 어둡고, 그러나 햇살이 닿는 곳마다 은빛 물결이 살아 움직이는 듯했다.\n"
"\n"
"갈매기는 한 번 더 날개를 접었다. 그 순간, 바람이 날개 아래로 밀어 올려주고, 다시 한 번 힘차게 펴졌다. 그 반복 속에서, 마치 바다가 그에게 날아갈 힘을 빌려주는 것 같았다.\n"
"\n"
"연안에 서 있는 낯선한 낯선한 �... 아니, 연안에서 낯선 �...\n"
"\n"
"---\n"
"\n"
"죄송합니다. 반복된 문장 때문에 자연스럽게 이어쓰기 어렵습니다. 원하시는 방향을 알려주시면 다시 작성드리겠습니다. 예를 들어:\n"
"\n"
"- **시적·수상적**으로 이어쓰기 (분위기, 감각 중심)\n"
"- **사건·이야기**로 이어쓰기 (갈매기를 따라가는 인물, 사건 발생)\n"
"- **단순히 한 문장만** 자연스럽게 확장하기\n"
"\n"
"어떤 방향이 좋으신가요?";

static void test_chat_malformed_utf8(bool detailed_debug) {
LOG_DBG("%s\n", __func__);

auto tst = peg_tester("models/templates/Qwen3.5-4B.jinja", detailed_debug);

// malformed UTF-8 inside the reasoning region is recovered as U+FFFD
tst.test("reasoning with bad \xeb\x82 bytes.</think>\n\nDone.")
.reasoning_format(COMMON_REASONING_FORMAT_AUTO)
.enable_thinking(true)
.expect_reasoning("reasoning with bad \xef\xbf\xbd bytes.")
.expect_content("Done.")
.run();

// reconstructed real failure: reasoning closed, malformed bytes in content
tst.test("short reasoning.\n</think>\n\n" + malformed_utf8_suffix)
.reasoning_format(COMMON_REASONING_FORMAT_AUTO)
.enable_thinking(true)
.expect_reasoning("short reasoning.")
.expect_content(malformed_utf8_suffix_sanitized)
.run();

// reasoning region itself never terminates (no </think>): the malformed
// bytes must still not abort the parse
tst.test(malformed_utf8_suffix)
.reasoning_format(COMMON_REASONING_FORMAT_AUTO)
.enable_thinking(true)
.expect_reasoning(malformed_utf8_suffix_sanitized)
.run();

// content-only path (thinking disabled / no reasoning extraction): the
// empty <think> prefill from the generation prompt stays in content
tst.test("plain \xeb\x82 text")
.reasoning_format(COMMON_REASONING_FORMAT_NONE)
.enable_thinking(false)
.expect_content("<think>\n\n</think>\n\nplain \xef\xbf\xbd text")
.run();

auto tmpls = read_templates("models/templates/Qwen3.5-4B.jinja");

common_chat_templates_inputs tool_inputs;
tool_inputs.messages = { message_user };
tool_inputs.add_generation_prompt = true;
tool_inputs.enable_thinking = true;
tool_inputs.reasoning_format = COMMON_REASONING_FORMAT_AUTO;
tool_inputs.tools = { run_in_terminal_tool };
// REQUIRED so a corrupted tool call cannot be silently dropped by the
// optional tool-call rule
tool_inputs.tool_choice = COMMON_CHAT_TOOL_CHOICE_REQUIRED;

auto tool_parser = make_peg_parser(tmpls.get(), tool_inputs, detailed_debug);

// negative: malformed UTF-8 inside a tool call argument must not be
// repaired into an executable call - the parse must still fail
{
bool threw = false;
try {
tool_parser.parse("<tool_call>\n<function=run_in_terminal>\n<parameter=command>\npw\xeb\x82" "d\n</parameter>\n</function>\n</tool_call>", false);
} catch (const std::exception &) {
threw = true;
}
if (!threw) {
throw std::runtime_error("expected failure for corrupted tool call argument");
}
}

// negative: malformed UTF-8 breaking a structural literal must still fail
{
bool threw = false;
try {
tool_parser.parse("<tool_call>\n<function=run_in_ter\xeb\x82minal>\n<parameter=command>\npwd\n</parameter>\n</function>\n</tool_call>", false);
} catch (const std::exception &) {
threw = true;
}
if (!threw) {
throw std::runtime_error("expected failure for corrupted tool call structure");
}
}

// negative: malformed UTF-8 inside a non-string (JSON) argument
{
common_chat_templates_inputs int_tool_inputs;
int_tool_inputs.messages = { message_user };
int_tool_inputs.add_generation_prompt = true;
int_tool_inputs.enable_thinking = false;
int_tool_inputs.reasoning_format = COMMON_REASONING_FORMAT_AUTO;
int_tool_inputs.tools = { special_function_tool };
int_tool_inputs.tool_choice = COMMON_CHAT_TOOL_CHOICE_REQUIRED;
auto int_tool_parser = make_peg_parser(tmpls.get(), int_tool_inputs, detailed_debug);
bool threw = false;
try {
int_tool_parser.parse("<tool_call>\n<function=special_function>\n<parameter=arg1>\n1\xeb\x82\n</parameter>\n</function>\n</tool_call>", false);
} catch (const std::exception &) {
threw = true;
}
if (!threw) {
throw std::runtime_error("expected failure for corrupted JSON tool argument");
}
}

// a trailing truncated sequence is preserved, not committed to U+FFFD:
// in partial input more bytes can still complete it, and on a final
// parse it keeps the same NEED_MORE semantics as before (content up to
// the tail is emitted, the incomplete bytes are withheld, no failure)
{
common_chat_templates_inputs partial_inputs;
partial_inputs.messages = { message_user };
partial_inputs.add_generation_prompt = true;
partial_inputs.enable_thinking = false;
partial_inputs.reasoning_format = COMMON_REASONING_FORMAT_NONE;
auto partial_parser = make_peg_parser(tmpls.get(), partial_inputs, detailed_debug);
auto msg = partial_parser.parse(std::string("plain text \xeb\x82"), /* is_partial = */ true);
assert_not_contains(msg.content, "\xef\xbf\xbd");
assert_not_contains(msg.content, "\xeb");

auto msg_final = partial_parser.parse(std::string("plain text \xeb\x82"), /* is_partial = */ false);
if (msg_final.content.find("plain text") == std::string::npos) {
throw std::runtime_error("final parse dropped content before incomplete tail");
}
assert_not_contains(msg_final.content, "\xef\xbf\xbd");
}

// generic content parser (empty arena -> content(rest) + end): malformed
// bytes in plain input are recovered the same way
{
common_chat_parser_params generic_params;
auto msg = common_chat_parse("plain \xeb\x82 text", false, generic_params);
if (msg.content != "plain \xef\xbf\xbd text") {
throw std::runtime_error("unexpected generic parse content: " + msg.content);
}
}

// structural mismatch without malformed UTF-8 still fails as before
{
common_chat_templates_inputs plain_inputs;
plain_inputs.messages = { message_user };
plain_inputs.add_generation_prompt = true;
plain_inputs.enable_thinking = false;
plain_inputs.reasoning_format = COMMON_REASONING_FORMAT_NONE;
plain_inputs.tools = { run_in_terminal_tool };
plain_inputs.tool_choice = COMMON_CHAT_TOOL_CHOICE_REQUIRED;
auto strict_parser = make_peg_parser(tmpls.get(), plain_inputs, detailed_debug);
bool threw = false;
try {
strict_parser.parse("<tool_call>\n<function=nonexistent_tool>\n</function>\n</tool_call>", false);
} catch (const std::exception &) {
threw = true;
}
if (!threw) {
throw std::runtime_error("expected failure for structural mismatch");
}
}
}

static void test_template_output_peg_parsers(bool detailed_debug) {
LOG_DBG("%s\n", __func__);

Expand Down Expand Up @@ -7239,6 +7437,7 @@ int main(int argc, char ** argv) {
test_reasoning_budget_tokens_per_request();
test_reasoning_budget_message_per_request();
test_template_output_peg_parsers(detailed_debug);
test_chat_malformed_utf8(detailed_debug);
std::cout << "\n[chat] All tests passed!" << '\n';
}
return 0;
Expand Down