chat : add dedicated Ling 3.0 (Bailing V3) parser (#28682)

* chat: add dedicated Ling 3.0 (Bailing V3) parser

Ling 3.0 Flash templates pre-open the think block in the generation
prompt, so the model never emits an opening <think>, and a tool call can
arrive before any </think>. The generated autoparser terminated reasoning
only at the close tag, which classified such tool calls entirely as
reasoning_content: clients received content="" with no tool_calls and
agent loops died as reasoning-only turns.

Adds a specialized parser that terminates reasoning at the think close
tag or at a <tool_call> start, mirroring the hand-written Qwen3-Coder and
Kimi K3 parsers and the reference vLLM/SGLang Ling3 parser (which treats
<tool_call> as an implicit reasoning terminator). Detection is gated on
the <role>...</role> section markers, unique to this family among the
tagged-argument templates.

Adds the Ling 3.0 Flash chat template and tests covering the
unclosed-think tool call (full parse and streaming), healthy closed-think
paths, trailing prose, parallel calls, marker-like strings in argument
values, string-union and non-string argument types, and
reasoning_format=none.

Assisted-by: Kimi Code

* tests : move Ling 3.0 test

---------

Co-authored-by: aetherbird <[email protected]>
Co-authored-by: Alde Rojas <[email protected]>
This commit is contained in:
Toby
2026-09-19 18:35:44 -05:00
committed by GitHub
co-authored by aetherbird Alde Rojas
parent e613ef2c81
commit 59657a613a
6 changed files with 543 additions and 0 deletions
+208
View File
@@ -4621,6 +4621,214 @@ static void test_template_output_peg_parsers(bool detailed_debug) {
}
}
// Ling 3.0 / Bailing V3 dedicated parser
{
auto tst = peg_tester("models/templates/inclusionai-ling-3.0-flash.jinja", detailed_debug);
const std::string get_time_call =
"<tool_call>get_time\n"
"<arg_key>city</arg_key>\n"
"<arg_value>Paris</arg_value>\n"
"</tool_call>";
// A tool call emitted before the think block is closed must be extracted,
// with the preceding text kept as reasoning.
tst.test("I need to check the time first.\n" + get_time_call)
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ get_time_tool })
.expect_reasoning("I need to check the time first.\n")
.expect_tool_calls({ { "get_time", R"({"city": "Paris"})", "" } })
.run();
// Closed think block, prose, then a tool call.
tst.test("Let me check the time.\n</think>\nChecking it now.\n" + get_time_call)
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ get_time_tool })
.expect_reasoning("Let me check the time.\n")
.expect_content("Checking it now.\n")
.expect_tool_calls({ { "get_time", R"({"city": "Paris"})", "" } })
.run();
// Prose after the last tool call is content, not a parse failure.
tst.test(get_time_call + "\nThe time has been checked.")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ get_time_tool })
.expect_content("\nThe time has been checked.")
.expect_tool_calls({ { "get_time", R"({"city": "Paris"})", "" } })
.run();
// Parallel tool calls.
tst.test("</think>\n" + get_time_call + "\n" + get_time_call)
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ get_time_tool })
.parallel_tool_calls(true)
.expect_content("")
.expect_tool_calls({
{ "get_time", R"({"city": "Paris"})", "" },
{ "get_time", R"({"city": "Paris"})", "" },
})
.run();
// Argument values may contain marker-like strings.
tst.test("check this\n</think>\n<tool_call>tool_2req_4opt\n"
"<arg_key>req1</arg_key>\n<arg_value>contains </think> and <tool_call> strings</arg_value>\n"
"<arg_key>req2</arg_key>\n<arg_value>1</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ tool_2req_4opt })
.expect_reasoning("check this\n")
.expect_tool_calls({
{ "tool_2req_4opt", R"({"req1": "contains </think> and <tool_call> strings", "req2": 1})", "" },
})
.run();
// reasoning_format=none keeps extracting tool calls.
tst.test("I need to check the time first.\n" + get_time_call)
.reasoning_format(COMMON_REASONING_FORMAT_NONE)
.tools({ get_time_tool })
.expect_content("I need to check the time first.\n")
.expect_tool_calls({ { "get_time", R"({"city": "Paris"})", "" } })
.run();
// With thinking off the template pre-closes the think block, so the model
// emits bare content: it must not be classified as reasoning.
tst.test("Here is the answer.\nNo think block at all.")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.enable_thinking(false)
.expect_reasoning("")
.expect_content("Here is the answer.\nNo think block at all.")
.run();
tst.test(get_time_call)
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.enable_thinking(false)
.tools({ get_time_tool })
.expect_reasoning("")
.expect_tool_calls({ { "get_time", R"({"city": "Paris"})", "" } })
.run();
// The end-of-turn token may arrive spelled out as text tokens instead of
// the single control token; it must not leak into content.
tst.test("Here is the answer.<|role_end|>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.enable_thinking(false)
.expect_content("Here is the answer.")
.run();
tst.test(get_time_call + "<|role_end|>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ get_time_tool })
.expect_tool_calls({ { "get_time", R"({"city": "Paris"})", "" } })
.run();
// Real output tolerates whitespace variation between tags (the template
// renders historical calls with no newline after the tool name).
tst.test("</think>\n<tool_call>get_time<arg_key>city</arg_key><arg_value>Paris</arg_value></tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ get_time_tool })
.expect_tool_calls({ { "get_time", R"({"city": "Paris"})", "" } })
.run();
// Required arguments may arrive in any order.
tst.test("</think>\n<tool_call>tool_2req_4opt\n"
"<arg_key>req2</arg_key>\n<arg_value>7</arg_value>\n"
"<arg_key>req1</arg_key>\n<arg_value>hello</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ tool_2req_4opt })
.expect_tool_calls({ { "tool_2req_4opt", R"({"req2": 7, "req1": "hello"})", "" } })
.run();
// Optional arguments may follow the required ones.
tst.test("</think>\n<tool_call>tool_2req_4opt\n"
"<arg_key>req1</arg_key>\n<arg_value>hello</arg_value>\n"
"<arg_key>req2</arg_key>\n<arg_value>7</arg_value>\n"
"<arg_key>opt1</arg_key>\n<arg_value>extra</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ tool_2req_4opt })
.expect_tool_calls({ { "tool_2req_4opt", R"({"req1": "hello", "req2": 7, "opt1": "extra"})", "" } })
.run();
// Non-string arguments parse as JSON.
tst.test("</think>\n<tool_call>magic_int\n"
"<arg_key>ref</arg_key>\n<arg_value>42</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ magic_int_tool })
.expect_tool_calls({ { "magic_int", R"({"ref": 42})", "" } })
.run();
// A nullable string accepts a JSON null and raw text.
tst.test("</think>\n<tool_call>set_nullable_str\n"
"<arg_key>name</arg_key>\n<arg_value>null</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ nullable_string_tool })
.expect_tool_calls({ { "set_nullable_str", R"({"name": null})", "" } })
.run();
tst.test("</think>\n<tool_call>set_nullable_str\n"
"<arg_key>name</arg_key>\n<arg_value>hello world</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ nullable_string_tool })
.expect_tool_calls({ { "set_nullable_str", R"({"name": "hello world"})", "" } })
.run();
// A raw string that starts like a JSON value must not be taken as JSON:
// the choice falls back to the string alternative.
tst.test("</think>\n<tool_call>set_nullable_str\n"
"<arg_key>name</arg_key>\n<arg_value>123 Main St</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ nullable_string_tool })
.expect_tool_calls({ { "set_nullable_str", R"({"name": "123 Main St"})", "" } })
.run();
// String unions: object and integer values parse as JSON, strings stay raw.
tst.test("</think>\n<tool_call>set_union\n"
"<arg_key>value</arg_key>\n<arg_value>{\"a\": 1}</arg_value>\n"
"<arg_key>amount</arg_key>\n<arg_value>7</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ string_union_tool })
.expect_tool_calls({ { "set_union", R"({"value": {"a": 1}, "amount": 7})", "" } })
.run();
tst.test("</think>\n<tool_call>set_union\n"
"<arg_key>value</arg_key>\n<arg_value>plain text</arg_value>\n"
"<arg_key>amount</arg_key>\n<arg_value>1abc</arg_value>\n"
"</tool_call>")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.tools({ string_union_tool })
.expect_tool_calls({ { "set_union", R"({"value": "plain text", "amount": "1abc"})", "" } })
.run();
// Continuation: the partial assistant turn is spliced back into the prompt.
common_chat_msg prefill = simple_assist_msg("", "I'm thinking");
tst.test("Hello, world!")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.enable_thinking(true)
.messages({ message_user, prefill })
.add_generation_prompt(false)
.continue_final_message(COMMON_CHAT_CONTINUATION_CONTENT)
.expect_reasoning("I'm thinking")
.expect_content("Hello, world!")
.run();
tst.test(" more</think>Hello, world!")
.reasoning_format(COMMON_REASONING_FORMAT_DEEPSEEK)
.enable_thinking(true)
.messages({ message_user, prefill })
.add_generation_prompt(false)
.continue_final_message(COMMON_CHAT_CONTINUATION_REASONING)
.expect_reasoning("I'm thinking more")
.expect_content("Hello, world!")
.run();
}
// Kimi-K3 tests - custom parser
// Unique feature: XTML tags built from <|open|>/<|close|>/<|sep|>, and a
// generation prompt that leaves the think section already open.