mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-07-21 10:15:53 +00:00
chat: trim messages sent to StepFun parser (fixes long reasoning loops) (#25238)
* chat: trim messages sent to StepFun parser (fixes long reasoning loops) * add regression test; remove duplicate template * chat: trim StepFun content parts before rendering The StepFun trim workaround ran on the already-rendered messages, where typed content parts have been concatenated into a single string, so the per-part whitespace could no longer be reached. Move the trim ahead of rendering and apply it to content_parts text as well as the string content and reasoning_content. Adds a content-parts regression test. Co-Authored-By: Piotr Wilkin <ilintar@gmail.com> Assisted-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: tarruda <tpadilha84@gmail.com>
This commit is contained in:
co-authored by
tarruda
parent
d4cff114c0
commit
2d973636e2
@@ -1887,7 +1887,6 @@ static void test_role_markers_all_templates(testing & t) {
|
||||
{ "Qwen-Qwen3-0.6B.jinja", "<|im_start|>user", "<|im_start|>assistant" },
|
||||
{ "Qwen-QwQ-32B.jinja", "<|im_start|>user", "<|im_start|>assistant" },
|
||||
{ "StepFun3.5-Flash.jinja", "<|im_start|>user", "<|im_start|>assistant" },
|
||||
{ "stepfun-ai-Step-3.5-Flash.jinja", "<|im_start|>user", "<|im_start|>assistant" },
|
||||
|
||||
// DeepSeek family
|
||||
{ "deepseek-ai-DeepSeek-R1-Distill-Llama-8B.jinja", "<|User|>", "<|Assistant|>" },
|
||||
|
||||
@@ -3155,6 +3155,59 @@ static void test_template_output_peg_parsers(bool detailed_debug) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
// StepFun trimming regression test (see https://github.com/ggml-org/llama.cpp/pull/25238)
|
||||
auto tmpls = read_templates("models/templates/StepFun3.5-Flash.jinja");
|
||||
|
||||
common_chat_msg message_chatbot = simple_assist_msg("Let me check.\n\n", "I am thinking.\n\n");
|
||||
|
||||
{
|
||||
common_chat_templates_inputs inputs;
|
||||
inputs.messages = { message_chatbot };
|
||||
inputs.add_generation_prompt = true;
|
||||
|
||||
auto params = common_chat_templates_apply(tmpls.get(), inputs);
|
||||
|
||||
if (params.prompt.find("Let me check.\n\n") != std::string::npos) {
|
||||
throw std::runtime_error("StepFun 3.5: content not trimmed");
|
||||
}
|
||||
|
||||
if (params.prompt.find("I am thinking.\n\n") != std::string::npos) {
|
||||
throw std::runtime_error("StepFun 3.5: reasoning_content not trimmed");
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
// Trimming must also reach typed (text) content parts, not just string content
|
||||
// (see https://github.com/ggml-org/llama.cpp/pull/25238)
|
||||
common_chat_msg message_parts;
|
||||
message_parts.role = "user";
|
||||
message_parts.content_parts = {
|
||||
{ /* .type = */ "text", /* .text = */ "First part.\n\n" },
|
||||
{ /* .type = */ "media_marker", /* .text = */ "<__media__>" },
|
||||
{ /* .type = */ "text", /* .text = */ "Second part.\n\n" },
|
||||
};
|
||||
|
||||
common_chat_templates_inputs inputs;
|
||||
inputs.messages = { message_parts };
|
||||
inputs.add_generation_prompt = true;
|
||||
|
||||
auto params = common_chat_templates_apply(tmpls.get(), inputs);
|
||||
|
||||
if (params.prompt.find("First part.\n\n") != std::string::npos ||
|
||||
params.prompt.find("Second part.\n\n") != std::string::npos) {
|
||||
throw std::runtime_error("StepFun 3.5: text content parts not trimmed");
|
||||
}
|
||||
|
||||
// the trimmed text itself must still be present
|
||||
if (params.prompt.find("First part.") == std::string::npos ||
|
||||
params.prompt.find("Second part.") == std::string::npos) {
|
||||
throw std::runtime_error("StepFun 3.5: text content parts missing after trim");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
{
|
||||
|
||||
Reference in New Issue
Block a user