diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index b53fc513de..921c8acf74 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -2473,6 +2473,39 @@ class LlamaCppBackend: _accumulated_predicted_ms = 0.0 _accumulated_predicted_n = 0 + # Flipped once a tool result has been appended to the conversation + # and the system message updated with a synthesise-now directive. + # Small models otherwise keep honouring the initial "prefer tools" + # nudge and loop on search forever, even when the result they need + # is already in context. + _synthesise_nudge_applied = False + # Phrased as a concrete action rather than an opt-out clause -- + # the "do not call more tools" wording actively harmed small-model + # synthesis rate in our benchmarks. + _SYNTHESISE_NUDGE = ( + " Tool results have been gathered. Now write the final answer to the" + " user's original question using what you have. Tool calls are no" + " longer needed for this turn." + ) + + def _apply_synthesise_nudge() -> None: + nonlocal _synthesise_nudge_applied + if _synthesise_nudge_applied: + return + for _msg in conversation: + if _msg.get("role") == "system": + _content = _msg.get("content") or "" + if _SYNTHESISE_NUDGE.strip() not in _content: + _msg["content"] = _content.rstrip() + _SYNTHESISE_NUDGE + _synthesise_nudge_applied = True + return + # No system message yet: insert one + conversation.insert( + 0, + {"role": "system", "content": _SYNTHESISE_NUDGE.lstrip()}, + ) + _synthesise_nudge_applied = True + def _strip_tool_markup(text: str, *, final: bool = False) -> str: if not auto_heal_tool_calls: return text @@ -3135,6 +3168,10 @@ class LlamaCppBackend: tool_msg["tool_call_id"] = tool_call_id conversation.append(tool_msg) + # First tool result of the loop: tell the model to + # synthesise an answer rather than continue searching. + _apply_synthesise_nudge() + # Clear tool status badge before next generation iteration yield {"type": "status", "text": ""} # Continue the loop to let model respond with context diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 4246f0056b..63ce790271 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -127,6 +127,34 @@ _TOOL_ACTION_NUDGE = ( " Do NOT output code blocks -- use the python tool instead." ) +# Softer variant for small models (<9B). The aggressive ALWAYS-CALL-TOOLS +# phrasing above causes small models to pick web_search on every factual +# question even when the answer sits in their training data, and to keep +# calling search after each result instead of synthesising. See +# tests/test_tool_loop_with_nudge.py for the measured behaviour. +_TOOL_ACTION_NUDGE_SMALL = ( + " Call tools only when you need current information or a specific" + " calculation. For questions within your knowledge, answer directly." + " Issue one tool call at a time rather than queuing several at once." +) + +# Appended whenever the current conversation already contains a tool +# result, to counteract the "prefer tools" nudge and push the model +# toward synthesising a final answer from what it has. Matters most for +# small models where the initial "prefer tools" directive is still +# dominating on the second and subsequent turns. +# +# Phrasing matters a lot here -- "do not call more tools" as an opt-out +# clause is actually worse than no nudge (measured 33% vs 57% synthesis +# rate on Qwen3.5-4B UD-Q4_K_XL). Reframing as a concrete action ("write +# the final answer to the user's original question using what you have") +# is what moves the needle: the same bench run jumps to 90% synthesis. +_TOOL_SYNTHESISE_NUDGE = ( + " Tool results have been gathered. Now write the final answer to the" + " user's original question using what you have. Tool calls are no" + " longer needed for this turn." +) + # Regex for stripping leaked tool-call XML from assistant messages/stream _TOOL_XML_RE = _re.compile( r".*?|.*?", @@ -1242,7 +1270,16 @@ async def openai_chat_completions( _nudge = "" if _nudge: - _nudge += _TOOL_ACTION_NUDGE + _nudge += ( + _TOOL_ACTION_NUDGE_SMALL if _is_small_model else _TOOL_ACTION_NUDGE + ) + # If the current conversation already has a tool result + # message, append the synthesise-now directive. Covers + # forked chats and long-running sessions where the prior + # "prefer tools" line keeps biasing the model into search + # loops instead of answering from what it already has. + if any(m.get("role") == "tool" for m in chat_messages): + _nudge += _TOOL_SYNTHESISE_NUDGE # Append nudge to system prompt (preserve user's prompt) if system_prompt: system_prompt = system_prompt.rstrip() + "\n\n" + _nudge @@ -2468,7 +2505,11 @@ async def anthropic_messages( _nudge = "" if _nudge: - _nudge += _TOOL_ACTION_NUDGE + _nudge += ( + _TOOL_ACTION_NUDGE_SMALL if _is_small_model else _TOOL_ACTION_NUDGE + ) + if any(m.get("role") == "tool" for m in openai_messages): + _nudge += _TOOL_SYNTHESISE_NUDGE # Inject into system prompt if openai_messages and openai_messages[0].get("role") == "system": openai_messages[0]["content"] = (