diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py
index b53fc513de..921c8acf74 100644
--- a/studio/backend/core/inference/llama_cpp.py
+++ b/studio/backend/core/inference/llama_cpp.py
@@ -2473,6 +2473,39 @@ class LlamaCppBackend:
_accumulated_predicted_ms = 0.0
_accumulated_predicted_n = 0
+ # Flipped once a tool result has been appended to the conversation
+ # and the system message updated with a synthesise-now directive.
+ # Small models otherwise keep honouring the initial "prefer tools"
+ # nudge and loop on search forever, even when the result they need
+ # is already in context.
+ _synthesise_nudge_applied = False
+ # Phrased as a concrete action rather than an opt-out clause --
+ # the "do not call more tools" wording actively harmed small-model
+ # synthesis rate in our benchmarks.
+ _SYNTHESISE_NUDGE = (
+ " Tool results have been gathered. Now write the final answer to the"
+ " user's original question using what you have. Tool calls are no"
+ " longer needed for this turn."
+ )
+
+ def _apply_synthesise_nudge() -> None:
+ nonlocal _synthesise_nudge_applied
+ if _synthesise_nudge_applied:
+ return
+ for _msg in conversation:
+ if _msg.get("role") == "system":
+ _content = _msg.get("content") or ""
+ if _SYNTHESISE_NUDGE.strip() not in _content:
+ _msg["content"] = _content.rstrip() + _SYNTHESISE_NUDGE
+ _synthesise_nudge_applied = True
+ return
+ # No system message yet: insert one
+ conversation.insert(
+ 0,
+ {"role": "system", "content": _SYNTHESISE_NUDGE.lstrip()},
+ )
+ _synthesise_nudge_applied = True
+
def _strip_tool_markup(text: str, *, final: bool = False) -> str:
if not auto_heal_tool_calls:
return text
@@ -3135,6 +3168,10 @@ class LlamaCppBackend:
tool_msg["tool_call_id"] = tool_call_id
conversation.append(tool_msg)
+ # First tool result of the loop: tell the model to
+ # synthesise an answer rather than continue searching.
+ _apply_synthesise_nudge()
+
# Clear tool status badge before next generation iteration
yield {"type": "status", "text": ""}
# Continue the loop to let model respond with context
diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py
index 4246f0056b..63ce790271 100644
--- a/studio/backend/routes/inference.py
+++ b/studio/backend/routes/inference.py
@@ -127,6 +127,34 @@ _TOOL_ACTION_NUDGE = (
" Do NOT output code blocks -- use the python tool instead."
)
+# Softer variant for small models (<9B). The aggressive ALWAYS-CALL-TOOLS
+# phrasing above causes small models to pick web_search on every factual
+# question even when the answer sits in their training data, and to keep
+# calling search after each result instead of synthesising. See
+# tests/test_tool_loop_with_nudge.py for the measured behaviour.
+_TOOL_ACTION_NUDGE_SMALL = (
+ " Call tools only when you need current information or a specific"
+ " calculation. For questions within your knowledge, answer directly."
+ " Issue one tool call at a time rather than queuing several at once."
+)
+
+# Appended whenever the current conversation already contains a tool
+# result, to counteract the "prefer tools" nudge and push the model
+# toward synthesising a final answer from what it has. Matters most for
+# small models where the initial "prefer tools" directive is still
+# dominating on the second and subsequent turns.
+#
+# Phrasing matters a lot here -- "do not call more tools" as an opt-out
+# clause is actually worse than no nudge (measured 33% vs 57% synthesis
+# rate on Qwen3.5-4B UD-Q4_K_XL). Reframing as a concrete action ("write
+# the final answer to the user's original question using what you have")
+# is what moves the needle: the same bench run jumps to 90% synthesis.
+_TOOL_SYNTHESISE_NUDGE = (
+ " Tool results have been gathered. Now write the final answer to the"
+ " user's original question using what you have. Tool calls are no"
+ " longer needed for this turn."
+)
+
# Regex for stripping leaked tool-call XML from assistant messages/stream
_TOOL_XML_RE = _re.compile(
r".*?|.*?",
@@ -1242,7 +1270,16 @@ async def openai_chat_completions(
_nudge = ""
if _nudge:
- _nudge += _TOOL_ACTION_NUDGE
+ _nudge += (
+ _TOOL_ACTION_NUDGE_SMALL if _is_small_model else _TOOL_ACTION_NUDGE
+ )
+ # If the current conversation already has a tool result
+ # message, append the synthesise-now directive. Covers
+ # forked chats and long-running sessions where the prior
+ # "prefer tools" line keeps biasing the model into search
+ # loops instead of answering from what it already has.
+ if any(m.get("role") == "tool" for m in chat_messages):
+ _nudge += _TOOL_SYNTHESISE_NUDGE
# Append nudge to system prompt (preserve user's prompt)
if system_prompt:
system_prompt = system_prompt.rstrip() + "\n\n" + _nudge
@@ -2468,7 +2505,11 @@ async def anthropic_messages(
_nudge = ""
if _nudge:
- _nudge += _TOOL_ACTION_NUDGE
+ _nudge += (
+ _TOOL_ACTION_NUDGE_SMALL if _is_small_model else _TOOL_ACTION_NUDGE
+ )
+ if any(m.get("role") == "tool" for m in openai_messages):
+ _nudge += _TOOL_SYNTHESISE_NUDGE
# Inject into system prompt
if openai_messages and openai_messages[0].get("role") == "system":
openai_messages[0]["content"] = (