Studio: stream reasoning tokens in the tool-loop generator (fixes DeepSeek thinking not streaming with a pill on) (#6947)

This commit is contained in:
oobabooga 2026-07-07 19:50:40 -03:00 committed by GitHub
commit a9db53e189
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 338 additions and 18 deletions

View file

@ -221,7 +221,7 @@ def test_structured_tool_call_after_visible_preface_is_executed(monkeypatch):
assert assistant_messages[-1]["tool_calls"][0]["function"]["name"] == "render_html"
def test_buffered_reasoning_answer_emits_backend_summary(monkeypatch):
def test_streamed_reasoning_answer_emits_backend_summary(monkeypatch):
stream = [
_sse({"reasoning_content": "I am thinking."}),
_sse({"reasoning_content": " Still thinking."}),
@ -240,17 +240,236 @@ def test_buffered_reasoning_answer_emits_backend_summary(monkeypatch):
)
)
content_texts = [e["text"] for e in events if e["type"] == "content"]
# Reasoning streams live during BUFFERING instead of arriving as one block:
# each reasoning delta is emitted immediately, wrapped in <think>.
assert content_texts[0] == "<think>I am thinking."
assert content_texts[1] == "<think>I am thinking. Still thinking."
# The final event closes the block and appends the answer.
assert content_texts[-1] == "<think>I am thinking. Still thinking.</think>Final answer."
summary_index = next(
i for i, event in enumerate(events) if event["type"] == "reasoning_summary"
)
content_index = next(i for i, event in enumerate(events) if event["type"] == "content")
assert summary_index < content_index
final_content_index = max(i for i, event in enumerate(events) if event["type"] == "content")
assert summary_index < final_content_index
assert events[summary_index]["duration_ms"] == 62000
assert (
events[content_index]["text"]
== "<think>I am thinking. Still thinking.</think>Final answer."
def test_reasoning_streams_incrementally_with_tools(monkeypatch):
# Regression (DeepSeek "thinking doesn't stream"): with a tool/pill active the
# tool-loop generator must stream reasoning token-by-token like the no-tool
# path, not accumulate it and dump one buffered <think> block.
stream = [
_sse({"reasoning_content": "Step one."}),
_sse({"reasoning_content": " Step two."}),
_sse({"reasoning_content": " Step three."}),
_sse({"content": "Done."}),
_done(),
]
payloads: list[dict] = []
backend = _make_backend(monkeypatch, [stream], payloads)
_patch_monotonic(monkeypatch, [1.0, 2.0, 3.0, 4.0, 4.0])
events = list(
backend.generate_chat_completion_with_tools(
messages = [{"role": "user", "content": "think then answer"}],
tools = [{"type": "function", "function": {"name": "web_search"}}],
max_tool_iterations = 1,
)
)
reasoning_stage = [
e["text"]
for e in events
if e["type"] == "content"
and e["text"].startswith("<think>")
and "</think>" not in e["text"]
]
# One live emission per reasoning delta -- not a single dump.
assert reasoning_stage == [
"<think>Step one.",
"<think>Step one. Step two.",
"<think>Step one. Step two. Step three.",
]
final = [e["text"] for e in events if e["type"] == "content"][-1]
assert final == "<think>Step one. Step two. Step three.</think>Done."
def test_reasoning_only_reply_matches_no_tool_path_with_tools(monkeypatch):
# A reasoning-only turn (whole answer in reasoning_content, no content, no
# tool) with a tool active streams the reasoning live, then resolves to the
# bare reasoning text -- identical to the no-tool generate_chat_completion
# path -- so the non-streaming drain still returns it as `content`, not an
# empty answer.
stream = [
_sse({"reasoning_content": "The capital of France is Paris."}),
_done(),
]
payloads: list[dict] = []
backend = _make_backend(monkeypatch, [stream], payloads)
_patch_monotonic(monkeypatch, [1.0, 5.0, 5.0])
events = list(
backend.generate_chat_completion_with_tools(
messages = [{"role": "user", "content": "just think"}],
tools = [{"type": "function", "function": {"name": "web_search"}}],
max_tool_iterations = 1,
)
)
content_texts = [e["text"] for e in events if e["type"] == "content"]
# Reasoning streamed live during BUFFERING (the fix).
assert content_texts[0] == "<think>The capital of France is Paris."
# Resolves to bare reasoning, matching the no-tool sibling.
assert content_texts[-1] == "The capital of France is Paris."
def test_reasoning_before_structured_tool_closes_think_block(monkeypatch):
# Regression: reasoning streamed live during BUFFERING must be closed with
# </think> before a structured tool_call drains, so consumers without a
# reasoning extractor (Anthropic /v1/messages) never receive an unclosed
# <think>. Mirrors the is_match (XML tool signal) path.
tool_stream = [
_sse({"reasoning_content": "Let me search."}),
*_structured_tool_call("web_search", {"query": "weather"}, "call_1"),
]
final_stream = [
_sse({"content": "It is sunny."}),
_done(),
]
payloads: list[dict] = []
backend = _make_backend(monkeypatch, [tool_stream, final_stream], payloads)
_patch_monotonic(monkeypatch, [1.0, 2.0, 3.0, 4.0, 4.0])
monkeypatch.setattr(
"core.inference.tools.execute_tool", lambda name, arguments, **_kwargs: "sunny"
)
events = list(
backend.generate_chat_completion_with_tools(
messages = [{"role": "user", "content": "weather?"}],
tools = [{"type": "function", "function": {"name": "web_search"}}],
max_tool_iterations = 1,
)
)
tool_start_index = next(i for i, e in enumerate(events) if e["type"] == "tool_start")
content_before_tool = [e["text"] for e in events[:tool_start_index] if e["type"] == "content"]
# Reasoning streamed live, then closed before the tool -- balanced block.
assert content_before_tool[0] == "<think>Let me search."
assert content_before_tool[-1] == "<think>Let me search.</think>"
def _replay_route_reasoning_extractor(cumulatives: list[str]) -> tuple[str, str]:
"""Replay the route's cumulative suffix-diff + reasoning extractor (the
shared core of routes/inference.py gguf_stream_chunks and the tool-loop
consumer) over content snapshots. Returns (visible, reasoning)."""
from routes.inference import _ResponsesReasoningExtractor
extractor = _ResponsesReasoningExtractor(parse_think_markers = True)
prev_text = ""
visible: list[str] = []
reasoning: list[str] = []
for cumulative in cumulatives:
new_text = cumulative[len(prev_text) :]
prev_text = cumulative
if not new_text:
continue
reasoning_delta, visible_delta = extractor.feed(new_text)
if reasoning_delta:
reasoning.append(reasoning_delta)
if visible_delta:
visible.append(visible_delta)
final_reasoning, final_visible = extractor.finish()
if final_reasoning:
reasoning.append(final_reasoning)
if final_visible:
visible.append(final_visible)
return "".join(visible), "".join(reasoning)
def test_reasoning_only_route_output_matches_no_tool_path(monkeypatch):
# Parity contract: a reasoning-only reply must reach the client identically
# whether tools are on or off. Both generators stream <think> live then
# resolve to the bare reasoning text; the route's suffix-diff + extractor
# must therefore produce the same (visible, reasoning) split for both.
stream = [
_sse({"reasoning_content": "The capital"}),
_sse({"reasoning_content": " of France is Paris."}),
_done(),
]
tool_backend = _make_backend(monkeypatch, [list(stream)], [])
_patch_monotonic(monkeypatch, [1.0, 2.0, 2.0])
tool_cumulatives = [
e["text"]
for e in tool_backend.generate_chat_completion_with_tools(
messages = [{"role": "user", "content": "capital of France?"}],
tools = [{"type": "function", "function": {"name": "web_search"}}],
max_tool_iterations = 1,
)
if e.get("type") == "content"
]
no_tool_backend = _make_backend(monkeypatch, [list(stream)], [])
no_tool_cumulatives = [
y
for y in no_tool_backend.generate_chat_completion(
messages = [{"role": "user", "content": "capital of France?"}],
)
if isinstance(y, str)
]
# Both paths stream the reasoning live with the same leading shape. (Raw
# yield lists aren't compared verbatim: the tool path emits a pre-existing
# duplicate trailing event that the route's suffix-diff dedupes.)
assert tool_cumulatives[:3] == no_tool_cumulatives[:3]
# The contract that matters: identical route-level output.
tool_out = _replay_route_reasoning_extractor(tool_cumulatives)
no_tool_out = _replay_route_reasoning_extractor(no_tool_cumulatives)
assert tool_out == no_tool_out
# Pin the shared contract so a change to either path shows up here.
_visible, reasoning = tool_out
assert reasoning == "The capital of France is Paris."
def test_reasoning_before_bare_json_tool_closes_think_block(monkeypatch):
# _drain_silently sibling of the structured-tool close: a bare-JSON tool call
# with a live reasoning prefix must also close </think> before draining, and
# must never leak the drained call text as content.
tool_stream = [
_sse({"reasoning_content": "Searching now."}),
_sse({"content": '{"name":"web_search","arguments":{"query":"weather"}}'}),
_done(),
]
final_stream = [
_sse({"content": "It is sunny."}),
_done(),
]
payloads: list[dict] = []
backend = _make_backend(monkeypatch, [tool_stream, final_stream], payloads)
_patch_monotonic(monkeypatch, [1.0, 2.0, 3.0, 4.0, 4.0])
monkeypatch.setattr(
"core.inference.tools.execute_tool", lambda name, arguments, **_kwargs: "sunny"
)
events = list(
backend.generate_chat_completion_with_tools(
messages = [{"role": "user", "content": "weather?"}],
tools = [{"type": "function", "function": {"name": "web_search"}}],
max_tool_iterations = 1,
)
)
tool_start_index = next(i for i, e in enumerate(events) if e["type"] == "tool_start")
content_before_tool = [e["text"] for e in events[:tool_start_index] if e["type"] == "content"]
assert content_before_tool[0] == "<think>Searching now."
assert content_before_tool[-1] == "<think>Searching now.</think>"
# The bare-JSON call text was drained, never surfaced as content.
assert not any('"name"' in t for t in content_before_tool)
def test_consumed_tool_final_pass_emits_latest_reasoning_summary(monkeypatch):
tool_stream = [