Studio: stream reasoning tokens in the tool-loop generator (fixes DeepSeek thinking not streaming with a pill on) (#6947)
This commit is contained in:
parent
304b8eca7a
commit
a9db53e189
4 changed files with 338 additions and 18 deletions
|
|
@ -221,7 +221,7 @@ def test_structured_tool_call_after_visible_preface_is_executed(monkeypatch):
|
|||
assert assistant_messages[-1]["tool_calls"][0]["function"]["name"] == "render_html"
|
||||
|
||||
|
||||
def test_buffered_reasoning_answer_emits_backend_summary(monkeypatch):
|
||||
def test_streamed_reasoning_answer_emits_backend_summary(monkeypatch):
|
||||
stream = [
|
||||
_sse({"reasoning_content": "I am thinking."}),
|
||||
_sse({"reasoning_content": " Still thinking."}),
|
||||
|
|
@ -240,17 +240,236 @@ def test_buffered_reasoning_answer_emits_backend_summary(monkeypatch):
|
|||
)
|
||||
)
|
||||
|
||||
content_texts = [e["text"] for e in events if e["type"] == "content"]
|
||||
# Reasoning streams live during BUFFERING instead of arriving as one block:
|
||||
# each reasoning delta is emitted immediately, wrapped in <think>.
|
||||
assert content_texts[0] == "<think>I am thinking."
|
||||
assert content_texts[1] == "<think>I am thinking. Still thinking."
|
||||
# The final event closes the block and appends the answer.
|
||||
assert content_texts[-1] == "<think>I am thinking. Still thinking.</think>Final answer."
|
||||
|
||||
summary_index = next(
|
||||
i for i, event in enumerate(events) if event["type"] == "reasoning_summary"
|
||||
)
|
||||
content_index = next(i for i, event in enumerate(events) if event["type"] == "content")
|
||||
assert summary_index < content_index
|
||||
final_content_index = max(i for i, event in enumerate(events) if event["type"] == "content")
|
||||
assert summary_index < final_content_index
|
||||
assert events[summary_index]["duration_ms"] == 62000
|
||||
assert (
|
||||
events[content_index]["text"]
|
||||
== "<think>I am thinking. Still thinking.</think>Final answer."
|
||||
|
||||
|
||||
def test_reasoning_streams_incrementally_with_tools(monkeypatch):
|
||||
# Regression (DeepSeek "thinking doesn't stream"): with a tool/pill active the
|
||||
# tool-loop generator must stream reasoning token-by-token like the no-tool
|
||||
# path, not accumulate it and dump one buffered <think> block.
|
||||
stream = [
|
||||
_sse({"reasoning_content": "Step one."}),
|
||||
_sse({"reasoning_content": " Step two."}),
|
||||
_sse({"reasoning_content": " Step three."}),
|
||||
_sse({"content": "Done."}),
|
||||
_done(),
|
||||
]
|
||||
payloads: list[dict] = []
|
||||
backend = _make_backend(monkeypatch, [stream], payloads)
|
||||
_patch_monotonic(monkeypatch, [1.0, 2.0, 3.0, 4.0, 4.0])
|
||||
|
||||
events = list(
|
||||
backend.generate_chat_completion_with_tools(
|
||||
messages = [{"role": "user", "content": "think then answer"}],
|
||||
tools = [{"type": "function", "function": {"name": "web_search"}}],
|
||||
max_tool_iterations = 1,
|
||||
)
|
||||
)
|
||||
|
||||
reasoning_stage = [
|
||||
e["text"]
|
||||
for e in events
|
||||
if e["type"] == "content"
|
||||
and e["text"].startswith("<think>")
|
||||
and "</think>" not in e["text"]
|
||||
]
|
||||
# One live emission per reasoning delta -- not a single dump.
|
||||
assert reasoning_stage == [
|
||||
"<think>Step one.",
|
||||
"<think>Step one. Step two.",
|
||||
"<think>Step one. Step two. Step three.",
|
||||
]
|
||||
final = [e["text"] for e in events if e["type"] == "content"][-1]
|
||||
assert final == "<think>Step one. Step two. Step three.</think>Done."
|
||||
|
||||
|
||||
def test_reasoning_only_reply_matches_no_tool_path_with_tools(monkeypatch):
|
||||
# A reasoning-only turn (whole answer in reasoning_content, no content, no
|
||||
# tool) with a tool active streams the reasoning live, then resolves to the
|
||||
# bare reasoning text -- identical to the no-tool generate_chat_completion
|
||||
# path -- so the non-streaming drain still returns it as `content`, not an
|
||||
# empty answer.
|
||||
stream = [
|
||||
_sse({"reasoning_content": "The capital of France is Paris."}),
|
||||
_done(),
|
||||
]
|
||||
payloads: list[dict] = []
|
||||
backend = _make_backend(monkeypatch, [stream], payloads)
|
||||
_patch_monotonic(monkeypatch, [1.0, 5.0, 5.0])
|
||||
|
||||
events = list(
|
||||
backend.generate_chat_completion_with_tools(
|
||||
messages = [{"role": "user", "content": "just think"}],
|
||||
tools = [{"type": "function", "function": {"name": "web_search"}}],
|
||||
max_tool_iterations = 1,
|
||||
)
|
||||
)
|
||||
|
||||
content_texts = [e["text"] for e in events if e["type"] == "content"]
|
||||
# Reasoning streamed live during BUFFERING (the fix).
|
||||
assert content_texts[0] == "<think>The capital of France is Paris."
|
||||
# Resolves to bare reasoning, matching the no-tool sibling.
|
||||
assert content_texts[-1] == "The capital of France is Paris."
|
||||
|
||||
|
||||
def test_reasoning_before_structured_tool_closes_think_block(monkeypatch):
|
||||
# Regression: reasoning streamed live during BUFFERING must be closed with
|
||||
# </think> before a structured tool_call drains, so consumers without a
|
||||
# reasoning extractor (Anthropic /v1/messages) never receive an unclosed
|
||||
# <think>. Mirrors the is_match (XML tool signal) path.
|
||||
tool_stream = [
|
||||
_sse({"reasoning_content": "Let me search."}),
|
||||
*_structured_tool_call("web_search", {"query": "weather"}, "call_1"),
|
||||
]
|
||||
final_stream = [
|
||||
_sse({"content": "It is sunny."}),
|
||||
_done(),
|
||||
]
|
||||
payloads: list[dict] = []
|
||||
backend = _make_backend(monkeypatch, [tool_stream, final_stream], payloads)
|
||||
_patch_monotonic(monkeypatch, [1.0, 2.0, 3.0, 4.0, 4.0])
|
||||
|
||||
monkeypatch.setattr(
|
||||
"core.inference.tools.execute_tool", lambda name, arguments, **_kwargs: "sunny"
|
||||
)
|
||||
|
||||
events = list(
|
||||
backend.generate_chat_completion_with_tools(
|
||||
messages = [{"role": "user", "content": "weather?"}],
|
||||
tools = [{"type": "function", "function": {"name": "web_search"}}],
|
||||
max_tool_iterations = 1,
|
||||
)
|
||||
)
|
||||
|
||||
tool_start_index = next(i for i, e in enumerate(events) if e["type"] == "tool_start")
|
||||
content_before_tool = [e["text"] for e in events[:tool_start_index] if e["type"] == "content"]
|
||||
# Reasoning streamed live, then closed before the tool -- balanced block.
|
||||
assert content_before_tool[0] == "<think>Let me search."
|
||||
assert content_before_tool[-1] == "<think>Let me search.</think>"
|
||||
|
||||
|
||||
def _replay_route_reasoning_extractor(cumulatives: list[str]) -> tuple[str, str]:
|
||||
"""Replay the route's cumulative suffix-diff + reasoning extractor (the
|
||||
shared core of routes/inference.py gguf_stream_chunks and the tool-loop
|
||||
consumer) over content snapshots. Returns (visible, reasoning)."""
|
||||
from routes.inference import _ResponsesReasoningExtractor
|
||||
|
||||
extractor = _ResponsesReasoningExtractor(parse_think_markers = True)
|
||||
prev_text = ""
|
||||
visible: list[str] = []
|
||||
reasoning: list[str] = []
|
||||
for cumulative in cumulatives:
|
||||
new_text = cumulative[len(prev_text) :]
|
||||
prev_text = cumulative
|
||||
if not new_text:
|
||||
continue
|
||||
reasoning_delta, visible_delta = extractor.feed(new_text)
|
||||
if reasoning_delta:
|
||||
reasoning.append(reasoning_delta)
|
||||
if visible_delta:
|
||||
visible.append(visible_delta)
|
||||
final_reasoning, final_visible = extractor.finish()
|
||||
if final_reasoning:
|
||||
reasoning.append(final_reasoning)
|
||||
if final_visible:
|
||||
visible.append(final_visible)
|
||||
return "".join(visible), "".join(reasoning)
|
||||
|
||||
|
||||
def test_reasoning_only_route_output_matches_no_tool_path(monkeypatch):
|
||||
# Parity contract: a reasoning-only reply must reach the client identically
|
||||
# whether tools are on or off. Both generators stream <think> live then
|
||||
# resolve to the bare reasoning text; the route's suffix-diff + extractor
|
||||
# must therefore produce the same (visible, reasoning) split for both.
|
||||
stream = [
|
||||
_sse({"reasoning_content": "The capital"}),
|
||||
_sse({"reasoning_content": " of France is Paris."}),
|
||||
_done(),
|
||||
]
|
||||
|
||||
tool_backend = _make_backend(monkeypatch, [list(stream)], [])
|
||||
_patch_monotonic(monkeypatch, [1.0, 2.0, 2.0])
|
||||
tool_cumulatives = [
|
||||
e["text"]
|
||||
for e in tool_backend.generate_chat_completion_with_tools(
|
||||
messages = [{"role": "user", "content": "capital of France?"}],
|
||||
tools = [{"type": "function", "function": {"name": "web_search"}}],
|
||||
max_tool_iterations = 1,
|
||||
)
|
||||
if e.get("type") == "content"
|
||||
]
|
||||
|
||||
no_tool_backend = _make_backend(monkeypatch, [list(stream)], [])
|
||||
no_tool_cumulatives = [
|
||||
y
|
||||
for y in no_tool_backend.generate_chat_completion(
|
||||
messages = [{"role": "user", "content": "capital of France?"}],
|
||||
)
|
||||
if isinstance(y, str)
|
||||
]
|
||||
|
||||
# Both paths stream the reasoning live with the same leading shape. (Raw
|
||||
# yield lists aren't compared verbatim: the tool path emits a pre-existing
|
||||
# duplicate trailing event that the route's suffix-diff dedupes.)
|
||||
assert tool_cumulatives[:3] == no_tool_cumulatives[:3]
|
||||
# The contract that matters: identical route-level output.
|
||||
tool_out = _replay_route_reasoning_extractor(tool_cumulatives)
|
||||
no_tool_out = _replay_route_reasoning_extractor(no_tool_cumulatives)
|
||||
assert tool_out == no_tool_out
|
||||
# Pin the shared contract so a change to either path shows up here.
|
||||
_visible, reasoning = tool_out
|
||||
assert reasoning == "The capital of France is Paris."
|
||||
|
||||
|
||||
def test_reasoning_before_bare_json_tool_closes_think_block(monkeypatch):
|
||||
# _drain_silently sibling of the structured-tool close: a bare-JSON tool call
|
||||
# with a live reasoning prefix must also close </think> before draining, and
|
||||
# must never leak the drained call text as content.
|
||||
tool_stream = [
|
||||
_sse({"reasoning_content": "Searching now."}),
|
||||
_sse({"content": '{"name":"web_search","arguments":{"query":"weather"}}'}),
|
||||
_done(),
|
||||
]
|
||||
final_stream = [
|
||||
_sse({"content": "It is sunny."}),
|
||||
_done(),
|
||||
]
|
||||
payloads: list[dict] = []
|
||||
backend = _make_backend(monkeypatch, [tool_stream, final_stream], payloads)
|
||||
_patch_monotonic(monkeypatch, [1.0, 2.0, 3.0, 4.0, 4.0])
|
||||
|
||||
monkeypatch.setattr(
|
||||
"core.inference.tools.execute_tool", lambda name, arguments, **_kwargs: "sunny"
|
||||
)
|
||||
|
||||
events = list(
|
||||
backend.generate_chat_completion_with_tools(
|
||||
messages = [{"role": "user", "content": "weather?"}],
|
||||
tools = [{"type": "function", "function": {"name": "web_search"}}],
|
||||
max_tool_iterations = 1,
|
||||
)
|
||||
)
|
||||
|
||||
tool_start_index = next(i for i, e in enumerate(events) if e["type"] == "tool_start")
|
||||
content_before_tool = [e["text"] for e in events[:tool_start_index] if e["type"] == "content"]
|
||||
assert content_before_tool[0] == "<think>Searching now."
|
||||
assert content_before_tool[-1] == "<think>Searching now.</think>"
|
||||
# The bare-JSON call text was drained, never surfaced as content.
|
||||
assert not any('"name"' in t for t in content_before_tool)
|
||||
|
||||
|
||||
def test_consumed_tool_final_pass_emits_latest_reasoning_summary(monkeypatch):
|
||||
tool_stream = [
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue