feat(chat): server-side timings, context display & source hover cards (#4467)

* feat(chat): add server-side timings and context display for GGUF

Extract timings/usage metadata from llama-server SSE stream and forward
through the full stack. Replace client-side estimates with accurate
server-reported metrics (prompt eval, tok/s, token counts, cache hits).
Add context window usage bar to chat top nav.

* feat(chat): source badges with hover cards and 2-row collapse

- Add hover cards to source badges showing favicon, title, URL and
  snippet description on hover
- Limit source badges to 2 rows with +X more expand/collapse
- Parse snippet from web search results for hover card descriptions
- Replace individual Source rendering with grouped SourcesGroup component

* fix(chat): add null guards for server timings edge cases

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* fix(chat): reset contextUsage on thread switch, remove unused context-display

* fix(chat): stop double-counting completion tokens in tool-calling path

* fix(chat): skip metadata events in llm_assist consumers

* fix(chat): hide context usage bar in compare mode

* fix(chat): harden timings pipeline and context usage persistence

Accumulate prompt_ms, predicted_ms, and predicted_n from intermediate
tool-detection passes so the final metadata reflects total server work.
Persist contextUsage in message metadata (Dexie) and restore on thread
load. Add type guard in gguf_stream_chunks for unexpected dict events.
Clear contextUsage when entering compare mode.

* feat(chat): make GGUF stream metadata OpenAI-compatible

* fix(chat): address PR review feedback

* feat(chat): address PR review feedback

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
This commit is contained in:
Wasim Yousef Said 2026-03-21 07:42:01 +01:00 committed by GitHub
commit 50cccfd55e
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
17 changed files with 840 additions and 91 deletions

View file

@ -1459,7 +1459,7 @@ class LlamaCppBackend:
stop: Optional[list[str]] = None,
cancel_event: Optional[threading.Event] = None,
enable_thinking: Optional[bool] = None,
) -> Generator[str, None, None]:
) -> Generator[str | dict, None, None]:
"""
Send a chat completion request to llama-server and stream tokens back.
@ -1490,10 +1490,14 @@ class LlamaCppBackend:
payload["max_tokens"] = max_tokens
if stop:
payload["stop"] = stop
payload["stream_options"] = {"include_usage": True}
url = f"{self.base_url}/v1/chat/completions"
cumulative = ""
in_thinking = False
_stream_done = False
_metadata_usage = None
_metadata_timings = None
try:
# _stream_with_retry uses a 120 s read timeout so prefill
@ -1536,12 +1540,20 @@ class LlamaCppBackend:
# as the main response, not as a thinking block.
cumulative = reasoning_text
yield cumulative
return
_stream_done = True
break # exit inner while
if not line.startswith("data: "):
continue
try:
data = json.loads(line[6:])
# Capture server timings/usage from final chunks
_chunk_timings = data.get("timings")
if _chunk_timings:
_metadata_timings = _chunk_timings
_chunk_usage = data.get("usage")
if _chunk_usage:
_metadata_usage = _chunk_usage
choices = data.get("choices", [])
if choices:
delta = choices[0].get("delta", {})
@ -1570,6 +1582,14 @@ class LlamaCppBackend:
logger.debug(
f"Skipping malformed SSE line: {line[:100]}"
)
if _stream_done:
break # exit outer for
if _metadata_usage or _metadata_timings:
yield {
"type": "metadata",
"usage": _metadata_usage,
"timings": _metadata_timings,
}
except httpx.ConnectError:
raise RuntimeError("Lost connection to llama-server")
@ -1614,6 +1634,9 @@ class LlamaCppBackend:
conversation = list(messages)
url = f"{self.base_url}/v1/chat/completions"
_accumulated_completion_tokens = 0
_accumulated_predicted_ms = 0.0
_accumulated_predicted_n = 0
for iteration in range(max_tool_iterations):
if cancel_event is not None and cancel_event.is_set():
@ -1710,6 +1733,13 @@ class LlamaCppBackend:
)
if finish_reason == "tool_calls" or (tool_calls and len(tool_calls) > 0):
# Only accumulate metrics for responses that are actually used
_accumulated_completion_tokens += data.get("usage", {}).get(
"completion_tokens", 0
)
_iter_timings = data.get("timings", {})
_accumulated_predicted_ms += _iter_timings.get("predicted_ms", 0)
_accumulated_predicted_n += _iter_timings.get("predicted_n", 0)
# Append the assistant message with tool_calls to conversation
assistant_msg = {"role": "assistant", "content": content_text}
if tool_calls:
@ -1805,6 +1835,14 @@ class LlamaCppBackend:
if iteration == 0 and content_text:
yield {"type": "status", "text": ""}
yield {"type": "content", "text": content_text}
_direct_usage = data.get("usage")
_direct_timings = data.get("timings")
if _direct_usage or _direct_timings:
yield {
"type": "metadata",
"usage": _direct_usage,
"timings": _direct_timings,
}
return
# Tools were called in previous iterations; do a final
@ -1834,6 +1872,7 @@ class LlamaCppBackend:
stream_payload["max_tokens"] = max_tokens
if stop:
stream_payload["stop"] = stop
stream_payload["stream_options"] = {"include_usage": True}
import re as _re_final
@ -1862,6 +1901,9 @@ class LlamaCppBackend:
in_thinking = False
has_content_tokens = False
reasoning_text = ""
_metadata_usage = None
_metadata_timings = None
_stream_done = False
try:
stream_timeout = httpx.Timeout(connect = 10, read = 0.5, write = 10, pool = 10)
@ -1899,12 +1941,20 @@ class LlamaCppBackend:
else:
cumulative = reasoning_text
yield {"type": "content", "text": cumulative}
return
_stream_done = True
break # exit inner while
if not line.startswith("data: "):
continue
try:
chunk_data = json.loads(line[6:])
# Capture server timings/usage from final chunks
_chunk_timings = chunk_data.get("timings")
if _chunk_timings:
_metadata_timings = _chunk_timings
_chunk_usage = chunk_data.get("usage")
if _chunk_usage:
_metadata_usage = _chunk_usage
choices = chunk_data.get("choices", [])
if choices:
delta = choices[0].get("delta", {})
@ -1934,6 +1984,42 @@ class LlamaCppBackend:
logger.debug(
f"Skipping malformed SSE line: {line[:100]}"
)
if _stream_done:
break # exit outer for
_final_usage = _metadata_usage or {}
_final_completion = _final_usage.get("completion_tokens", 0)
_final_prompt = _final_usage.get("prompt_tokens", 0)
_total_completion = (
_final_completion + _accumulated_completion_tokens
)
if _metadata_usage or _metadata_timings:
_merged_timings = (
dict(_metadata_timings) if _metadata_timings else {}
)
if _accumulated_predicted_ms or _accumulated_predicted_n:
_merged_timings["predicted_ms"] = (
_merged_timings.get("predicted_ms", 0)
+ _accumulated_predicted_ms
)
_total_predicted_n = (
_merged_timings.get("predicted_n", 0)
+ _accumulated_predicted_n
)
_merged_timings["predicted_n"] = _total_predicted_n
_total_predicted_ms = _merged_timings["predicted_ms"]
if _total_predicted_ms > 0:
_merged_timings["predicted_per_second"] = (
_total_predicted_n / (_total_predicted_ms / 1000.0)
)
yield {
"type": "metadata",
"usage": {
"prompt_tokens": _final_prompt,
"completion_tokens": _total_completion,
"total_tokens": _final_prompt + _total_completion,
},
"timings": _merged_timings,
}
except httpx.ConnectError:
raise RuntimeError("Lost connection to llama-server")