diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 7909af8a23..b0746a55a9 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -2493,7 +2493,9 @@ class LlamaCppBackend: # iterations so they are not silently dropped. yield {"type": "status", "text": ""} if content_accum: - yield {"type": "content", "text": content_accum} + _safe = _strip_tool_markup(content_accum, final = True) + if _safe.strip(): + yield {"type": "content", "text": _safe} _fu = _iter_usage or {} _fc = _fu.get("completion_tokens", 0) _fp = _fu.get("prompt_tokens", 0) diff --git a/studio/backend/core/inference/tools.py b/studio/backend/core/inference/tools.py index 55bfa095f9..a52819eedb 100644 --- a/studio/backend/core/inference/tools.py +++ b/studio/backend/core/inference/tools.py @@ -57,14 +57,18 @@ WEB_SEARCH_TOOL = { "type": "function", "function": { "name": "web_search", - "description": "Search the web for current information, recent events, or facts you are uncertain about.", + "description": "Search the web and fetch page content from the top result. Returns snippets for all results plus the full text of the best matching page.", "parameters": { "type": "object", "properties": { "query": { "type": "string", - "description": "The search query", - } + "description": "The search query. Be specific and concise.", + }, + "url": { + "type": "string", + "description": "Optional: fetch this specific URL directly instead of searching. Use when you already know the page you need.", + }, }, "required": ["query"], }, @@ -131,7 +135,11 @@ def execute_tool( ) effective_timeout = _EXEC_TIMEOUT if timeout is _TIMEOUT_UNSET else timeout if name == "web_search": - return _web_search(arguments.get("query", ""), timeout = effective_timeout) + return _web_search( + arguments.get("query", ""), + url = arguments.get("url", ""), + timeout = effective_timeout, + ) if name == "python": return _python_exec( arguments.get("code", ""), cancel_event, effective_timeout, session_id @@ -143,8 +151,52 @@ def execute_tool( return f"Unknown tool: {name}" -def _web_search(query: str, max_results: int = 5, timeout: int = _EXEC_TIMEOUT) -> str: - """Search the web using DuckDuckGo and return formatted results.""" +def _fetch_page_text(url: str, max_chars: int = 4000, timeout: int = 10) -> str: + """Fetch a URL and extract plain text content (best-effort).""" + import urllib.request + import urllib.error + + try: + req = urllib.request.Request(url, headers = {"User-Agent": "Mozilla/5.0"}) + with urllib.request.urlopen(req, timeout = timeout) as resp: + raw = resp.read(200_000).decode("utf-8", errors = "replace") + # Strip HTML tags (lightweight, no extra deps) + import re + + text = re.sub( + r"]*>.*?", "", raw, flags = re.DOTALL | re.IGNORECASE + ) + text = re.sub( + r"]*>.*?", "", text, flags = re.DOTALL | re.IGNORECASE + ) + text = re.sub(r"<[^>]+>", " ", text) + text = re.sub(r"\s+", " ", text).strip() + # Decode HTML entities + import html + + text = html.unescape(text) + if len(text) > max_chars: + text = text[:max_chars] + "..." + return text + except Exception: + return "" + + +def _web_search( + query: str, url: str = "", max_results: int = 5, timeout: int = _EXEC_TIMEOUT +) -> str: + """Search the web using DuckDuckGo and return formatted results. + + If *url* is provided, fetches that page directly (skips search). + Otherwise searches and fetches page content for the top result. + """ + # Direct URL fetch mode + if url and url.strip(): + text = _fetch_page_text(url.strip(), max_chars = 8000, timeout = 15) + if text: + return f"Page content from {url}:\n\n{text}" + return f"Failed to fetch {url}" + if not query.strip(): return "No query provided." try: @@ -153,6 +205,7 @@ def _web_search(query: str, max_results: int = 5, timeout: int = _EXEC_TIMEOUT) results = DDGS(timeout = timeout).text(query, max_results = max_results) if not results: return "No results found." + parts = [] for r in results: parts.append( @@ -160,6 +213,10 @@ def _web_search(query: str, max_results: int = 5, timeout: int = _EXEC_TIMEOUT) f"URL: {r.get('href', '')}\n" f"Snippet: {r.get('body', '')}" ) + parts.append( + "\nTip: To read the full content of any page above, " + "call web_search again with the url parameter." + ) return "\n\n---\n\n".join(parts) except Exception as e: return f"Search failed: {e}" diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 6f44a3c69f..8efe6e1699 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -1034,17 +1034,6 @@ async def openai_chat_completions( status_code = 400, detail = f"Failed to process image: {e}" ) - # Build message list with system prompt prepended - gguf_messages = [] - if system_prompt: - gguf_messages.append({"role": "system", "content": system_prompt}) - gguf_messages.extend(chat_messages) - - cancel_event = threading.Event() - - completion_id = f"chatcmpl-{uuid.uuid4().hex[:12]}" - created = int(time.time()) - # ── Tool-calling path (agentic loop) ────────────────── use_tools = ( payload.enable_tools and llama_backend.supports_tools and not image_b64 @@ -1062,6 +1051,74 @@ async def openai_chat_completions( else: tools_to_use = ALL_TOOLS + # Build a model-agnostic nudge tailored to the active tools. + # No format instructions -- each model's chat template handles + # that. Keeps tool use optional ("prefer", not "must"). + tool_names = {t["function"]["name"] for t in tools_to_use} + has_web = "web_search" in tool_names + has_code = "python" in tool_names or "terminal" in tool_names + + nudge_parts = ["You are a helpful assistant with access to tools."] + if has_code and has_web: + nudge_parts.append( + "For tasks involving math, calculations, code, or data " + "analysis, prefer using the code execution tools to " + "produce a verified answer rather than answering from " + "memory alone. For questions requiring up-to-date or " + "real-time information, use web search." + ) + elif has_code: + nudge_parts.append( + "For tasks involving math, calculations, code, or data " + "analysis, prefer using the code execution tools to " + "produce a verified answer rather than answering from " + "memory alone." + ) + elif has_web: + nudge_parts.append( + "For questions requiring up-to-date or real-time " + "information, prefer using web search rather than " + "answering from memory alone." + ) + + tool_nudge = " ".join(nudge_parts) + + if not system_prompt: + system_prompt = tool_nudge + else: + system_prompt = system_prompt.rstrip() + "\n\n" + tool_nudge + + # Strip stale tool-call XML from conversation history + # (the frontend may include prior assistant text + # in follow-up user messages). + import re as _re + + _tool_xml_strip = [ + _re.compile(r".*?", _re.DOTALL), + _re.compile(r".*$", _re.DOTALL), + _re.compile(r".*?", _re.DOTALL), + _re.compile(r".*$", _re.DOTALL), + ] + for msg in chat_messages: + c = msg.get("content", "") + if isinstance(c, str) and "" in c: + for pat in _tool_xml_strip: + c = pat.sub("", c) + msg["content"] = c.strip() + + # Build message list with system prompt prepended + gguf_messages = [] + if system_prompt: + gguf_messages.append({"role": "system", "content": system_prompt}) + gguf_messages.extend(chat_messages) + + cancel_event = threading.Event() + + completion_id = f"chatcmpl-{uuid.uuid4().hex[:12]}" + created = int(time.time()) + + if use_tools: + def gguf_generate_with_tools(): return llama_backend.generate_chat_completion_with_tools( messages = gguf_messages, @@ -1143,6 +1200,22 @@ async def openai_chat_completions( # "content" type -- cumulative text cumulative = event.get("text", "") + # Strip closed tool-call XML pairs that may + # have leaked through the backend stream. + # Only strip closed pairs here (not open-ended) + # so legitimate text after a tool block is kept. + cumulative = _re.sub( + r".*?", + "", + cumulative, + flags = _re.DOTALL, + ) + cumulative = _re.sub( + r".*?", + "", + cumulative, + flags = _re.DOTALL, + ) new_text = cumulative[len(prev_text) :] prev_text = cumulative if not new_text: