diff --git a/studio/backend/core/inference/_html_to_md.py b/studio/backend/core/inference/_html_to_md.py new file mode 100644 index 0000000000..d96b8168e2 --- /dev/null +++ b/studio/backend/core/inference/_html_to_md.py @@ -0,0 +1,439 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 + +""" +Minimal HTML-to-Markdown converter using only the standard library. + +Replaces the external ``html2text`` (GPL-3.0) dependency with a ~250-line +``html.parser.HTMLParser`` subclass. Covers headings, links, bold/italic, +lists, tables, blockquotes, code blocks, and entity decoding. +""" + +from __future__ import annotations + +import html +import re +from html.parser import HTMLParser + +__all__ = ["html_to_markdown"] + +_SKIP_TAGS = frozenset({"script", "style", "head", "noscript", "svg", "math"}) +_BLOCK_TAGS = frozenset( + { + "p", + "div", + "section", + "article", + "header", + "footer", + "main", + "aside", + "nav", + "figure", + "figcaption", + "details", + "summary", + "dl", + "dt", + "dd", + } +) +_HEADING_TAGS = frozenset({"h1", "h2", "h3", "h4", "h5", "h6"}) +_INLINE_EMPHASIS = {"strong": "**", "b": "**", "em": "*", "i": "*"} + + +class _MarkdownRenderer(HTMLParser): + """HTMLParser subclass that emits Markdown tokens into a list.""" + + def __init__(self): + super().__init__(convert_charrefs = False) + self._out: list[str] = [] + self._skip_depth: int = 0 + + # Link state + self._link_href: str | None = None + self._link_text_parts: list[str] = [] + self._in_link: bool = False + + # List state + self._list_stack: list[str] = [] # "ul" or "ol" + self._ol_counter: list[int] = [] + + # Table state + self._in_table: bool = False + self._current_row: list[str] = [] + self._cell_parts: list[str] = [] + self._in_cell: bool = False + self._header_row_done: bool = False + self._row_has_th: bool = False + self._is_first_row: bool = False + + # Pre/code state + self._in_pre: bool = False + self._pre_parts: list[str] = [] + self._in_inline_code: bool = False + + # Blockquote state -- stack of output buffers so nested + # blockquotes each collect their own content and get prefixed + # with the correct number of ">" markers on close. + self._bq_stack: list[list[str]] = [] + + # ------------------------------------------------------------------ + def _emit(self, text: str) -> None: + if self._in_link: + self._link_text_parts.append(text) + elif self._in_cell: + self._cell_parts.append(text) + elif self._in_pre: + self._pre_parts.append(text) + elif self._bq_stack: + self._bq_stack[-1].append(text) + else: + self._out.append(text) + + # ------------------------------------------------------------------ + def _prefix_blockquote(self, content: str) -> str: + """Prefix every line of *content* with ``> ``.""" + # Strip trailing whitespace first, then collapse blank lines + content = re.sub(r"[ \t]+$", "", content, flags = re.MULTILINE) + content = re.sub(r"\n{3,}", "\n\n", content).strip() + if not content: + return "" + lines = content.split("\n") + prefixed: list[str] = [] + for line in lines: + if line.strip(): + prefixed.append("> " + line) + else: + prefixed.append(">") + return "\n".join(prefixed) + + # ------------------------------------------------------------------ + # Table helpers -- flush open cells and rows so that HTML with + # omitted optional end tags (, ) does not lose data. + # ------------------------------------------------------------------ + def _finish_cell(self) -> None: + if not self._in_cell: + return + self._in_cell = False + cell_text = "".join(self._cell_parts).strip().replace("\n", " ") + cell_text = cell_text.replace("|", "\\|") + self._current_row.append(cell_text) + self._cell_parts = [] + + def _finish_row(self) -> None: + if not self._current_row: + return + line = "| " + " | ".join(self._current_row) + " |" + self._emit(line + "\n") + if not self._header_row_done and (self._row_has_th or self._is_first_row): + sep = "| " + " | ".join("---" for _ in self._current_row) + " |" + self._emit(sep + "\n") + self._header_row_done = True + self._is_first_row = False + self._current_row = [] + self._row_has_th = False + + # ------------------------------------------------------------------ + # Link text helper -- normalize whitespace so block-level content + # inside an does not produce multiline Markdown link labels. + # ------------------------------------------------------------------ + def _finish_link(self) -> None: + text = re.sub(r"\s+", " ", "".join(self._link_text_parts)).strip() + href = self._link_href or "" + self._in_link = False + if href and text: + self._emit(f"[{text}]({href})") + elif text: + self._emit(text) + + # ------------------------------------------------------------------ + # Tag handlers + # ------------------------------------------------------------------ + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + tag = tag.lower() + + if tag in _SKIP_TAGS: + self._skip_depth += 1 + return + if self._skip_depth: + return + + attr_dict = dict(attrs) + + if tag in _HEADING_TAGS: + level = int(tag[1]) + self._emit("\n\n" + "#" * level + " ") + + elif tag == "a": + self._link_href = attr_dict.get("href") + self._link_text_parts = [] + self._in_link = True + + elif tag in _INLINE_EMPHASIS: + self._emit(_INLINE_EMPHASIS[tag]) + + elif tag == "br": + self._emit("\n") + + elif tag in _BLOCK_TAGS: + self._emit("\n\n") + + elif tag == "hr": + self._emit("\n\n---\n\n") + + elif tag == "blockquote": + self._emit("\n\n") + self._bq_stack.append([]) + + elif tag == "ul": + self._list_stack.append("ul") + self._emit("\n") + + elif tag == "ol": + self._list_stack.append("ol") + start_attr = attr_dict.get("start") + try: + start = int(start_attr) if start_attr is not None else 1 + except (ValueError, TypeError): + start = 1 + self._ol_counter.append(start - 1) + self._emit("\n") + + elif tag == "li": + indent = " " * max(0, len(self._list_stack) - 1) + if self._list_stack and self._list_stack[-1] == "ol": + if self._ol_counter: + self._ol_counter[-1] += 1 + self._emit(f"\n{indent}{self._ol_counter[-1]}. ") + else: + self._emit(f"\n{indent}1. ") + else: + self._emit(f"\n{indent}* ") + + elif tag == "pre": + self._pre_parts = [] + self._in_pre = True + + elif tag == "code" and not self._in_pre: + self._in_inline_code = True + self._emit("`") + + elif tag == "table": + self._in_table = True + self._header_row_done = False + self._is_first_row = True + self._emit("\n\n") + + elif tag == "tr": + # Flush any open cell/row from a previous row that may + # have omitted its optional or end tags. + self._finish_cell() + self._finish_row() + + elif tag in ("th", "td"): + # Flush any open cell (handles omitted /
spans
+ if self._in_inline_code:
+ self._emit(data)
+ return
+ # Collapse all whitespace (including newlines) per HTML rules
+ text = re.sub(r"\s+", " ", data)
+ # Suppress whitespace-only text nodes between table structural
+ # elements (indentation from source HTML) to prevent leading
+ # spaces from breaking Markdown table row alignment.
+ if self._in_table and not self._in_cell and not text.strip():
+ return
+ self._emit(text)
+
+ def handle_entityref(self, name: str) -> None:
+ if self._skip_depth:
+ return
+ self._emit(html.unescape(f"&{name};"))
+
+ def handle_charref(self, name: str) -> None:
+ if self._skip_depth:
+ return
+ self._emit(html.unescape(f"{name};"))
+
+ # ------------------------------------------------------------------
+ # Flush pending buffers (handles truncated HTML from capped fetches)
+ # ------------------------------------------------------------------
+ def flush_pending(self) -> None:
+ """Flush any open side-buffers into ``_out``.
+
+ Called after ``close()`` to recover content from truncated HTML
+ where closing tags were never seen (common when ``_fetch_page_text``
+ caps the download by byte count).
+ """
+ # Flush innermost buffers first so their content propagates outward.
+
+ if self._in_link:
+ self._finish_link()
+
+ if self._in_inline_code:
+ self._in_inline_code = False
+ self._emit("`")
+
+ self._finish_cell()
+ self._finish_row()
+
+ if self._in_pre:
+ raw = "".join(self._pre_parts)
+ self._in_pre = False
+ block = "```\n" + raw + "\n```"
+ self._emit("\n\n" + block + "\n\n")
+
+ # Flatten any open blockquote buffers (innermost first)
+ while self._bq_stack:
+ content = "".join(self._bq_stack.pop())
+ prefixed = self._prefix_blockquote(content)
+ if not prefixed:
+ continue
+ if self._bq_stack:
+ self._bq_stack[-1].append("\n\n" + prefixed + "\n\n")
+ else:
+ self._out.append("\n\n" + prefixed + "\n\n")
+
+
+# ------------------------------------------------------------------
+# Post-processing
+# ------------------------------------------------------------------
+def _cleanup(text: str) -> str:
+ """Normalize whitespace and blank lines in the final output.
+
+ Preserves content inside fenced code blocks verbatim so that
+ intentional blank lines in ```` content are not collapsed.
+ """
+ lines = text.split("\n")
+ out: list[str] = []
+ in_fence = False
+ blank_run = 0
+
+ for line in lines:
+ stripped = line.rstrip(" \t")
+ if stripped.startswith("```"):
+ in_fence = not in_fence
+ blank_run = 0
+ out.append(stripped)
+ continue
+
+ if in_fence:
+ # Preserve code block content exactly as-is
+ out.append(line)
+ continue
+
+ if not stripped:
+ blank_run += 1
+ if blank_run <= 1:
+ out.append("")
+ continue
+
+ blank_run = 0
+ out.append(stripped)
+
+ return "\n".join(out).strip()
+
+
+# ------------------------------------------------------------------
+# Public API
+# ------------------------------------------------------------------
+def html_to_markdown(source_html: str) -> str:
+ """Convert an HTML string to Markdown.
+
+ Handles headings, links, bold/italic, lists (ordered and unordered),
+ tables, blockquotes, code blocks, and HTML entities. ``]*>",
- "",
- raw_html,
- flags = _re.DOTALL | _re.IGNORECASE,
- )
- text = _re.sub(
- r"]*>", "", text, flags = _re.DOTALL | _re.IGNORECASE
- )
- text = _re.sub(r"<[^>]+>", " ", text)
- text = _re.sub(r"\s+", " ", text).strip()
+ text = html_to_markdown(raw_html)
if not text:
return "(page returned no readable text)"
diff --git a/studio/backend/tests/tool_calling_benchmark_results.md b/studio/backend/tests/tool_calling_benchmark_results.md
deleted file mode 100644
index c2b0687895..0000000000
--- a/studio/backend/tests/tool_calling_benchmark_results.md
+++ /dev/null
@@ -1,62 +0,0 @@
-# GGUF Tool Calling Benchmark Results
-
-Prompt: "List and categorize all the songs that charted #3 on the Billboard Hot 100 in 2015."
-10 runs per configuration, web search + code execution + thinking enabled.
-GPU: NVIDIA B200, CUDA_VISIBLE_DEVICES=2.
-
-Ground truth: 4 songs peaked at #3 in 2015 -- "Love Me like You Do" (Ellie Goulding), "Earned It" (The Weeknd), "Watch Me" (Silento), "Drag Me Down" (One Direction).
-
-## Cartesian Grid: Model x Quant x KV Cache
-
-| Model | Quant | KV Cache | OK/10 | Avg Time | Avg Tools | XML Leaks | URL Fetch | Peak3 Avg | All 4/4 | Best Songs |
-|-------|-------|----------|-------|----------|-----------|-----------|-----------|-----------|---------|------------|
-| 4B | UD-Q4_K_XL | f16 | 10/10 | 9.8s | 3.5 | 0/10 | 4/10 | 0.8/4 | 2/10 | 9 |
-| 4B | UD-Q4_K_XL | bf16 | 10/10 | 10.6s | 4.5 | 0/10 | 4/10 | 0.4/4 | 1/10 | 5 |
-| 4B | Q8_0 | f16 | 10/10 | 4.9s | 2.4 | 0/10 | 8/10 | 0.4/4 | 1/10 | 5 |
-| 4B | Q8_0 | bf16 | 10/10 | 8.0s | 3.0 | 0/10 | 5/10 | 0.0/4 | 0/10 | 0 |
-| 9B | UD-Q4_K_XL | f16 | 10/10 | 6.7s | 2.0 | 0/10 | 5/10 | 0.0/4 | 0/10 | 3 |
-| 9B | UD-Q4_K_XL | bf16 | 9/10 | 49.5s | 2.4 | 0/10 | 5/10 | 0.0/4 | 0/10 | 1 |
-| 9B | Q8_0 | f16 | 10/10 | 7.4s | 2.5 | 0/10 | 5/10 | 0.0/4 | 0/10 | 2 |
-| 9B | Q8_0 | bf16 | 10/10 | 10.4s | 2.7 | 0/10 | 6/10 | 1.0/4 | 2/10 | 15 |
-| **27B** | **UD-Q4_K_XL** | **bf16** | **9/10** | **131.1s** | **13.8** | **0/10** | **7/10** | **2.7/4** | **6/10** | **27** |
-| 27B | UD-Q4_K_XL | f16 | 7/10 | 201.6s | 14.1 | 0/10 | 8/10 | 2.0/4 | 5/10 | 26 |
-| 27B | Q8_0 | f16 | 4/10 | 312.5s | 16.0 | 1/10 | 10/10 | 2.4/4 | 6/10 | 28 |
-| 27B | Q8_0 | bf16 | 5/10 | 258.4s | 16.5 | 2/10 | 10/10 | 0.9/4 | 1/10 | 27 |
-| 35B-A3B | UD-Q4_K_XL | f16 | 3/10 | 353.6s | 14.7 | 1/10 | 6/10 | 1.2/4 | 3/10 | 27 |
-| 35B-A3B | UD-Q4_K_XL | bf16 | 3/10 | 356.2s | 17.2 | 1/10 | 8/10 | 1.6/4 | 4/10 | 27 |
-| 35B-A3B | Q8_0 | f16 | 2/10 | 372.1s | 17.6 | 1/10 | 7/10 | 1.2/4 | 3/10 | 26 |
-| 35B-A3B | Q8_0 | bf16 | 6/10 | 267.7s | 17.5 | 1/10 | 8/10 | 2.4/4 | 6/10 | 27 |
-
-**Column definitions:**
-- **Peak3 Avg**: Average number of correct peak-#3 songs found per run (out of 4)
-- **All 4/4**: Runs where all 4 correct songs were identified
-- **Best Songs**: Maximum number of Billboard 2015 songs mentioned in any single run (out of 31 tracked)
-- **URL Fetch**: Runs where the model used web_search with `url` parameter to fetch full page content
-
-## Key Findings
-
-1. **27B UD-Q4_K_XL + bf16 KV is the sweet spot.** 6/10 runs found all 4 correct songs, 0 XML leaks, 131s average. Best balance of accuracy, speed, and reliability.
-
-2. **Larger models use tools more effectively.** 27B and 35B-A3B models used 13-17 tool calls per query (vs 2-4 for 4B/9B), performing multiple searches and URL fetches to find the answer.
-
-3. **27B Q8_0 had the highest raw accuracy (6/10 all-4/4) but lower reliability** -- only 4/10 OK runs due to timeouts on long agentic chains. The UD-Q4_K_XL quant is more practical.
-
-4. **4B models were fastest (5-10s) but least accurate.** They occasionally found all 4 songs (2/10 best case) when they happened to fetch the right Wikipedia page.
-
-5. **9B was surprisingly weaker than 4B on this task.** It used fewer tool calls and rarely extracted song data from fetched pages. The 9B model may need higher temperature or different prompting for this specific task type.
-
-6. **35B-A3B had reliability issues.** Most runs timed out or errored due to slow per-token generation with many tool iterations. When it completed (2-6/10 OK), accuracy was comparable to 27B.
-
-7. **bf16 KV cache had mixed effects.** For 27B it improved both speed (131s vs 202s) and accuracy (6/10 vs 5/10 all-4/4). For smaller models it had no consistent benefit.
-
-8. **XML leaks are nearly eliminated.** 0/10 for all 4B and 9B configs, and only 1-2/10 for the largest models (which generate much more text in complex agentic loops).
-
-## Before vs After (4B UD-Q4_K_XL, f16 KV)
-
-| Metric | Before Changes | After Changes |
-|--------|---------------|---------------|
-| XML leaks | 10/10 | 0/10 |
-| URL fetches | 0/10 | 4/10 |
-| Peak3 accuracy | 0.0/4 | 0.8/4 |
-| Runs with all 4 songs | 0/10 | 2/10 |
-| Avg time | 12.3s | 9.8s |