unsloth/tests/python/test_rag_parsers.py
Roland Tannous c4b5889e53 Studio: layout-aware RAG parsers + heading-aware chunking (Phase 3A)
Replace bare-pypdf/python-docx/BeautifulSoup extraction with Markdown-
preserving parsers so the chunker can split on real heading boundaries
instead of running paragraphs together.

Parsers
- pdf.py:  pymupdf + pymupdf4llm.to_markdown() per page; pypdf kept as
           fallback when pymupdf can't open the file.
- docx.py: mammoth.convert_to_html() + markdownify, with an explicit
           style_map so Title/Heading 1..6 become h1..h6 in the output.
- html.py: BeautifulSoup pre-scrub (drop script/style) then markdownify
           so <h*>, <table>, <ul> convert faithfully.
- text.py: signature update only; TXT/MD pass through unchanged.
- parsers/__init__.py: new ParsedImage + ParseResult dataclass; parse()
  signature is now parse(path, *, want_images=False) -> ParseResult.
  ParseResult is iterable over .pages for backward compat.

Chunker
- chunking.py: prepend Markdown heading separators ("\n# " .. "\n#### ")
  to the priority list so heading-aware splits happen for free once the
  parsers emit Markdown.

Ingestion
- ingestion.py: single call site updated to consume ParseResult.pages.

Deps (no-torch-runtime.txt)
+ pymupdf>=1.24, pymupdf4llm>=0.0.17, mammoth>=1.7, markdownify>=0.13
- pypdf kept as a fallback path.

Tests
- test_rag_parsers.py asserts Markdown headings survive PDF/DOCX/HTML
  extraction; also exercises ParseResult iteration backward-compat.
- test_rag_chunking.py: new case verifying chunks start at Markdown
  heading boundaries when the input is Markdown.

Foundation for Phase 3B-late (heading-aware spans for late chunking)
and Phase 3B-multimodal (want_images=True enables image extraction in
the same parser layer). No schema or opt-in flags in this commit.
2026-05-24 11:10:13 +04:00

135 lines
4 KiB
Python

"""Document parser tests — each format skipped if its lib is unavailable.
Phase 3A: parsers now return ParseResult (iterable over .pages) and
emit Markdown so the chunker can split on heading boundaries.
"""
import sys
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
STUDIO_BACKEND = REPO_ROOT / "studio" / "backend"
if str(STUDIO_BACKEND) not in sys.path:
sys.path.insert(0, str(STUDIO_BACKEND))
def test_text_parser_utf8(tmp_path):
from core.rag.parsers import parse
file = tmp_path / "sample.txt"
file.write_text("hello world\n\nsecond paragraph", encoding = "utf-8")
result = parse(file)
assert len(result) == 1
assert "hello world" in result.pages[0].text
assert "second paragraph" in result.pages[0].text
assert result.images == []
def test_markdown_parser_preserves_headings(tmp_path):
from core.rag.parsers import parse
file = tmp_path / "sample.md"
file.write_text("# Title\n\nBody text with **emphasis**.", encoding = "utf-8")
result = parse(file)
assert result.pages
# Markdown should pass through unchanged — heading marker preserved.
assert "# Title" in result.pages[0].text
def test_unsupported_format_raises(tmp_path):
from core.rag.parsers import UnsupportedFormatError, parse
file = tmp_path / "weird.xyz"
file.write_text("nope")
with pytest.raises(UnsupportedFormatError):
parse(file)
def test_html_parser_emits_markdown_headings(tmp_path):
pytest.importorskip("bs4")
pytest.importorskip("lxml")
pytest.importorskip("markdownify")
from core.rag.parsers import parse
file = tmp_path / "sample.html"
file.write_text(
"<html><body>"
"<script>alert(1)</script>"
"<h1>Main Title</h1>"
"<h2>Sub Section</h2>"
"<p>visible text</p>"
"<ul><li>one</li><li>two</li></ul>"
"</body></html>",
encoding = "utf-8",
)
result = parse(file)
assert result.pages
md = result.pages[0].text
# markdownify converts <h1> → '# ', <h2> → '## '
assert "# Main Title" in md
assert "## Sub Section" in md
assert "visible text" in md
# script content scrubbed
assert "alert" not in md
# list items become Markdown bullets
assert "one" in md and "two" in md
def test_pdf_parser_extracts_pages(tmp_path):
pytest.importorskip("pymupdf")
pytest.importorskip("pymupdf4llm")
from core.rag.parsers import parse
pypdf = pytest.importorskip("pypdf")
from pypdf import PdfWriter
file = tmp_path / "tiny.pdf"
writer = PdfWriter()
writer.add_blank_page(width = 72, height = 72)
with open(file, "wb") as f:
writer.write(f)
# Blank page yields no extractable text — should return empty pages
# without error.
result = parse(file)
assert isinstance(result.pages, list)
assert isinstance(result.images, list)
def test_docx_parser_emits_markdown_headings(tmp_path):
pytest.importorskip("docx")
pytest.importorskip("mammoth")
pytest.importorskip("markdownify")
from docx import Document
file = tmp_path / "sample.docx"
doc = Document()
doc.add_heading("Top Level Heading", level = 1)
doc.add_paragraph("First paragraph here.")
doc.add_heading("Sub Heading", level = 2)
doc.add_paragraph("Second paragraph here.")
doc.save(str(file))
from core.rag.parsers import parse
result = parse(file)
assert result.pages
md = result.pages[0].text
# mammoth via _STYLE_MAP maps Heading 1/2 → h1/h2 → '# '/'## '.
assert "# Top Level Heading" in md
assert "## Sub Heading" in md
assert "First paragraph" in md
assert "Second paragraph" in md
def test_parse_result_is_iterable_for_backcompat(tmp_path):
"""Code that does `for page in parse(path)` should keep working."""
from core.rag.parsers import parse
file = tmp_path / "sample.txt"
file.write_text("hello", encoding = "utf-8")
result = parse(file)
pages = list(result)
assert len(pages) == 1
assert pages[0].text == "hello"