swap doclaynet.pdf to the full paper, add real-PDF split/merge test

This commit is contained in:
Yiorgis Gozadinos 2026-05-25 12:09:30 +03:00
parent 6bcc2f6357
commit 9ed24ad53e
No known key found for this signature in database
5 changed files with 111 additions and 28 deletions

View file

@ -130,3 +130,30 @@ def vcr_config():
"filter_headers": ["authorization", "x-api-key"],
"decode_compressed_response": True,
}
@pytest.fixture(scope="session")
def doclaynet_first_page_pdf(tmp_path_factory) -> Path:
"""One-page extract of ``tests/data/doclaynet.pdf`` (the full DocLayNet
arXiv paper). Most existing tests only need a small PDF with at least
one picture; this avoids running docling over all nine pages of the
paper just to assert ``pictures != []``. The full paper is used
directly by the split-and-merge integration test."""
import pypdfium2 as pdfium
src_path = Path(__file__).parent / "data" / "doclaynet.pdf"
out_dir = tmp_path_factory.mktemp("doclaynet")
out_path = out_dir / "page0.pdf"
src = pdfium.PdfDocument(str(src_path))
try:
dst = pdfium.PdfDocument.new()
try:
dst.import_pages(src, [0])
with open(out_path, "wb") as f:
dst.save(f)
finally:
dst.close()
finally:
src.close()
return out_path

Binary file not shown.

View file

@ -610,22 +610,20 @@ This is content.
@pytest.mark.vcr()
@pytest.mark.asyncio
async def test_local_and_serve_chunkers_produce_same_output():
async def test_local_and_serve_chunkers_produce_same_output(doclaynet_first_page_pdf):
"""Test that local and serve chunkers produce identical output for the same document.
Note: Labels are resolved from the DoclingDocument since docling-serve API
only returns ref strings, not labels. See:
https://github.com/docling-project/docling-serve/issues/448
"""
from pathlib import Path
from haiku.rag.chunkers.docling_local import DoclingLocalChunker
from haiku.rag.chunkers.docling_serve import DoclingServeChunker
from haiku.rag.converters.docling_serve import DoclingServeConverter
# Use docling-serve to convert the PDF (ensures same conversion for both chunkers)
converter = DoclingServeConverter(Config)
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
doc = await converter.convert_file(pdf_path)
# Create both chunkers with same config
@ -681,7 +679,7 @@ async def test_local_and_serve_chunkers_produce_same_output():
@pytest.mark.vcr()
@pytest.mark.asyncio
async def test_serve_chunker_accepts_picture_laden_docling():
async def test_serve_chunker_accepts_picture_laden_docling(doclaynet_first_page_pdf):
"""Round-trip a picture-bearing PDF through docling-serve's chunker.
Catches the schema-shape failure we hit in production: when the
@ -691,8 +689,6 @@ async def test_serve_chunker_accepts_picture_laden_docling():
the docling-serve container's docling-core and the local one would
surface as ``Input document document.json is not valid``.
"""
from pathlib import Path
from haiku.rag.chunkers.docling_serve import DoclingServeChunker
from haiku.rag.converters.docling_serve import DoclingServeConverter
@ -702,7 +698,7 @@ async def test_serve_chunker_accepts_picture_laden_docling():
config.processing.chunker_type = "hybrid"
converter = DoclingServeConverter(config)
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
doc = await converter.convert_file(pdf_path)
# Sanity: at least one picture has bytes inlined as a data URI — that's

View file

@ -1127,13 +1127,13 @@ This is paragraph four about topic C.
@pytest.mark.vcr()
async def test_client_visualize_chunk_with_pdf(temp_db_path):
async def test_client_visualize_chunk_with_pdf(temp_db_path, doclaynet_first_page_pdf):
"""Test visualize_chunk returns images with bounding boxes for PDF documents."""
from PIL.Image import Image as PILImage
from haiku.rag.config import AppConfig
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
config = AppConfig()
config.processing.conversion_options.do_ocr = False

View file

@ -627,10 +627,12 @@ class TestDoclingLocalConverter:
)
@pytest.mark.asyncio
async def test_convert_pdf_with_picture_images(self, config):
async def test_convert_pdf_with_picture_images(
self, config, doclaynet_first_page_pdf
):
"""Picture bytes are produced by the local converter for PDFs that
contain figures."""
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
converter = DoclingLocalConverter(config)
doc = await converter.convert_file(pdf_path)
@ -643,9 +645,59 @@ class TestDoclingLocalConverter:
)
@pytest.mark.asyncio
async def test_convert_pdf_without_page_images(self, config):
"""Test PDF conversion excludes page images when disabled."""
async def test_split_and_merge_matches_single_pass(self, config):
"""Real-PDF integration test for split_pages: convert the full
9-page DocLayNet arXiv paper single-pass, then again via
``convert_pdf_with_splitting`` with slice_size=1 (one slice per
page), and assert the merged result is equivalent to single-pass
on totals + per-page-number coverage + self_ref uniqueness +
markdown export.
Slow runs docling-local 10 times against a multi-page PDF. The
contract this pins is the highest-risk one: that splitting at the
byte level and merging via DoclingDocument.concatenate produces a
document semantically indistinguishable from a single-pass convert.
"""
from haiku.rag.converters.pdf_split import convert_pdf_with_splitting
pdf_path = Path("tests/data/doclaynet.pdf")
config.processing.conversion_options.do_ocr = False
converter = DoclingLocalConverter(config)
baseline = await converter.convert_file(pdf_path)
merged = await convert_pdf_with_splitting(
converter, pdf_path, source_uri=None, slice_size=1
)
# Same totals across every list the consumer cares about.
assert len(merged.texts) == len(baseline.texts)
assert len(merged.pictures) == len(baseline.pictures)
assert len(merged.tables) == len(baseline.tables)
assert sorted(merged.pages.keys()) == sorted(baseline.pages.keys())
# Page numbers cover the same range — this is the key thing
# concatenate handles via its internal page_delta.
def _page_nos(doc):
return {p.page_no for t in doc.texts for p in t.prov}
assert _page_nos(merged) == _page_nos(baseline)
# self_refs unique across the merged doc — concatenate re-indexes
# them per-slice, so a duplicate here is a real merger bug.
merged_refs = [t.self_ref for t in merged.texts]
assert len(set(merged_refs)) == len(merged_refs)
# Strongest assertion: rendered markdown matches byte-for-byte.
# If this fails, the split/merge introduced ordering or content
# drift the count-based asserts above didn't catch.
assert merged.export_to_markdown() == baseline.export_to_markdown()
@pytest.mark.asyncio
async def test_convert_pdf_without_page_images(
self, config, doclaynet_first_page_pdf
):
"""Test PDF conversion excludes page images when disabled."""
pdf_path = doclaynet_first_page_pdf
config.processing.conversion_options.generate_page_images = False
converter = DoclingLocalConverter(config)
@ -659,9 +711,9 @@ class TestDoclingLocalConverter:
)
@pytest.mark.asyncio
async def test_convert_pdf_with_page_images(self, config):
async def test_convert_pdf_with_page_images(self, config, doclaynet_first_page_pdf):
"""Test PDF conversion includes page images when enabled."""
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
config.processing.conversion_options.generate_page_images = True
converter = DoclingLocalConverter(config)
@ -805,9 +857,11 @@ class TestDoclingLocalConverter:
@pytest.mark.asyncio
@pytest.mark.vcr()
async def test_picture_description_end_to_end(self, config):
async def test_picture_description_end_to_end(
self, config, doclaynet_first_page_pdf
):
"""End-to-end test: convert PDF with VLM picture descriptions."""
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
# Disable OCR (not needed for native PDF, avoids model downloads)
config.processing.conversion_options.do_ocr = False
@ -1321,12 +1375,14 @@ class TestDoclingServeConverterIntegration:
@pytest.mark.asyncio
@pytest.mark.integration
async def test_picture_description_end_to_end(self, config):
async def test_picture_description_end_to_end(
self, config, doclaynet_first_page_pdf
):
"""End-to-end test: convert PDF with VLM picture descriptions via docling-serve.
Note: Not using VCR because this test involves polling with changing task IDs.
"""
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
config.processing.pictures = "description"
config.processing.conversion_options.picture_description.model.provider = (
"ollama"
@ -1359,9 +1415,11 @@ class TestDoclingServeConverterIntegration:
@pytest.mark.vcr()
@pytest.mark.asyncio
async def test_convert_pdf_without_page_images(self, config):
async def test_convert_pdf_without_page_images(
self, config, doclaynet_first_page_pdf
):
"""Test PDF conversion excludes page images when disabled."""
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
config.processing.conversion_options.generate_page_images = False
converter = DoclingServeConverter(config)
@ -1376,9 +1434,9 @@ class TestDoclingServeConverterIntegration:
@pytest.mark.vcr()
@pytest.mark.asyncio
async def test_convert_pdf_with_page_images(self, config):
async def test_convert_pdf_with_page_images(self, config, doclaynet_first_page_pdf):
"""Test PDF conversion includes page images when enabled."""
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
config.processing.conversion_options.generate_page_images = True
converter = DoclingServeConverter(config)
@ -1393,9 +1451,9 @@ class TestDoclingServeConverterIntegration:
@pytest.mark.vcr()
@pytest.mark.asyncio
async def test_convert_pdf_with_ocr_engine(self, config):
async def test_convert_pdf_with_ocr_engine(self, config, doclaynet_first_page_pdf):
"""Test PDF conversion with explicit OCR engine selection."""
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
config.processing.conversion_options.ocr_engine = "easyocr"
converter = DoclingServeConverter(config)
@ -1406,7 +1464,9 @@ class TestDoclingServeConverterIntegration:
@pytest.mark.vcr()
@pytest.mark.asyncio
async def test_convert_pdf_with_picture_images(self, config):
async def test_convert_pdf_with_picture_images(
self, config, doclaynet_first_page_pdf
):
"""Picture bytes are produced for PDFs that contain figures.
docling-serve only emits picture image bytes via the
@ -1416,7 +1476,7 @@ class TestDoclingServeConverterIntegration:
into ``data:`` URIs so the result is shape-equivalent to the local
converter.
"""
pdf_path = Path("tests/data/doclaynet.pdf")
pdf_path = doclaynet_first_page_pdf
converter = DoclingServeConverter(config)
doc = await converter.convert_file(pdf_path)