swap doclaynet.pdf to the full paper, add real-PDF split/merge test
This commit is contained in:
parent
6bcc2f6357
commit
9ed24ad53e
5 changed files with 111 additions and 28 deletions
|
|
@ -130,3 +130,30 @@ def vcr_config():
|
|||
"filter_headers": ["authorization", "x-api-key"],
|
||||
"decode_compressed_response": True,
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def doclaynet_first_page_pdf(tmp_path_factory) -> Path:
|
||||
"""One-page extract of ``tests/data/doclaynet.pdf`` (the full DocLayNet
|
||||
arXiv paper). Most existing tests only need a small PDF with at least
|
||||
one picture; this avoids running docling over all nine pages of the
|
||||
paper just to assert ``pictures != []``. The full paper is used
|
||||
directly by the split-and-merge integration test."""
|
||||
import pypdfium2 as pdfium
|
||||
|
||||
src_path = Path(__file__).parent / "data" / "doclaynet.pdf"
|
||||
out_dir = tmp_path_factory.mktemp("doclaynet")
|
||||
out_path = out_dir / "page0.pdf"
|
||||
|
||||
src = pdfium.PdfDocument(str(src_path))
|
||||
try:
|
||||
dst = pdfium.PdfDocument.new()
|
||||
try:
|
||||
dst.import_pages(src, [0])
|
||||
with open(out_path, "wb") as f:
|
||||
dst.save(f)
|
||||
finally:
|
||||
dst.close()
|
||||
finally:
|
||||
src.close()
|
||||
return out_path
|
||||
|
|
|
|||
Binary file not shown.
|
|
@ -610,22 +610,20 @@ This is content.
|
|||
|
||||
@pytest.mark.vcr()
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_and_serve_chunkers_produce_same_output():
|
||||
async def test_local_and_serve_chunkers_produce_same_output(doclaynet_first_page_pdf):
|
||||
"""Test that local and serve chunkers produce identical output for the same document.
|
||||
|
||||
Note: Labels are resolved from the DoclingDocument since docling-serve API
|
||||
only returns ref strings, not labels. See:
|
||||
https://github.com/docling-project/docling-serve/issues/448
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
from haiku.rag.chunkers.docling_local import DoclingLocalChunker
|
||||
from haiku.rag.chunkers.docling_serve import DoclingServeChunker
|
||||
from haiku.rag.converters.docling_serve import DoclingServeConverter
|
||||
|
||||
# Use docling-serve to convert the PDF (ensures same conversion for both chunkers)
|
||||
converter = DoclingServeConverter(Config)
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
doc = await converter.convert_file(pdf_path)
|
||||
|
||||
# Create both chunkers with same config
|
||||
|
|
@ -681,7 +679,7 @@ async def test_local_and_serve_chunkers_produce_same_output():
|
|||
|
||||
@pytest.mark.vcr()
|
||||
@pytest.mark.asyncio
|
||||
async def test_serve_chunker_accepts_picture_laden_docling():
|
||||
async def test_serve_chunker_accepts_picture_laden_docling(doclaynet_first_page_pdf):
|
||||
"""Round-trip a picture-bearing PDF through docling-serve's chunker.
|
||||
|
||||
Catches the schema-shape failure we hit in production: when the
|
||||
|
|
@ -691,8 +689,6 @@ async def test_serve_chunker_accepts_picture_laden_docling():
|
|||
the docling-serve container's docling-core and the local one would
|
||||
surface as ``Input document document.json is not valid``.
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
from haiku.rag.chunkers.docling_serve import DoclingServeChunker
|
||||
from haiku.rag.converters.docling_serve import DoclingServeConverter
|
||||
|
||||
|
|
@ -702,7 +698,7 @@ async def test_serve_chunker_accepts_picture_laden_docling():
|
|||
config.processing.chunker_type = "hybrid"
|
||||
|
||||
converter = DoclingServeConverter(config)
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
doc = await converter.convert_file(pdf_path)
|
||||
|
||||
# Sanity: at least one picture has bytes inlined as a data URI — that's
|
||||
|
|
|
|||
|
|
@ -1127,13 +1127,13 @@ This is paragraph four about topic C.
|
|||
|
||||
|
||||
@pytest.mark.vcr()
|
||||
async def test_client_visualize_chunk_with_pdf(temp_db_path):
|
||||
async def test_client_visualize_chunk_with_pdf(temp_db_path, doclaynet_first_page_pdf):
|
||||
"""Test visualize_chunk returns images with bounding boxes for PDF documents."""
|
||||
from PIL.Image import Image as PILImage
|
||||
|
||||
from haiku.rag.config import AppConfig
|
||||
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
config = AppConfig()
|
||||
config.processing.conversion_options.do_ocr = False
|
||||
|
||||
|
|
|
|||
|
|
@ -627,10 +627,12 @@ class TestDoclingLocalConverter:
|
|||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_convert_pdf_with_picture_images(self, config):
|
||||
async def test_convert_pdf_with_picture_images(
|
||||
self, config, doclaynet_first_page_pdf
|
||||
):
|
||||
"""Picture bytes are produced by the local converter for PDFs that
|
||||
contain figures."""
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
converter = DoclingLocalConverter(config)
|
||||
|
||||
doc = await converter.convert_file(pdf_path)
|
||||
|
|
@ -643,9 +645,59 @@ class TestDoclingLocalConverter:
|
|||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_convert_pdf_without_page_images(self, config):
|
||||
"""Test PDF conversion excludes page images when disabled."""
|
||||
async def test_split_and_merge_matches_single_pass(self, config):
|
||||
"""Real-PDF integration test for split_pages: convert the full
|
||||
9-page DocLayNet arXiv paper single-pass, then again via
|
||||
``convert_pdf_with_splitting`` with slice_size=1 (one slice per
|
||||
page), and assert the merged result is equivalent to single-pass
|
||||
on totals + per-page-number coverage + self_ref uniqueness +
|
||||
markdown export.
|
||||
|
||||
Slow — runs docling-local 10 times against a multi-page PDF. The
|
||||
contract this pins is the highest-risk one: that splitting at the
|
||||
byte level and merging via DoclingDocument.concatenate produces a
|
||||
document semantically indistinguishable from a single-pass convert.
|
||||
"""
|
||||
from haiku.rag.converters.pdf_split import convert_pdf_with_splitting
|
||||
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
config.processing.conversion_options.do_ocr = False
|
||||
converter = DoclingLocalConverter(config)
|
||||
|
||||
baseline = await converter.convert_file(pdf_path)
|
||||
merged = await convert_pdf_with_splitting(
|
||||
converter, pdf_path, source_uri=None, slice_size=1
|
||||
)
|
||||
|
||||
# Same totals across every list the consumer cares about.
|
||||
assert len(merged.texts) == len(baseline.texts)
|
||||
assert len(merged.pictures) == len(baseline.pictures)
|
||||
assert len(merged.tables) == len(baseline.tables)
|
||||
assert sorted(merged.pages.keys()) == sorted(baseline.pages.keys())
|
||||
|
||||
# Page numbers cover the same range — this is the key thing
|
||||
# concatenate handles via its internal page_delta.
|
||||
def _page_nos(doc):
|
||||
return {p.page_no for t in doc.texts for p in t.prov}
|
||||
|
||||
assert _page_nos(merged) == _page_nos(baseline)
|
||||
|
||||
# self_refs unique across the merged doc — concatenate re-indexes
|
||||
# them per-slice, so a duplicate here is a real merger bug.
|
||||
merged_refs = [t.self_ref for t in merged.texts]
|
||||
assert len(set(merged_refs)) == len(merged_refs)
|
||||
|
||||
# Strongest assertion: rendered markdown matches byte-for-byte.
|
||||
# If this fails, the split/merge introduced ordering or content
|
||||
# drift the count-based asserts above didn't catch.
|
||||
assert merged.export_to_markdown() == baseline.export_to_markdown()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_convert_pdf_without_page_images(
|
||||
self, config, doclaynet_first_page_pdf
|
||||
):
|
||||
"""Test PDF conversion excludes page images when disabled."""
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
config.processing.conversion_options.generate_page_images = False
|
||||
converter = DoclingLocalConverter(config)
|
||||
|
||||
|
|
@ -659,9 +711,9 @@ class TestDoclingLocalConverter:
|
|||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_convert_pdf_with_page_images(self, config):
|
||||
async def test_convert_pdf_with_page_images(self, config, doclaynet_first_page_pdf):
|
||||
"""Test PDF conversion includes page images when enabled."""
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
config.processing.conversion_options.generate_page_images = True
|
||||
converter = DoclingLocalConverter(config)
|
||||
|
||||
|
|
@ -805,9 +857,11 @@ class TestDoclingLocalConverter:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.vcr()
|
||||
async def test_picture_description_end_to_end(self, config):
|
||||
async def test_picture_description_end_to_end(
|
||||
self, config, doclaynet_first_page_pdf
|
||||
):
|
||||
"""End-to-end test: convert PDF with VLM picture descriptions."""
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
|
||||
# Disable OCR (not needed for native PDF, avoids model downloads)
|
||||
config.processing.conversion_options.do_ocr = False
|
||||
|
|
@ -1321,12 +1375,14 @@ class TestDoclingServeConverterIntegration:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.integration
|
||||
async def test_picture_description_end_to_end(self, config):
|
||||
async def test_picture_description_end_to_end(
|
||||
self, config, doclaynet_first_page_pdf
|
||||
):
|
||||
"""End-to-end test: convert PDF with VLM picture descriptions via docling-serve.
|
||||
|
||||
Note: Not using VCR because this test involves polling with changing task IDs.
|
||||
"""
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
config.processing.pictures = "description"
|
||||
config.processing.conversion_options.picture_description.model.provider = (
|
||||
"ollama"
|
||||
|
|
@ -1359,9 +1415,11 @@ class TestDoclingServeConverterIntegration:
|
|||
|
||||
@pytest.mark.vcr()
|
||||
@pytest.mark.asyncio
|
||||
async def test_convert_pdf_without_page_images(self, config):
|
||||
async def test_convert_pdf_without_page_images(
|
||||
self, config, doclaynet_first_page_pdf
|
||||
):
|
||||
"""Test PDF conversion excludes page images when disabled."""
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
config.processing.conversion_options.generate_page_images = False
|
||||
converter = DoclingServeConverter(config)
|
||||
|
||||
|
|
@ -1376,9 +1434,9 @@ class TestDoclingServeConverterIntegration:
|
|||
|
||||
@pytest.mark.vcr()
|
||||
@pytest.mark.asyncio
|
||||
async def test_convert_pdf_with_page_images(self, config):
|
||||
async def test_convert_pdf_with_page_images(self, config, doclaynet_first_page_pdf):
|
||||
"""Test PDF conversion includes page images when enabled."""
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
config.processing.conversion_options.generate_page_images = True
|
||||
converter = DoclingServeConverter(config)
|
||||
|
||||
|
|
@ -1393,9 +1451,9 @@ class TestDoclingServeConverterIntegration:
|
|||
|
||||
@pytest.mark.vcr()
|
||||
@pytest.mark.asyncio
|
||||
async def test_convert_pdf_with_ocr_engine(self, config):
|
||||
async def test_convert_pdf_with_ocr_engine(self, config, doclaynet_first_page_pdf):
|
||||
"""Test PDF conversion with explicit OCR engine selection."""
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
config.processing.conversion_options.ocr_engine = "easyocr"
|
||||
converter = DoclingServeConverter(config)
|
||||
|
||||
|
|
@ -1406,7 +1464,9 @@ class TestDoclingServeConverterIntegration:
|
|||
|
||||
@pytest.mark.vcr()
|
||||
@pytest.mark.asyncio
|
||||
async def test_convert_pdf_with_picture_images(self, config):
|
||||
async def test_convert_pdf_with_picture_images(
|
||||
self, config, doclaynet_first_page_pdf
|
||||
):
|
||||
"""Picture bytes are produced for PDFs that contain figures.
|
||||
|
||||
docling-serve only emits picture image bytes via the
|
||||
|
|
@ -1416,7 +1476,7 @@ class TestDoclingServeConverterIntegration:
|
|||
into ``data:`` URIs so the result is shape-equivalent to the local
|
||||
converter.
|
||||
"""
|
||||
pdf_path = Path("tests/data/doclaynet.pdf")
|
||||
pdf_path = doclaynet_first_page_pdf
|
||||
converter = DoclingServeConverter(config)
|
||||
|
||||
doc = await converter.convert_file(pdf_path)
|
||||
|
|
|
|||
Loading…
Reference in a new issue