Merge pull request #536 from ggozad/fix/reuse-docling-converter

Reuse the local docling converter across documents
This commit is contained in:
Yiorgis Gozadinos 2026-08-13 12:43:40 +03:00 committed by GitHub
commit 10fd89f547
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 221 additions and 12 deletions

View file

@ -6,6 +6,7 @@
- `cross-encoder` reranking no longer ties the scores of strongly-relevant candidates, which left their order to the sort. Scores remain 0-1. - `cross-encoder` reranking no longer ties the scores of strongly-relevant candidates, which left their order to the sort. Scores remain 0-1.
- `haiku-rag` and `haiku-ingester` CLI startup no longer imports `lancedb`, `pyarrow` and `pydantic_ai`. - `haiku-rag` and `haiku-ingester` CLI startup no longer imports `lancedb`, `pyarrow` and `pydantic_ai`.
- `haiku.rag.store` no longer re-exports `Store`; import it from `haiku.rag.store.engine`. - `haiku.rag.store` no longer re-exports `Store`; import it from `haiku.rag.store.engine`.
- `docling-local` reuses one docling `DocumentConverter` per set of conversion options instead of building one per document, so local layout, table and OCR models are no longer loaded per document. Conversions through a shared converter are serialized.
## [0.73.0] - 2026-08-06 ## [0.73.0] - 2026-08-06

View file

@ -1,6 +1,10 @@
"""Local docling converter implementation.""" """Local docling converter implementation."""
import asyncio import asyncio
import hashlib
import threading
from collections.abc import Iterator
from contextlib import contextmanager
from pathlib import Path from pathlib import Path
from typing import TYPE_CHECKING, ClassVar from typing import TYPE_CHECKING, ClassVar
@ -10,11 +14,24 @@ from haiku.rag.converters.text_utils import TextFileHandler, docling_safe_name
if TYPE_CHECKING: if TYPE_CHECKING:
from docling.datamodel.base_models import InputFormat from docling.datamodel.base_models import InputFormat
from docling.datamodel.pipeline_options import PdfPipelineOptions
from docling.document_converter import DocumentConverter as DoclingDocConverter
from docling.document_converter import FormatOption from docling.document_converter import FormatOption
from docling_core.types.doc.document import DoclingDocument from docling_core.types.doc.document import DoclingDocument
from haiku.rag.config.models import ConversionOptions from haiku.rag.config.models import ConversionOptions
# Docling builds its layout, table and OCR models per DocumentConverter and
# caches pipelines per instance, so a converter per document reloads every model
# per document. StandardPdfPipeline also keeps per-run state on the instance, so
# the lock spans the conversion, not just the lookup.
_CONVERTER_LOCK = threading.Lock()
_CONVERTERS: dict[str, "DoclingDocConverter"] = {}
# HTML and Markdown backend options carry the per-document source_uri. Both run
# SimplePipeline, which loads no models, so they get a converter per call.
_URI_AWARE_EXTENSIONS = frozenset({".html", ".xhtml", ".md", ".qmd", ".rmd"})
class DoclingLocalConverter(DocumentConverter): class DoclingLocalConverter(DocumentConverter):
"""Converter that uses local docling for document conversion. """Converter that uses local docling for document conversion.
@ -142,7 +159,9 @@ class DoclingLocalConverter(DocumentConverter):
return pipeline_options return pipeline_options
def _build_format_options( def _build_format_options(
self, source_uri: str | None = None self,
source_uri: str | None = None,
pipeline_options: "PdfPipelineOptions | None" = None,
) -> "dict[InputFormat, FormatOption]": ) -> "dict[InputFormat, FormatOption]":
"""Per-format options shared between file and text conversion paths. """Per-format options shared between file and text conversion paths.
@ -155,6 +174,8 @@ class DoclingLocalConverter(DocumentConverter):
source_uri: Origin URI used by the HTML and Markdown backends to source_uri: Origin URI used by the HTML and Markdown backends to
resolve relative `<img src="/path">` references (e.g. when resolve relative `<img src="/path">` references (e.g. when
ingesting a downloaded HTML page). ingesting a downloaded HTML page).
pipeline_options: Wired into every format option; built from
configuration when omitted.
""" """
from docling.backend.docling_parse_backend import DoclingParseDocumentBackend from docling.backend.docling_parse_backend import DoclingParseDocumentBackend
from docling.datamodel.backend_options import ( from docling.datamodel.backend_options import (
@ -173,7 +194,8 @@ class DoclingLocalConverter(DocumentConverter):
from pydantic import AnyUrl from pydantic import AnyUrl
opts = self.config.processing.conversion_options opts = self.config.processing.conversion_options
pipeline_options = self._build_pipeline_options() if pipeline_options is None:
pipeline_options = self._build_pipeline_options()
fetch = opts.fetch_remote_images fetch = opts.fetch_remote_images
source_url = AnyUrl(source_uri) if source_uri else None source_url = AnyUrl(source_uri) if source_uri else None
@ -203,19 +225,51 @@ class DoclingLocalConverter(DocumentConverter):
InputFormat.PPTX: PowerpointFormatOption(pipeline_options=pipeline_options), InputFormat.PPTX: PowerpointFormatOption(pipeline_options=pipeline_options),
} }
def _sync_convert_docling_file( @contextmanager
self, path: Path, source_uri: str | None = None def _shared_converter(self) -> Iterator["DoclingDocConverter"]:
) -> "DoclingDocument": """Yield the converter shared by every conversion with these pipeline
"""Synchronous conversion of docling-supported files.""" options, holding the lock for the caller's conversion.
`serialize_as_any` is required for the key: without it pydantic
serializes the nested option models as their declared type, rendering
them as `{}` and hiding `table_mode` and the OCR engine.
"""
from docling.document_converter import ( from docling.document_converter import (
DocumentConverter as DoclingDocConverter, DocumentConverter as DoclingDocConverter,
) )
converter = DoclingDocConverter( pipeline_options = self._build_pipeline_options()
format_options=self._build_format_options(source_uri=source_uri) key = hashlib.md5(
) pipeline_options.model_dump_json(serialize_as_any=True).encode("utf-8"),
result = converter.convert(path) usedforsecurity=False,
return result.document ).hexdigest()
with _CONVERTER_LOCK:
converter = _CONVERTERS.get(key)
if converter is None:
converter = _CONVERTERS[key] = DoclingDocConverter(
format_options=self._build_format_options(
pipeline_options=pipeline_options
)
)
yield converter
def _sync_convert_docling_file(
self, path: Path, source_uri: str | None = None
) -> "DoclingDocument":
"""Synchronous conversion of docling-supported files."""
if path.suffix.lower() in _URI_AWARE_EXTENSIONS:
from docling.document_converter import (
DocumentConverter as DoclingDocConverter,
)
converter = DoclingDocConverter(
format_options=self._build_format_options(source_uri=source_uri)
)
return converter.convert(path).document
with self._shared_converter() as converter:
return converter.convert(path).document
async def convert_file( async def convert_file(
self, path: Path, source_uri: str | None = None self, path: Path, source_uri: str | None = None

View file

@ -1,7 +1,11 @@
"""Tests for document converters.""" """Tests for document converters."""
import asyncio
import tempfile import tempfile
import threading
import time
from pathlib import Path from pathlib import Path
from types import SimpleNamespace
from unittest.mock import AsyncMock, Mock, patch from unittest.mock import AsyncMock, Mock, patch
import httpx import httpx
@ -10,7 +14,7 @@ from docling_core.types.doc.document import DoclingDocument
from haiku.rag.config import AppConfig from haiku.rag.config import AppConfig
from haiku.rag.config.models import ModelConfig from haiku.rag.config.models import ModelConfig
from haiku.rag.converters import get_converter from haiku.rag.converters import docling_local, get_converter
from haiku.rag.converters.base import vlm_api_url from haiku.rag.converters.base import vlm_api_url
from haiku.rag.converters.docling_local import DoclingLocalConverter from haiku.rag.converters.docling_local import DoclingLocalConverter
from haiku.rag.converters.docling_serve import DoclingServeConverter from haiku.rag.converters.docling_serve import DoclingServeConverter
@ -1007,6 +1011,156 @@ class TestDoclingLocalConverter:
) )
class TestSharedDoclingConverter:
"""Reuse of the docling converter across documents. CSV inputs resolve to
SimplePipeline, so no models load here.
"""
@pytest.fixture
def config(self):
return AppConfig()
@pytest.fixture(autouse=True)
def clear_cache(self):
"""A cached converter is an instance of the patched subclass below, so
it must not outlive the test that cached it."""
docling_local._CONVERTERS.clear()
yield
docling_local._CONVERTERS.clear()
@pytest.fixture
def docling_calls(self, monkeypatch):
"""Count converter constructions and overlapping conversions."""
from docling.document_converter import DocumentConverter
record = SimpleNamespace(constructions=0, format_options=[], overlaps=0)
depth = 0
depth_lock = threading.Lock()
class RecordingConverter(DocumentConverter):
def __init__(self, *args, format_options=None, **kwargs):
super().__init__(*args, format_options=format_options, **kwargs)
record.constructions += 1
record.format_options.append(format_options)
def convert(self, *args, **kwargs):
nonlocal depth
with depth_lock:
depth += 1
if depth > 1:
record.overlaps += 1
try:
time.sleep(0.05)
return super().convert(*args, **kwargs)
finally:
with depth_lock:
depth -= 1
monkeypatch.setattr(
"docling.document_converter.DocumentConverter", RecordingConverter
)
return record
@pytest.fixture
def csv_file(self, tmp_path):
source = tmp_path / "rows.csv"
source.write_text("a,b\n1,2\n")
return source
@pytest.mark.asyncio
async def test_converter_is_reused_across_documents(
self, config, csv_file, docling_calls
):
"""A converter per ingested document must not mean a docling converter
per document, which is what reloads the models."""
for _ in range(2):
doc = await DoclingLocalConverter(config).convert_file(
csv_file, source_uri=csv_file.as_uri()
)
assert isinstance(doc, DoclingDocument)
assert docling_calls.constructions == 1
@pytest.mark.asyncio
async def test_concurrent_conversions_do_not_overlap(
self, config, csv_file, docling_calls
):
"""StandardPdfPipeline keeps per-run state on the instance, so two
conversions may never share one converter at the same time."""
await asyncio.gather(
*(
DoclingLocalConverter(config).convert_file(
csv_file, source_uri=csv_file.as_uri()
)
for _ in range(2)
)
)
assert docling_calls.constructions == 1
assert docling_calls.overlaps == 0
@pytest.mark.asyncio
async def test_markup_conversion_keeps_its_own_source_uri(
self, config, tmp_path, docling_calls
):
"""HTML and Markdown backends resolve relative image references from
`source_uri`, so each conversion gets a converter carrying its own."""
from docling.datamodel.base_models import InputFormat
for name in ("one.md", "two.md"):
source = tmp_path / name
source.write_text("# heading\n\ntext\n")
await DoclingLocalConverter(config).convert_file(
source, source_uri=source.as_uri()
)
assert docling_calls.constructions == 2
uris = [
str(options[InputFormat.MD].backend_options.source_uri)
for options in docling_calls.format_options
]
assert uris[0].endswith("one.md")
assert uris[1].endswith("two.md")
@pytest.mark.asyncio
@pytest.mark.parametrize("mode", ["fast", "accurate"])
async def test_differing_pipeline_options_get_their_own_converter(
self, config, csv_file, docling_calls, mode
):
"""`table_mode` lives in a nested option model, which pydantic renders
as `{}` unless serialized with `serialize_as_any`."""
config.processing.conversion_options.table_mode = mode
other = config.model_copy(deep=True)
other.processing.conversion_options.table_mode = (
"accurate" if mode == "fast" else "fast"
)
for cfg in (config, other):
await DoclingLocalConverter(cfg).convert_file(
csv_file, source_uri=csv_file.as_uri()
)
assert docling_calls.constructions == 2
@pytest.mark.asyncio
async def test_config_differences_outside_the_pipeline_still_reuse(
self, config, csv_file, docling_calls, tmp_path
):
"""Settings the pipeline never sees must not strand a second set of
models."""
other = config.model_copy(deep=True)
other.storage.data_dir = tmp_path / "elsewhere"
other.qa.model.name = "some-other-model"
other.qa.max_searches = config.qa.max_searches + 3
for cfg in (config, other):
await DoclingLocalConverter(cfg).convert_file(
csv_file, source_uri=csv_file.as_uri()
)
assert docling_calls.constructions == 1
class TestDoclingServeConverter: class TestDoclingServeConverter:
"""Tests for DoclingServeConverter (mocked).""" """Tests for DoclingServeConverter (mocked)."""