Pass OCR options from conversion_options to docling-serve chunker
This commit is contained in:
parent
b6431322cf
commit
b22ac33dde
4 changed files with 55 additions and 2 deletions
|
|
@ -1,6 +1,12 @@
|
||||||
# Changelog
|
# Changelog
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **docling-serve Chunker OCR Options**: The docling-serve chunker now respects OCR settings from `conversion_options`
|
||||||
|
- Passes `do_ocr`, `force_ocr`, `ocr_engine`, and `ocr_lang` to the chunking API
|
||||||
|
- Allows disabling OCR via config when running docling-serve in read-only containers
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
- **CI**: Cache HuggingFace tokenizer to prevent flaky test failures when HuggingFace has transient outages
|
- **CI**: Cache HuggingFace tokenizer to prevent flaky test failures when HuggingFace has transient outages
|
||||||
|
|
|
||||||
|
|
@ -230,6 +230,8 @@ providers:
|
||||||
|
|
||||||
Conversion options work identically for both local and remote processing.
|
Conversion options work identically for both local and remote processing.
|
||||||
|
|
||||||
|
**Note:** When using `chunker: docling-serve`, OCR options (`do_ocr`, `force_ocr`, `ocr_engine`, `ocr_lang`) from `conversion_options` are passed to the chunking API. This is useful when running docling-serve in a read-only container where OCR model downloads fail—set `do_ocr: false` to disable OCR entirely.
|
||||||
|
|
||||||
### Chunking Strategies
|
### Chunking Strategies
|
||||||
|
|
||||||
**Hybrid chunking** (default):
|
**Hybrid chunking** (default):
|
||||||
|
|
|
||||||
|
|
@ -61,9 +61,13 @@ class DoclingServeChunker(DocumentChunker):
|
||||||
)
|
)
|
||||||
self.chunker_type = config.processing.chunker_type
|
self.chunker_type = config.processing.chunker_type
|
||||||
|
|
||||||
def _build_chunking_data(self) -> dict[str, str]:
|
def _build_chunking_data(self) -> dict[str, str | list[str]]:
|
||||||
"""Build form data for chunking request."""
|
"""Build form data for chunking request."""
|
||||||
return {
|
opts = self.config.processing.conversion_options
|
||||||
|
data: dict[str, str | list[str]] = {
|
||||||
|
"convert_do_ocr": str(opts.do_ocr).lower(),
|
||||||
|
"convert_force_ocr": str(opts.force_ocr).lower(),
|
||||||
|
"convert_ocr_engine": opts.ocr_engine,
|
||||||
"chunking_max_tokens": str(self.config.processing.chunk_size),
|
"chunking_max_tokens": str(self.config.processing.chunk_size),
|
||||||
"chunking_tokenizer": self.config.processing.chunking_tokenizer,
|
"chunking_tokenizer": self.config.processing.chunking_tokenizer,
|
||||||
"chunking_merge_peers": str(
|
"chunking_merge_peers": str(
|
||||||
|
|
@ -73,6 +77,9 @@ class DoclingServeChunker(DocumentChunker):
|
||||||
self.config.processing.chunking_use_markdown_tables
|
self.config.processing.chunking_use_markdown_tables
|
||||||
).lower(),
|
).lower(),
|
||||||
}
|
}
|
||||||
|
if opts.ocr_lang:
|
||||||
|
data["convert_ocr_lang"] = opts.ocr_lang
|
||||||
|
return data
|
||||||
|
|
||||||
async def _call_chunk_api(self, document: "DoclingDocument") -> list[dict]:
|
async def _call_chunk_api(self, document: "DoclingDocument") -> list[dict]:
|
||||||
"""Call docling-serve chunking API and return raw chunk data.
|
"""Call docling-serve chunking API and return raw chunk data.
|
||||||
|
|
|
||||||
|
|
@ -351,6 +351,10 @@ class TestDoclingServeChunker:
|
||||||
config.processing.chunk_size = 512
|
config.processing.chunk_size = 512
|
||||||
config.processing.chunking_merge_peers = False
|
config.processing.chunking_merge_peers = False
|
||||||
config.processing.chunking_use_markdown_tables = True
|
config.processing.chunking_use_markdown_tables = True
|
||||||
|
config.processing.conversion_options.do_ocr = False
|
||||||
|
config.processing.conversion_options.force_ocr = True
|
||||||
|
config.processing.conversion_options.ocr_engine = "tesseract"
|
||||||
|
config.processing.conversion_options.ocr_lang = ["en", "de"]
|
||||||
chunker = DoclingServeChunker(config)
|
chunker = DoclingServeChunker(config)
|
||||||
|
|
||||||
result_data = {"chunks": [{"text": "Chunk 1", "chunk_index": 0}]}
|
result_data = {"chunks": [{"text": "Chunk 1", "chunk_index": 0}]}
|
||||||
|
|
@ -370,6 +374,40 @@ class TestDoclingServeChunker:
|
||||||
assert data["chunking_max_tokens"] == "512"
|
assert data["chunking_max_tokens"] == "512"
|
||||||
assert data["chunking_merge_peers"] == "false"
|
assert data["chunking_merge_peers"] == "false"
|
||||||
assert data["chunking_use_markdown_tables"] == "true"
|
assert data["chunking_use_markdown_tables"] == "true"
|
||||||
|
# OCR options from conversion_options
|
||||||
|
assert data["convert_do_ocr"] == "false"
|
||||||
|
assert data["convert_force_ocr"] == "true"
|
||||||
|
assert data["convert_ocr_engine"] == "tesseract"
|
||||||
|
assert data["convert_ocr_lang"] == ["en", "de"]
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
@patch("haiku.rag.providers.docling_serve.httpx.AsyncClient")
|
||||||
|
async def test_chunk_omits_empty_ocr_lang(self, mock_client_class, config):
|
||||||
|
"""Test that ocr_lang is omitted when empty (default)."""
|
||||||
|
# Ensure ocr_lang is empty (default)
|
||||||
|
config.processing.conversion_options.ocr_lang = []
|
||||||
|
chunker = DoclingServeChunker(config)
|
||||||
|
|
||||||
|
result_data = {"chunks": [{"text": "Chunk 1", "chunk_index": 0}]}
|
||||||
|
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(result_data)
|
||||||
|
|
||||||
|
mock_client = AsyncMock()
|
||||||
|
mock_client.post = AsyncMock(return_value=submit_resp)
|
||||||
|
mock_client.get = AsyncMock(side_effect=[poll_resp, result_resp])
|
||||||
|
mock_client_class.return_value.__aenter__.return_value = mock_client
|
||||||
|
|
||||||
|
converter = get_converter(Config)
|
||||||
|
doc = await converter.convert_text("# Test", name="test.md")
|
||||||
|
await chunker.chunk(doc)
|
||||||
|
|
||||||
|
call_kwargs = mock_client.post.call_args.kwargs
|
||||||
|
data = call_kwargs["data"]
|
||||||
|
# OCR options should use defaults
|
||||||
|
assert data["convert_do_ocr"] == "true"
|
||||||
|
assert data["convert_force_ocr"] == "false"
|
||||||
|
assert data["convert_ocr_engine"] == "auto"
|
||||||
|
# ocr_lang should NOT be present when empty
|
||||||
|
assert "convert_ocr_lang" not in data
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
@patch("haiku.rag.providers.docling_serve.httpx.AsyncClient")
|
@patch("haiku.rag.providers.docling_serve.httpx.AsyncClient")
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue