Cache the chunking tokenizer to avoid HF Hub 429
This commit is contained in:
parent
d512f9fbd1
commit
0f4c4af495
3 changed files with 34 additions and 2 deletions
|
|
@ -19,6 +19,7 @@
|
||||||
- **HTML text ingest path picks up converter options.** `convert_text(format="html"/"md")` previously used a bare `DoclingDocConverter()` with zero format options — the wix corpus ingest path. It now uses the same shared `_build_format_options()` helper as the file path.
|
- **HTML text ingest path picks up converter options.** `convert_text(format="html"/"md")` previously used a bare `DoclingDocConverter()` with zero format options — the wix corpus ingest path. It now uses the same shared `_build_format_options()` helper as the file path.
|
||||||
- **Relative `<img>` paths resolve during URL ingest.** `HaikuRAG.convert()` and the converter `convert_file` / `convert_text` methods now thread a `source_uri` through to `HTMLBackendOptions.source_uri` / `MarkdownBackendOptions.source_uri`. URL ingest uses the originating URL; file ingest uses `file://`; raw text accepts an optional override. docling-serve accepts the kwarg as a no-op (its API has no equivalent option).
|
- **Relative `<img>` paths resolve during URL ingest.** `HaikuRAG.convert()` and the converter `convert_file` / `convert_text` methods now thread a `source_uri` through to `HTMLBackendOptions.source_uri` / `MarkdownBackendOptions.source_uri`. URL ingest uses the originating URL; file ingest uses `file://`; raw text accepts an optional override. docling-serve accepts the kwarg as a no-op (its API has no equivalent option).
|
||||||
- **CLI tracebacks no longer dump per-frame locals.** The Typer app now passes `pretty_exceptions_show_locals=False`, so exceptions involving a `DoclingDocument` (or any large object) print readable rich tracebacks instead of pages of inline base64 image URIs. Set `_TYPER_STANDARD_TRACEBACK=1` for plain Python tracebacks.
|
- **CLI tracebacks no longer dump per-frame locals.** The Typer app now passes `pretty_exceptions_show_locals=False`, so exceptions involving a `DoclingDocument` (or any large object) print readable rich tracebacks instead of pages of inline base64 image URIs. Set `_TYPER_STANDARD_TRACEBACK=1` for plain Python tracebacks.
|
||||||
|
- **Batch ingest no longer hits HF Hub's 429 rate limit.** The chunking tokenizer is now loaded once per process via `@functools.cache` instead of once per chunker instance.
|
||||||
|
|
||||||
### Documentation
|
### Documentation
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,3 +1,4 @@
|
||||||
|
from functools import cache
|
||||||
from typing import TYPE_CHECKING, cast
|
from typing import TYPE_CHECKING, cast
|
||||||
|
|
||||||
from haiku.rag.chunkers.base import DocumentChunker
|
from haiku.rag.chunkers.base import DocumentChunker
|
||||||
|
|
@ -9,6 +10,16 @@ if TYPE_CHECKING:
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
|
||||||
|
|
||||||
|
@cache
|
||||||
|
def _get_tokenizer(name: str):
|
||||||
|
# `AutoTokenizer.from_pretrained` triggers an HF Hub `model_info` request
|
||||||
|
# per call. Batch ingest builds one chunker per document, so without this
|
||||||
|
# cache HF rate-limits at 1000 requests / 5 minutes.
|
||||||
|
from transformers import AutoTokenizer
|
||||||
|
|
||||||
|
return AutoTokenizer.from_pretrained(name)
|
||||||
|
|
||||||
|
|
||||||
def _create_markdown_serializer_provider(use_markdown_tables: bool = True):
|
def _create_markdown_serializer_provider(use_markdown_tables: bool = True):
|
||||||
"""Create a markdown serializer provider with configurable table rendering.
|
"""Create a markdown serializer provider with configurable table rendering.
|
||||||
|
|
||||||
|
|
@ -64,7 +75,6 @@ class DoclingLocalChunker(DocumentChunker):
|
||||||
from docling_core.transforms.chunker.tokenizer.huggingface import (
|
from docling_core.transforms.chunker.tokenizer.huggingface import (
|
||||||
HuggingFaceTokenizer,
|
HuggingFaceTokenizer,
|
||||||
)
|
)
|
||||||
from transformers import AutoTokenizer
|
|
||||||
|
|
||||||
self.config = config
|
self.config = config
|
||||||
self.chunk_size = config.processing.chunk_size
|
self.chunk_size = config.processing.chunk_size
|
||||||
|
|
@ -72,7 +82,7 @@ class DoclingLocalChunker(DocumentChunker):
|
||||||
self.tokenizer_name = config.processing.chunking_tokenizer
|
self.tokenizer_name = config.processing.chunking_tokenizer
|
||||||
|
|
||||||
if self.chunker_type == "hybrid":
|
if self.chunker_type == "hybrid":
|
||||||
hf_tokenizer = AutoTokenizer.from_pretrained(self.tokenizer_name)
|
hf_tokenizer = _get_tokenizer(self.tokenizer_name)
|
||||||
tokenizer = HuggingFaceTokenizer(
|
tokenizer = HuggingFaceTokenizer(
|
||||||
tokenizer=hf_tokenizer, max_tokens=self.chunk_size
|
tokenizer=hf_tokenizer, max_tokens=self.chunk_size
|
||||||
)
|
)
|
||||||
|
|
|
||||||
|
|
@ -84,6 +84,27 @@ def test_get_chunker_invalid():
|
||||||
get_chunker(config)
|
get_chunker(config)
|
||||||
|
|
||||||
|
|
||||||
|
def test_tokenizer_cached_across_chunker_instances():
|
||||||
|
"""Repeated DoclingLocalChunker instantiations share one loaded tokenizer.
|
||||||
|
|
||||||
|
Each `AutoTokenizer.from_pretrained` call triggers an `HfApi.model_info`
|
||||||
|
HTTP request to check for revision drift. Batch ingest creates one
|
||||||
|
chunker per document, which without caching hits HF Hub's 1000-per-5min
|
||||||
|
limit and crashes with HTTP 429.
|
||||||
|
"""
|
||||||
|
from haiku.rag.chunkers.docling_local import _get_tokenizer
|
||||||
|
|
||||||
|
_get_tokenizer.cache_clear()
|
||||||
|
|
||||||
|
DoclingLocalChunker()
|
||||||
|
DoclingLocalChunker()
|
||||||
|
DoclingLocalChunker()
|
||||||
|
|
||||||
|
info = _get_tokenizer.cache_info()
|
||||||
|
assert info.misses == 1
|
||||||
|
assert info.hits == 2
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_local_chunker_hierarchical(qa_corpus: Dataset):
|
async def test_local_chunker_hierarchical(qa_corpus: Dataset):
|
||||||
"""Test DoclingLocalChunker with hierarchical chunking."""
|
"""Test DoclingLocalChunker with hierarchical chunking."""
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue