haiku.rag/haiku_rag_slim/haiku/rag/client/processing.py
Yiorgis Gozadinos d1691d3942
Share document preparation and HTTP acquisition
Five call sites repeated the same post-conversion preparation: store the
Docling representation and resolve a title when none was supplied.
_prepare_and_title now owns that sequence. update_document continues to
call _prepare_document_from_docling directly because an explicit update
must preserve an existing empty title.

create_document, both content-replacement branches of update_document,
and source ingestion embedded eagerly before passing chunks to a
persistence funnel that checked them again. The funnels now own
embedding, including the checks required by import_document and
import_documents for caller-supplied chunks.

Move the document.embed span into ensure_chunks_embedded after its early
return. Every path that performs embedding is now instrumented, while
operations whose chunks are already embedded emit no span.

convert() previously used its own HTTP client and temporary-file path.
Route URL conversion through HTTPSource, matching source ingestion, and
move _write_fetch_body to processing.py so both paths share temporary
file handling without an import cycle.

Add walk_files for filesystem enumeration and use it from both
FSSource.discover and one-shot directory ingestion. Symlink escape
filtering now has one implementation.
2026-08-20 10:50:33 +03:00

398 lines
14 KiB
Python

import asyncio
import io
import logging
import tempfile
from pathlib import Path
from typing import TYPE_CHECKING
from urllib.parse import urlparse
import logfire
from haiku.rag.client.exceptions import UnsupportedSourceError
from haiku.rag.config import AppConfig
from haiku.rag.converters import get_converter
from haiku.rag.store.models.chunk import Chunk
from haiku.rag.store.models.document_item import _picture_description_text
if TYPE_CHECKING:
from docling_core.types.doc.document import DoclingDocument, PictureItem
from haiku.rag.embeddings import EmbedderWrapper
logger = logging.getLogger(__name__)
def _warn_if_descriptions_missing(
config: AppConfig, doc: "DoclingDocument", source: str
) -> None:
"""Warn when picture-description was requested but produced nothing.
docling-serve swallows VLM errors (network failures, missing models,
etc.) and returns a successful conversion with empty descriptions.
docling-local can do the same when the VLM endpoint is unreachable.
Surface the silent failure: when ``processing.pictures="description"``
AND the document has at least one picture AND zero descriptions came
back, log a clear warning so the user can fix their VLM config before
a thousand-document ingest produces an empty corpus.
"""
if config.processing.pictures != "description":
return
if not doc.pictures:
return
described = sum(1 for p in doc.pictures if _picture_description_text(p))
if described == 0:
model = config.processing.conversion_options.picture_description.model
logger.warning(
"processing.pictures='description' but no descriptions came back "
"for %s (%d pictures, 0 described). The VLM call likely failed "
"silently inside the converter. Check that the VLM at %s is "
"reachable from the converter and that the model name '%s' "
"resolves on the server.",
source,
len(doc.pictures),
model.base_url or "<provider default>",
model.name,
)
def _write_fetch_body_sync(body: bytes, suffix: str) -> Path:
with tempfile.NamedTemporaryFile(
mode="wb", suffix=suffix, delete=False
) as temp_file:
temp_file.write(body)
temp_file.flush()
return Path(temp_file.name)
async def _write_fetch_body(body: bytes, suffix: str) -> Path:
return await asyncio.to_thread(_write_fetch_body_sync, body, suffix)
async def convert(
config: AppConfig,
source: Path | str,
*,
format: str = "md",
source_uri: str | None = None,
) -> "DoclingDocument":
"""Convert a file, URL, or text to DoclingDocument.
Args:
config: Application configuration.
source: One of:
- Path: Local file path to convert
- str (URL): HTTP/HTTPS URL to download and convert
- str (text): Raw text content to convert
format: The format of text content ("md", "html", or "plain").
Defaults to "md". Use "plain" for plain text without parsing.
Only used when source is raw text (not a file path or URL).
Files and URLs determine format from extension/content-type.
source_uri: Origin URI used by docling's HTML/Markdown backends to
resolve relative `<img src="/path">` references. When omitted,
defaults to the URL (URL ingest) or `file://` URI (file ingest);
raw text input has no origin so no default is derived.
Returns:
DoclingDocument from the converted source.
Raises:
ValueError: If the file doesn't exist or has unsupported extension.
httpx.RequestError: If URL download fails.
"""
converter = get_converter(config)
async def _convert_file(
file_path: Path, effective_uri: str | None
) -> "DoclingDocument":
"""Dispatch through split-and-merge for large PDFs when configured,
otherwise call the converter directly."""
if file_path.suffix.lower() == ".pdf" and config.processing.split_pages > 0:
from haiku.rag.converters.pdf_split import convert_pdf_with_splitting
return await convert_pdf_with_splitting(
converter, file_path, effective_uri, config.processing.split_pages
)
return await converter.convert_file(file_path, source_uri=effective_uri)
# Path object - convert file directly
if isinstance(source, Path):
if not source.exists():
raise UnsupportedSourceError(f"File does not exist: {source}")
if source.suffix.lower() not in converter.supported_extensions:
raise UnsupportedSourceError(f"Unsupported file extension: {source.suffix}")
effective_uri = source_uri or source.absolute().as_uri()
doc = await _convert_file(source, effective_uri)
_warn_if_descriptions_missing(config, doc, str(source))
return doc
# String - check if URL or text
parsed = urlparse(source)
if parsed.scheme in ("http", "https"):
# One HTTP acquisition path: the same adapter the ingester fetches with.
from haiku.rag.ingester.sources.http import HTTPSource
fetcher = HTTPSource(source_id="convert")
try:
result = await fetcher.fetch(source)
finally:
await fetcher.aclose()
file_extension = get_extension_from_content_type_or_url(
source, result.content_type
)
if file_extension not in converter.supported_extensions:
raise UnsupportedSourceError(
f"Unsupported content type/extension: "
f"{result.content_type}/{file_extension}"
)
temp_path = await _write_fetch_body(result.body, file_extension)
try:
doc = await _convert_file(temp_path, source_uri or source)
_warn_if_descriptions_missing(config, doc, source)
return doc
finally:
temp_path.unlink(missing_ok=True)
elif parsed.scheme == "file":
# file:// URI
file_path = Path(parsed.path)
if not file_path.exists():
raise UnsupportedSourceError(f"File does not exist: {file_path}")
if file_path.suffix.lower() not in converter.supported_extensions:
raise UnsupportedSourceError(
f"Unsupported file extension: {file_path.suffix}"
)
effective_uri = source_uri or file_path.absolute().as_uri()
doc = await _convert_file(file_path, effective_uri)
_warn_if_descriptions_missing(config, doc, str(file_path))
return doc
else:
# Raw text content — HTML and markdown can still embed pictures
# via <img>/![](...) so the same description check applies.
doc = await converter.convert_text(source, format=format, source_uri=source_uri)
_warn_if_descriptions_missing(config, doc, "<text input>")
return doc
def _merge_picture_chunks(
docling_document: "DoclingDocument",
text_chunks: list[Chunk],
document_id: str | None,
existing_picture_data: dict[str, bytes] | None,
min_picture_size: int,
) -> list[Chunk]:
picture_chunks = build_picture_chunks(
docling_document,
document_id=document_id,
existing_picture_data=existing_picture_data,
min_picture_size=min_picture_size,
)
if not picture_chunks:
for i, c in enumerate(text_chunks):
c.order = i
return text_chunks
positions = {
item.self_ref: pos
for pos, (item, _level) in enumerate(docling_document.iterate_items())
}
def first_pos(c: Chunk) -> int:
refs = (c.metadata or {}).get("doc_item_refs") or []
return positions.get(refs[0], len(positions)) if refs else len(positions)
merged = sorted(text_chunks + picture_chunks, key=first_pos)
for i, c in enumerate(merged):
c.order = i
return merged
async def chunk(
config: AppConfig,
docling_document: "DoclingDocument",
*,
embedder: "EmbedderWrapper",
existing_picture_data: dict[str, bytes] | None = None,
document_id: str | None = None,
) -> list[Chunk]:
"""Chunk a DoclingDocument into Chunks.
When the configured embedder supports images, also emit one synthetic
Chunk per ``PictureItem`` with available bytes (see ``build_picture_chunks``)
and merge them with text chunks in structural (``iterate_items()``) order.
``chunk.order`` is the index in the merged list.
``existing_picture_data`` (snapshot keyed by ``self_ref``) supplies bytes
for pictures whose ``image.uri`` has been stripped — used by the rebuild
path where the docling is loaded from the stored blob.
"""
from haiku.rag.chunkers import get_chunker
chunker = get_chunker(config)
text_chunks = await chunker.chunk(docling_document)
if not embedder.supports_images:
for i, c in enumerate(text_chunks):
c.order = i
return text_chunks
return await asyncio.to_thread(
_merge_picture_chunks,
docling_document,
text_chunks,
document_id,
existing_picture_data,
config.processing.min_picture_size,
)
def _min_picture_side(picture: "PictureItem", data: bytes) -> float | None:
"""The picture's smaller pixel dimension, or None when it can't be
determined. Uses ``ImageRef.size`` when the live image is present; falls
back to a PIL header read of the bytes (rebuild path, where picture URIs
have been stripped)."""
if picture.image is not None:
return min(picture.image.size.width, picture.image.size.height)
from PIL import Image as PILImage
from PIL import UnidentifiedImageError
try:
with PILImage.open(io.BytesIO(data)) as img:
return min(img.size)
except UnidentifiedImageError:
return None
def build_picture_chunks(
docling_document: "DoclingDocument",
*,
document_id: str | None = None,
existing_picture_data: dict[str, bytes] | None = None,
min_picture_size: int = 0,
) -> list[Chunk]:
"""Emit one synthetic ``Chunk`` per distinct ``PictureItem`` with available
bytes.
Bytes come from ``picture.image.uri`` (live data URI on a freshly-converted
docling) or from ``existing_picture_data`` keyed by ``self_ref`` (snapshot
taken before a delete-and-re-extract cycle, when the live docling has had
its picture URIs stripped). Pictures with no available bytes are skipped.
Pictures whose bytes were already seen in this document are skipped — the
first occurrence carries the chunk, so a watermark repeated on every page
embeds once. Pictures whose smaller side is under ``min_picture_size``
pixels are skipped entirely (``0`` disables the size filter; pictures
whose size can't be determined are kept).
The bytes ride on ``Chunk._picture_data`` (a PrivateAttr — not serialized)
so ``embed_chunks`` can route them through ``embed_image``. The
``order`` field is left at its default (0); the caller (``chunk()``)
reassigns it after merging with text chunks in structural order.
"""
from haiku.rag.store.models.document_item import (
_decode_picture_bytes,
extract_item_text,
)
existing = existing_picture_data or {}
seen: set[bytes] = set()
chunks: list[Chunk] = []
for picture in docling_document.pictures:
picture_data = _decode_picture_bytes(picture)
if picture_data is None:
picture_data = existing.get(picture.self_ref)
if picture_data is None:
continue
if picture_data in seen:
continue
seen.add(picture_data)
if min_picture_size > 0:
side = _min_picture_side(picture, picture_data)
if side is not None and side < min_picture_size:
continue
text = extract_item_text(picture, docling_document) or ""
page_numbers: list[int] = []
for p in picture.prov:
if p.page_no not in page_numbers:
page_numbers.append(p.page_no)
metadata = {
"doc_item_refs": [picture.self_ref],
"labels": ["picture"],
"page_numbers": sorted(page_numbers),
"headings": None,
}
chunk = Chunk(
document_id=document_id,
content=text,
metadata=metadata,
)
chunk._picture_data = picture_data
chunks.append(chunk)
return chunks
async def ensure_chunks_embedded(
config: AppConfig, chunks: list[Chunk], embedder: "EmbedderWrapper"
) -> list[Chunk]:
"""Ensure all chunks have embeddings, embedding any that don't.
Chunks that already have embeddings are passed through unchanged; missing
embeddings are filled in in-place in the returned list (preserving order).
"""
from haiku.rag.embeddings import embed_chunks
chunks_to_embed = [c for c in chunks if c.embedding is None]
if not chunks_to_embed:
return chunks
with logfire.span("document.embed", chunks=len(chunks_to_embed)):
embedded = await embed_chunks(chunks_to_embed, embedder, config)
# embed_chunks preserves input order; fill positionally, since duplicate
# chunk texts across documents make a content-keyed lookup ambiguous.
filled = iter(embedded)
return [ch if ch.embedding is not None else next(filled) for ch in chunks]
def get_extension_from_content_type_or_url(url: str, content_type: str) -> str:
"""Determine file extension from HTTP Content-Type header or URL suffix.
Returns the mapped extension for known content types, falling back to the
URL path suffix, and finally `.html` for generic web content.
"""
content_type_map = {
"text/html": ".html",
"text/plain": ".txt",
"text/markdown": ".md",
"application/pdf": ".pdf",
"application/json": ".json",
"text/csv": ".csv",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document": ".docx",
"application/vnd.openxmlformats-officedocument.presentationml.presentation": ".pptx",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": ".xlsx",
}
for ct, ext in content_type_map.items():
if ct in content_type:
return ext
parsed_url = urlparse(url)
path = Path(parsed_url.path)
if path.suffix:
return path.suffix.lower()
return ".html"