Merge pull request #115 from ggozad/fix/docling-lazy-load

Lazy-load  docling-related functions, speed up client
This commit is contained in:
Yiorgis Gozadinos 2025-10-23 11:58:18 +03:00 committed by GitHub
commit 417a173318
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 45 additions and 14 deletions

View file

@ -9,7 +9,6 @@ from urllib.parse import urlparse
import httpx import httpx
from haiku.rag.config import Config from haiku.rag.config import Config
from haiku.rag.reader import FileReader
from haiku.rag.reranking import get_reranker from haiku.rag.reranking import get_reranker
from haiku.rag.store.engine import Store from haiku.rag.store.engine import Store
from haiku.rag.store.models.chunk import Chunk from haiku.rag.store.models.chunk import Chunk
@ -17,7 +16,6 @@ from haiku.rag.store.models.document import Document
from haiku.rag.store.repositories.chunk import ChunkRepository from haiku.rag.store.repositories.chunk import ChunkRepository
from haiku.rag.store.repositories.document import DocumentRepository from haiku.rag.store.repositories.document import DocumentRepository
from haiku.rag.store.repositories.settings import SettingsRepository from haiku.rag.store.repositories.settings import SettingsRepository
from haiku.rag.utils import text_to_docling_document
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@ -91,6 +89,9 @@ class HaikuRAG:
Returns: Returns:
The created Document instance. The created Document instance.
""" """
# Lazy import to avoid loading docling
from haiku.rag.utils import text_to_docling_document
# Convert content to DoclingDocument for processing # Convert content to DoclingDocument for processing
docling_document = text_to_docling_document(content) docling_document = text_to_docling_document(content)
@ -127,6 +128,8 @@ class HaikuRAG:
ValueError: If the file/URL cannot be parsed or doesn't exist ValueError: If the file/URL cannot be parsed or doesn't exist
httpx.RequestError: If URL request fails httpx.RequestError: If URL request fails
""" """
# Lazy import to avoid loading docling
from haiku.rag.reader import FileReader
# Normalize metadata # Normalize metadata
metadata = metadata or {} metadata = metadata or {}
@ -181,6 +184,9 @@ class HaikuRAG:
Raises: Raises:
ValueError: If the file cannot be parsed or doesn't exist ValueError: If the file cannot be parsed or doesn't exist
""" """
# Lazy import to avoid loading docling
from haiku.rag.reader import FileReader
metadata = metadata or {} metadata = metadata or {}
if source_path.suffix.lower() not in FileReader.extensions: if source_path.suffix.lower() not in FileReader.extensions:
@ -256,6 +262,9 @@ class HaikuRAG:
ValueError: If the content cannot be parsed ValueError: If the content cannot be parsed
httpx.RequestError: If URL request fails httpx.RequestError: If URL request fails
""" """
# Lazy import to avoid loading docling
from haiku.rag.reader import FileReader
metadata = metadata or {} metadata = metadata or {}
async with httpx.AsyncClient() as client: async with httpx.AsyncClient() as client:
@ -379,6 +388,9 @@ class HaikuRAG:
async def update_document(self, document: Document) -> Document: async def update_document(self, document: Document) -> Document:
"""Update an existing document.""" """Update an existing document."""
# Lazy import to avoid loading docling
from haiku.rag.utils import text_to_docling_document
# Convert content to DoclingDocument # Convert content to DoclingDocument
docling_document = text_to_docling_document(document.content) docling_document = text_to_docling_document(document.content)
@ -597,6 +609,9 @@ class HaikuRAG:
Yields: Yields:
int: The ID of the document currently being processed int: The ID of the document currently being processed
""" """
# Lazy import to avoid loading docling
from haiku.rag.utils import text_to_docling_document
await self.chunk_repository.delete_all() await self.chunk_repository.delete_all()
self.store.recreate_embeddings_table() self.store.recreate_embeddings_table()

View file

@ -1,21 +1,27 @@
import logging import logging
from pathlib import Path from pathlib import Path
from typing import TYPE_CHECKING
from watchfiles import Change, DefaultFilter, awatch from watchfiles import Change, DefaultFilter, awatch
from haiku.rag.client import HaikuRAG from haiku.rag.client import HaikuRAG
from haiku.rag.reader import FileReader
from haiku.rag.store.models.document import Document from haiku.rag.store.models.document import Document
if TYPE_CHECKING:
from haiku.rag.reader import FileReader
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
class FileFilter(DefaultFilter): class FileFilter(DefaultFilter):
def __init__(self, *, ignore_paths: list[Path] | None = None) -> None: def __init__(self, *, ignore_paths: list[Path] | None = None) -> None:
# Lazy import to avoid loading docling
from haiku.rag.reader import FileReader
self.extensions = tuple(FileReader.extensions) self.extensions = tuple(FileReader.extensions)
super().__init__(ignore_paths=ignore_paths) super().__init__(ignore_paths=ignore_paths)
def __call__(self, change: "Change", path: str) -> bool: def __call__(self, change: Change, path: str) -> bool:
return path.endswith(self.extensions) and super().__call__(change, path) return path.endswith(self.extensions) and super().__call__(change, path)
@ -50,11 +56,15 @@ class FileWatcher:
uri = file.as_uri() uri = file.as_uri()
existing_doc = await self.client.get_document_by_uri(uri) existing_doc = await self.client.get_document_by_uri(uri)
if existing_doc: if existing_doc:
doc = await self.client.create_document_from_source(str(file)) result = await self.client.create_document_from_source(str(file))
# Since we're passing a file (not directory), result should be a single Document
doc = result if isinstance(result, Document) else result[0]
logger.info(f"Updated document {existing_doc.id} from {file}") logger.info(f"Updated document {existing_doc.id} from {file}")
return doc return doc
else: else:
doc = await self.client.create_document_from_source(str(file)) result = await self.client.create_document_from_source(str(file))
# Since we're passing a file (not directory), result should be a single Document
doc = result if isinstance(result, Document) else result[0]
logger.info(f"Created new document {doc.id} from {file}") logger.info(f"Created new document {doc.id} from {file}")
return doc return doc
except Exception as e: except Exception as e:

View file

@ -1,17 +1,19 @@
import inspect import inspect
import json import json
import logging import logging
from typing import TYPE_CHECKING
from uuid import uuid4 from uuid import uuid4
from docling_core.types.doc.document import DoclingDocument
from lancedb.rerankers import RRFReranker from lancedb.rerankers import RRFReranker
from haiku.rag.chunker import chunker
from haiku.rag.config import Config from haiku.rag.config import Config
from haiku.rag.embeddings import get_embedder from haiku.rag.embeddings import get_embedder
from haiku.rag.store.engine import DocumentRecord, Store from haiku.rag.store.engine import DocumentRecord, Store
from haiku.rag.store.models.chunk import Chunk from haiku.rag.store.models.chunk import Chunk
from haiku.rag.utils import load_callable, text_to_docling_document from haiku.rag.utils import load_callable
if TYPE_CHECKING:
from docling_core.types.doc.document import DoclingDocument
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@ -142,9 +144,13 @@ class ChunkRepository:
return chunks return chunks
async def create_chunks_for_document( async def create_chunks_for_document(
self, document_id: str, document: DoclingDocument self, document_id: str, document: "DoclingDocument"
) -> list[Chunk]: ) -> list[Chunk]:
"""Create chunks and embeddings for a document from DoclingDocument.""" """Create chunks and embeddings for a document from DoclingDocument."""
# Lazy imports to avoid loading docling during module import
from haiku.rag.chunker import chunker
from haiku.rag.utils import text_to_docling_document
# Optionally preprocess markdown before chunking # Optionally preprocess markdown before chunking
processed_document = document processed_document = document
preprocessor_path = Config.MARKDOWN_PREPROCESSOR preprocessor_path = Config.MARKDOWN_PREPROCESSOR

View file

@ -4,12 +4,12 @@ from datetime import datetime
from typing import TYPE_CHECKING from typing import TYPE_CHECKING
from uuid import uuid4 from uuid import uuid4
from docling_core.types.doc.document import DoclingDocument
from haiku.rag.store.engine import DocumentRecord, Store from haiku.rag.store.engine import DocumentRecord, Store
from haiku.rag.store.models.document import Document from haiku.rag.store.models.document import Document
if TYPE_CHECKING: if TYPE_CHECKING:
from docling_core.types.doc.document import DoclingDocument
from haiku.rag.store.models.chunk import Chunk from haiku.rag.store.models.chunk import Chunk
@ -171,7 +171,7 @@ class DocumentRepository:
async def _create_with_docling( async def _create_with_docling(
self, self,
entity: Document, entity: Document,
docling_document: DoclingDocument, docling_document: "DoclingDocument",
chunks: list["Chunk"] | None = None, chunks: list["Chunk"] | None = None,
) -> Document: ) -> Document:
"""Create a document with its chunks and embeddings.""" """Create a document with its chunks and embeddings."""
@ -211,7 +211,7 @@ class DocumentRepository:
raise raise
async def _update_with_docling( async def _update_with_docling(
self, entity: Document, docling_document: DoclingDocument self, entity: Document, docling_document: "DoclingDocument"
) -> Document: ) -> Document:
"""Update a document and regenerate its chunks.""" """Update a document and regenerate its chunks."""
assert entity.id is not None, "Document ID is required for update" assert entity.id is not None, "Document ID is required for update"