Merge pull request #138 from ggozad/fix/create-with-chunks

Do not create unecessary DoclingDocument when chunks are provided.
This commit is contained in:
Yiorgis Gozadinos 2025-11-13 13:20:39 +02:00 committed by GitHub
commit d2771399a7
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 39 additions and 28 deletions

View file

@ -1,8 +1,13 @@
# Changelog
## [Unreleased]
### Changed
### Fixed
- **Document Creation**: Optimized `create_document` to skip unnecessary DoclingDocument conversion when chunks are pre-provided
- **FileReader**: Error messages now include both original exception details and file path for easier debugging
## [0.15.0] - 2025-11-07

View file

@ -73,7 +73,7 @@ class HaikuRAG:
title=title,
metadata=metadata or {},
)
return await self.document_repository._create_with_docling(
return await self.document_repository._create_and_chunk(
document, docling_document, chunks
)
@ -96,19 +96,24 @@ class HaikuRAG:
Returns:
The created Document instance.
"""
# Lazy import to avoid loading docling
from haiku.rag.utils import text_to_docling_document
# Convert content to DoclingDocument for processing
docling_document = text_to_docling_document(content)
document = Document(
content=content,
uri=uri,
title=title,
metadata=metadata or {},
)
return await self.document_repository._create_with_docling(
# Only create docling_document if we need to generate chunks
if chunks is None:
# Lazy import to avoid loading docling
from haiku.rag.utils import text_to_docling_document
docling_document = text_to_docling_document(content)
else:
# Chunks already provided, no conversion needed
docling_document = None
return await self.document_repository._create_and_chunk(
document, docling_document, chunks
)
@ -239,7 +244,7 @@ class HaikuRAG:
existing_doc.metadata = metadata
if title is not None:
existing_doc.title = title
return await self.document_repository._update_with_docling(
return await self.document_repository._update_and_rechunk(
existing_doc, docling_document
)
else:
@ -333,7 +338,7 @@ class HaikuRAG:
existing_doc.metadata = metadata
if title is not None:
existing_doc.title = title
return await self.document_repository._update_with_docling(
return await self.document_repository._update_and_rechunk(
existing_doc, docling_document
)
else:
@ -405,7 +410,7 @@ class HaikuRAG:
# Convert content to DoclingDocument
docling_document = text_to_docling_document(document.content)
return await self.document_repository._update_with_docling(
return await self.document_repository._update_and_rechunk(
document, docling_document
)

View file

@ -182,10 +182,10 @@ class DocumentRepository:
"documents", schema=DocumentRecord
)
async def _create_with_docling(
async def _create_and_chunk(
self,
entity: Document,
docling_document: "DoclingDocument",
docling_document: "DoclingDocument | None",
chunks: list["Chunk"] | None = None,
) -> Document:
"""Create a document with its chunks and embeddings."""
@ -199,6 +199,9 @@ class DocumentRepository:
try:
# Create chunks if not provided
if chunks is None:
assert docling_document is not None, (
"docling_document is required when chunks are not provided"
)
assert created_doc.id is not None, (
"Document ID should not be None after creation"
)
@ -224,7 +227,7 @@ class DocumentRepository:
self.store.restore_table_versions(versions)
raise
async def _update_with_docling(
async def _update_and_rechunk(
self, entity: Document, docling_document: "DoclingDocument"
) -> Document:
"""Update a document and regenerate its chunks."""

View file

@ -26,7 +26,7 @@ async def test_chunk_repository_operations(qa_corpus: Dataset, temp_db_path):
from haiku.rag.utils import text_to_docling_document
docling_document = text_to_docling_document(document_text, name="test.md")
created_document = await doc_repo._create_with_docling(document, docling_document)
created_document = await doc_repo._create_and_chunk(document, docling_document)
assert created_document.id is not None
# Test getting chunks by document ID

View file

@ -29,7 +29,7 @@ async def test_create_document_with_chunks(qa_corpus: Dataset, temp_db_path):
docling_document = text_to_docling_document(document_text, name="test.md")
# Create the document with chunks in the database
created_document = await doc_repo._create_with_docling(document, docling_document)
created_document = await doc_repo._create_and_chunk(document, docling_document)
# Verify the document was created
assert created_document.id is not None

View file

@ -36,9 +36,7 @@ async def test_search_qa_corpus(qa_corpus: Dataset, temp_db_path):
from haiku.rag.utils import text_to_docling_document
docling_document = text_to_docling_document(document_text, name="test.md")
created_document = await doc_repo._create_with_docling(
document, docling_document
)
created_document = await doc_repo._create_and_chunk(document, docling_document)
documents.append((created_document, doc_data))
# Test with first few unique documents
@ -86,7 +84,7 @@ async def test_chunks_include_document_info(temp_db_path):
from haiku.rag.utils import text_to_docling_document
docling_document = text_to_docling_document(document.content, name="test.md")
created_document = await doc_repo._create_with_docling(document, docling_document)
created_document = await doc_repo._create_and_chunk(document, docling_document)
# Search for chunks
results = await chunk_repo.search("test document", limit=1, search_type="hybrid")
@ -124,7 +122,7 @@ async def test_chunks_include_document_title(temp_db_path):
from haiku.rag.utils import text_to_docling_document
dl = text_to_docling_document(document.content, name="title-test.md")
await doc_repo._create_with_docling(document, dl)
await doc_repo._create_and_chunk(document, dl)
# Perform a search that should find this document
results = await chunk_repo.search("custom title", limit=3, search_type="hybrid")
@ -158,7 +156,7 @@ async def test_search_score_types(temp_db_path):
from haiku.rag.utils import text_to_docling_document
docling_document = text_to_docling_document(content, name="test.md")
await doc_repo._create_with_docling(document, docling_document)
await doc_repo._create_and_chunk(document, docling_document)
query = "machine learning"

View file

@ -37,7 +37,7 @@ async def test_version_rollback_on_create_failure(temp_db_path):
dl_doc = text_to_docling_document(content, name="test.md")
with pytest.raises(RuntimeError):
await repo._create_with_docling(doc, dl_doc)
await repo._create_and_chunk(doc, dl_doc)
# State should be restored (no documents/chunks)
docs = await repo.list_all()
@ -66,7 +66,7 @@ async def test_version_rollback_on_update_failure(temp_db_path):
base_content = "Base content"
base_doc = Document(content=base_content)
base_dl = text_to_docling_document(base_content, name="base.md")
created = await repo._create_with_docling(base_doc, base_dl)
created = await repo._create_and_chunk(base_doc, base_dl)
# Force new chunk creation to fail during update after writing
orig = repo.chunk_repository.create_chunks_for_document
@ -83,7 +83,7 @@ async def test_version_rollback_on_update_failure(temp_db_path):
updated_dl = text_to_docling_document(updated_content, name="updated.md")
with pytest.raises(RuntimeError):
await repo._update_with_docling(created, updated_dl)
await repo._update_and_rechunk(created, updated_dl)
# Content and chunks should remain the original
persisted = await repo.get_by_id(created.id) # type: ignore[arg-type]
@ -142,12 +142,12 @@ async def test_vacuum_with_retention_threshold(temp_db_path):
# Create first document
doc1 = Document(content="First document")
dl_doc1 = text_to_docling_document("First document", name="doc1.md")
await repo._create_with_docling(doc1, dl_doc1)
await repo._create_and_chunk(doc1, dl_doc1)
# Create second document
doc2 = Document(content="Second document")
dl_doc2 = text_to_docling_document("Second document", name="doc2.md")
await repo._create_with_docling(doc2, dl_doc2)
await repo._create_and_chunk(doc2, dl_doc2)
# Get initial version counts (should have multiple versions from creates)
initial_doc_versions = len(list(store.documents_table.list_versions()))
@ -212,7 +212,7 @@ async def test_vacuum_completes_before_context_exit(temp_db_path, monkeypatch):
for i in range(3):
doc = Document(content=f"Test document {i}")
dl_doc = text_to_docling_document(f"Test document {i}", name=f"test{i}.md")
await client.document_repository._create_with_docling(doc, dl_doc)
await client.document_repository._create_and_chunk(doc, dl_doc)
# After context exit, automatic vacuum should have kept versions minimal
store = Store(temp_db_path)