Update client serialization for compressed docling documents

This commit is contained in:
Yiorgis Gozadinos 2026-01-11 09:23:56 +02:00
parent be715dd456
commit 9c4fd08c47
No known key found for this signature in database
5 changed files with 69 additions and 46 deletions

View file

@ -17,6 +17,7 @@ import httpx
from haiku.rag.config import AppConfig, Config from haiku.rag.config import AppConfig, Config
from haiku.rag.converters import get_converter from haiku.rag.converters import get_converter
from haiku.rag.reranking import get_reranker from haiku.rag.reranking import get_reranker
from haiku.rag.store.compression import compress_json
from haiku.rag.store.engine import Store from haiku.rag.store.engine import Store
from haiku.rag.store.models.chunk import Chunk, SearchResult from haiku.rag.store.models.chunk import Chunk, SearchResult
from haiku.rag.store.models.document import Document from haiku.rag.store.models.document import Document
@ -372,7 +373,7 @@ class HaikuRAG:
embedded_chunks = await embed_chunks(chunks, self._config) embedded_chunks = await embed_chunks(chunks, self._config)
# Store markdown export as content for better display/readability # Store markdown export as content for better display/readability
# The original content is preserved in docling_document_json # The original content is preserved in docling_document
stored_content = docling_document.export_to_markdown() stored_content = docling_document.export_to_markdown()
# Create document model # Create document model
@ -381,7 +382,7 @@ class HaikuRAG:
uri=uri, uri=uri,
title=title, title=title,
metadata=metadata or {}, metadata=metadata or {},
docling_document_json=docling_document.model_dump_json(), docling_document=compress_json(docling_document.model_dump_json()),
docling_version=docling_document.version, docling_version=docling_document.version,
) )
@ -417,7 +418,7 @@ class HaikuRAG:
uri=uri, uri=uri,
title=title, title=title,
metadata=metadata or {}, metadata=metadata or {},
docling_document_json=docling_document.model_dump_json(), docling_document=compress_json(docling_document.model_dump_json()),
docling_version=docling_document.version, docling_version=docling_document.version,
) )
@ -550,7 +551,9 @@ class HaikuRAG:
# Update existing document and rechunk # Update existing document and rechunk
existing_doc.content = docling_document.export_to_markdown() existing_doc.content = docling_document.export_to_markdown()
existing_doc.metadata = metadata existing_doc.metadata = metadata
existing_doc.docling_document_json = docling_document.model_dump_json() existing_doc.docling_document = compress_json(
docling_document.model_dump_json()
)
existing_doc.docling_version = docling_document.version existing_doc.docling_version = docling_document.version
if title is not None: if title is not None:
existing_doc.title = title existing_doc.title = title
@ -564,7 +567,7 @@ class HaikuRAG:
uri=uri, uri=uri,
title=title, title=title,
metadata=metadata, metadata=metadata,
docling_document_json=docling_document.model_dump_json(), docling_document=compress_json(docling_document.model_dump_json()),
docling_version=docling_document.version, docling_version=docling_document.version,
) )
return await self._store_document_with_chunks(document, embedded_chunks) return await self._store_document_with_chunks(document, embedded_chunks)
@ -657,7 +660,9 @@ class HaikuRAG:
# Update existing document and rechunk # Update existing document and rechunk
existing_doc.content = docling_document.export_to_markdown() existing_doc.content = docling_document.export_to_markdown()
existing_doc.metadata = metadata existing_doc.metadata = metadata
existing_doc.docling_document_json = docling_document.model_dump_json() existing_doc.docling_document = compress_json(
docling_document.model_dump_json()
)
existing_doc.docling_version = docling_document.version existing_doc.docling_version = docling_document.version
if title is not None: if title is not None:
existing_doc.title = title existing_doc.title = title
@ -671,7 +676,7 @@ class HaikuRAG:
uri=url, uri=url,
title=title, title=title,
metadata=metadata, metadata=metadata,
docling_document_json=docling_document.model_dump_json(), docling_document=compress_json(docling_document.model_dump_json()),
docling_version=docling_document.version, docling_version=docling_document.version,
) )
return await self._store_document_with_chunks(document, embedded_chunks) return await self._store_document_with_chunks(document, embedded_chunks)
@ -788,7 +793,9 @@ class HaikuRAG:
# Store docling data if provided # Store docling data if provided
if docling_document is not None: if docling_document is not None:
existing_doc.content = docling_document.export_to_markdown() existing_doc.content = docling_document.export_to_markdown()
existing_doc.docling_document_json = docling_document.model_dump_json() existing_doc.docling_document = compress_json(
docling_document.model_dump_json()
)
existing_doc.docling_version = docling_document.version existing_doc.docling_version = docling_document.version
elif content is not None: elif content is not None:
existing_doc.content = content existing_doc.content = content
@ -798,7 +805,9 @@ class HaikuRAG:
# DoclingDocument provided without chunks - chunk and embed using primitives # DoclingDocument provided without chunks - chunk and embed using primitives
if docling_document is not None: if docling_document is not None:
existing_doc.content = docling_document.export_to_markdown() existing_doc.content = docling_document.export_to_markdown()
existing_doc.docling_document_json = docling_document.model_dump_json() existing_doc.docling_document = compress_json(
docling_document.model_dump_json()
)
existing_doc.docling_version = docling_document.version existing_doc.docling_version = docling_document.version
new_chunks = await self.chunk(docling_document) new_chunks = await self.chunk(docling_document)
@ -810,7 +819,9 @@ class HaikuRAG:
# Content provided without chunks - convert, chunk, and embed using primitives # Content provided without chunks - convert, chunk, and embed using primitives
existing_doc.content = content # type: ignore[assignment] existing_doc.content = content # type: ignore[assignment]
converted_docling = await self.convert(existing_doc.content) converted_docling = await self.convert(existing_doc.content)
existing_doc.docling_document_json = converted_docling.model_dump_json() existing_doc.docling_document = compress_json(
converted_docling.model_dump_json()
)
existing_doc.docling_version = converted_docling.version existing_doc.docling_version = converted_docling.version
new_chunks = await self.chunk(converted_docling) new_chunks = await self.chunk(converted_docling)
@ -1485,7 +1496,7 @@ class HaikuRAG:
uri=doc.uri, uri=doc.uri,
title=doc.title, title=doc.title,
metadata=json.dumps(doc.metadata), metadata=json.dumps(doc.metadata),
docling_document_json=doc.docling_document_json, docling_document=doc.docling_document,
docling_version=doc.docling_version, docling_version=doc.docling_version,
created_at=doc.created_at.isoformat() if doc.created_at else now, created_at=doc.created_at.isoformat() if doc.created_at else now,
updated_at=now, updated_at=now,
@ -1523,7 +1534,7 @@ class HaikuRAG:
embedded_chunks = await embed_chunks(chunks, self._config) embedded_chunks = await embed_chunks(chunks, self._config)
# Update document fields # Update document fields
doc.docling_document_json = docling_document.model_dump_json() doc.docling_document = compress_json(docling_document.model_dump_json())
doc.docling_version = docling_document.version doc.docling_version = docling_document.version
# Prepare chunks with document_id and order # Prepare chunks with document_id and order
@ -1602,7 +1613,7 @@ class HaikuRAG:
chunks = await self.chunk(docling_document) chunks = await self.chunk(docling_document)
embedded_chunks = await embed_chunks(chunks, self._config) embedded_chunks = await embed_chunks(chunks, self._config)
doc.docling_document_json = docling_document.model_dump_json() doc.docling_document = compress_json(docling_document.model_dump_json())
doc.docling_version = docling_document.version doc.docling_version = docling_document.version
# Prepare chunks with document_id and order # Prepare chunks with document_id and order

View file

@ -8,6 +8,7 @@ from datasets import Dataset
from haiku.rag.client import HaikuRAG from haiku.rag.client import HaikuRAG
from haiku.rag.config import Config from haiku.rag.config import Config
from haiku.rag.store.compression import decompress_json
from haiku.rag.store.models.chunk import Chunk from haiku.rag.store.models.chunk import Chunk
from haiku.rag.store.models.document import Document from haiku.rag.store.models.document import Document
@ -790,13 +791,15 @@ async def test_client_create_document_stores_docling_json(temp_db_path):
) )
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
assert doc.docling_version is not None assert doc.docling_version is not None
# Verify JSON is valid and can be parsed # Verify JSON is valid and can be parsed
import json import json
parsed = json.loads(doc.docling_document_json) from haiku.rag.store.compression import decompress_json
parsed = json.loads(decompress_json(doc.docling_document))
assert "version" in parsed assert "version" in parsed
assert parsed["version"] == doc.docling_version assert parsed["version"] == doc.docling_version
@ -824,7 +827,8 @@ async def test_client_import_document_stores_docling_data(temp_db_path):
assert doc.id is not None assert doc.id is not None
assert "Content from docling document" in doc.content assert "Content from docling document" in doc.content
assert doc.docling_document_json == docling_doc.model_dump_json() assert doc.docling_document is not None
assert decompress_json(doc.docling_document) == docling_doc.model_dump_json()
assert doc.docling_version == docling_doc.version assert doc.docling_version == docling_doc.version
@ -840,13 +844,13 @@ async def test_client_create_document_from_file_stores_docling_json(temp_db_path
assert isinstance(doc, Document) assert isinstance(doc, Document)
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
assert doc.docling_version is not None assert doc.docling_version is not None
# Verify the stored document also has the JSON # Verify the stored document also has the JSON
retrieved = await client.get_document_by_id(doc.id) retrieved = await client.get_document_by_id(doc.id)
assert retrieved is not None assert retrieved is not None
assert retrieved.docling_document_json == doc.docling_document_json assert retrieved.docling_document == doc.docling_document
assert retrieved.docling_version == doc.docling_version assert retrieved.docling_version == doc.docling_version
@ -857,17 +861,17 @@ async def test_client_update_document_stores_docling_json(temp_db_path):
# Create initial document # Create initial document
doc = await client.create_document(content="Initial content") doc = await client.create_document(content="Initial content")
assert doc.id is not None assert doc.id is not None
original_json = doc.docling_document_json original_json = doc.docling_document
# Update content via update_document # Update content via update_document
updated_doc = await client.update_document( updated_doc = await client.update_document(
document_id=doc.id, content="New content via fields update" document_id=doc.id, content="New content via fields update"
) )
assert updated_doc.docling_document_json is not None assert updated_doc.docling_document is not None
assert updated_doc.docling_version is not None assert updated_doc.docling_version is not None
# JSON should be different because content changed # JSON should be different because content changed
assert updated_doc.docling_document_json != original_json assert updated_doc.docling_document != original_json
@pytest.mark.vcr() @pytest.mark.vcr()
@ -879,7 +883,7 @@ async def test_client_update_document_with_custom_chunks_no_docling_json(
# Create initial document # Create initial document
doc = await client.create_document(content="Initial content") doc = await client.create_document(content="Initial content")
assert doc.id is not None assert doc.id is not None
original_json = doc.docling_document_json original_json = doc.docling_document
# Update with custom chunks # Update with custom chunks
custom_chunks = [Chunk(content="Custom chunk", order=0)] custom_chunks = [Chunk(content="Custom chunk", order=0)]
@ -888,7 +892,7 @@ async def test_client_update_document_with_custom_chunks_no_docling_json(
) )
# Docling JSON should remain unchanged (no conversion when custom chunks provided) # Docling JSON should remain unchanged (no conversion when custom chunks provided)
assert updated_doc.docling_document_json == original_json assert updated_doc.docling_document == original_json
@pytest.mark.vcr() @pytest.mark.vcr()
@ -942,7 +946,11 @@ async def test_client_update_document_with_docling_rechunks(temp_db_path):
# Content should be extracted from docling document # Content should be extracted from docling document
assert "Completely different text" in updated_doc.content assert "Completely different text" in updated_doc.content
assert updated_doc.docling_document_json == docling_doc.model_dump_json() assert updated_doc.docling_document is not None
assert (
decompress_json(updated_doc.docling_document)
== docling_doc.model_dump_json()
)
assert updated_doc.docling_version == docling_doc.version assert updated_doc.docling_version == docling_doc.version
# Chunks should be regenerated # Chunks should be regenerated
@ -981,7 +989,11 @@ async def test_client_update_document_docling_with_chunks(temp_db_path):
# Content should be extracted from docling (since content wasn't provided) # Content should be extracted from docling (since content wasn't provided)
assert "Text from docling" in updated_doc.content assert "Text from docling" in updated_doc.content
assert updated_doc.docling_document_json == docling_doc.model_dump_json() assert updated_doc.docling_document is not None
assert (
decompress_json(updated_doc.docling_document)
== docling_doc.model_dump_json()
)
# Custom chunks should be used (not rechunked from docling) # Custom chunks should be used (not rechunked from docling)
chunks = await client.chunk_repository.get_by_document_id(doc.id) chunks = await client.chunk_repository.get_by_document_id(doc.id)
@ -1001,7 +1013,7 @@ async def test_client_file_update_stores_docling_json(temp_db_path):
# Create initial document # Create initial document
doc1 = await client.create_document_from_source(temp_path) doc1 = await client.create_document_from_source(temp_path)
assert isinstance(doc1, Document) assert isinstance(doc1, Document)
original_json = doc1.docling_document_json original_json = doc1.docling_document
original_version = doc1.docling_version original_version = doc1.docling_version
# Modify file # Modify file
@ -1013,8 +1025,8 @@ async def test_client_file_update_stores_docling_json(temp_db_path):
assert doc2.id == doc1.id # Same document assert doc2.id == doc1.id # Same document
# Docling JSON should be updated # Docling JSON should be updated
assert doc2.docling_document_json is not None assert doc2.docling_document is not None
assert doc2.docling_document_json != original_json assert doc2.docling_document != original_json
assert doc2.docling_version == original_version # Version stays same assert doc2.docling_version == original_version # Version stays same
@ -1038,7 +1050,7 @@ async def test_client_visualize_chunk_no_bounding_boxes(temp_db_path):
) )
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
chunks = await client.chunk_repository.get_by_document_id(doc.id) chunks = await client.chunk_repository.get_by_document_id(doc.id)
assert len(chunks) >= 1 assert len(chunks) >= 1
@ -1095,7 +1107,7 @@ async def test_client_visualize_chunk_with_pdf(temp_db_path):
doc = await client.create_document_from_source(pdf_path) doc = await client.create_document_from_source(pdf_path)
assert isinstance(doc, Document) assert isinstance(doc, Document)
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
chunks = await client.chunk_repository.get_by_document_id(doc.id) chunks = await client.chunk_repository.get_by_document_id(doc.id)
assert len(chunks) > 0 assert len(chunks) > 0
@ -1345,7 +1357,7 @@ async def test_client_create_document_with_html_format(temp_db_path):
) )
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
# Verify the DoclingDocument has proper structure # Verify the DoclingDocument has proper structure
docling_doc = doc.get_docling_document() docling_doc = doc.get_docling_document()

View file

@ -613,7 +613,7 @@ This is paragraph four about topic C.
) )
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
# Get chunks which should have doc_item_refs # Get chunks which should have doc_item_refs
chunks = await client.chunk_repository.get_by_document_id(doc.id) chunks = await client.chunk_repository.get_by_document_id(doc.id)

View file

@ -20,7 +20,7 @@ def vcr_cassette_dir():
return str(Path(__file__).parent / "cassettes" / "test_converters") return str(Path(__file__).parent / "cassettes" / "test_converters")
def create_mock_docling_document_json(name: str = "test") -> dict: def create_mock_docling_document(name: str = "test") -> dict:
"""Create a minimal valid DoclingDocument JSON structure for mocking.""" """Create a minimal valid DoclingDocument JSON structure for mocking."""
return { return {
"schema_name": "DoclingDocument", "schema_name": "DoclingDocument",
@ -476,7 +476,7 @@ class TestDoclingServeConverter:
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_convert_text_success(self, converter): async def test_convert_text_success(self, converter):
"""Test successful text conversion via docling-serve async workflow.""" """Test successful text conversion via docling-serve async workflow."""
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -498,7 +498,7 @@ class TestDoclingServeConverter:
config.providers.docling_serve.api_key = "test-key" config.providers.docling_serve.api_key = "test-key"
converter = DoclingServeConverter(config) converter = DoclingServeConverter(config)
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -527,7 +527,7 @@ class TestDoclingServeConverter:
config.processing.conversion_options.images_scale = 3.0 config.processing.conversion_options.images_scale = 3.0
converter = DoclingServeConverter(config) converter = DoclingServeConverter(config)
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -622,7 +622,7 @@ class TestDoclingServeConverter:
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_convert_file_pdf(self, converter): async def test_convert_file_pdf(self, converter):
"""Test converting PDF file via docling-serve async workflow.""" """Test converting PDF file via docling-serve async workflow."""
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -645,7 +645,7 @@ class TestDoclingServeConverter:
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_convert_file_text(self, converter): async def test_convert_file_text(self, converter):
"""Test converting text file (reads locally, sends to docling-serve).""" """Test converting text file (reads locally, sends to docling-serve)."""
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -734,7 +734,7 @@ class TestDoclingServeConverterPictureDescription:
config.prompts.picture_description = "Test prompt for picture description" config.prompts.picture_description = "Test prompt for picture description"
converter = DoclingServeConverter(config) converter = DoclingServeConverter(config)
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -767,7 +767,7 @@ class TestDoclingServeConverterPictureDescription:
"""Test that picture description is disabled by default.""" """Test that picture description is disabled by default."""
converter = DoclingServeConverter(config) converter = DoclingServeConverter(config)
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:

View file

@ -10,7 +10,7 @@ async def test_rebuild_full(qa_corpus: Dataset, temp_db_path):
async with HaikuRAG(temp_db_path, create=True) as client: async with HaikuRAG(temp_db_path, create=True) as client:
doc = await client.create_document(content=qa_corpus["document_extracted"][0]) doc = await client.create_document(content=qa_corpus["document_extracted"][0])
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
chunks_before = await client.chunk_repository.get_by_document_id(doc.id) chunks_before = await client.chunk_repository.get_by_document_id(doc.id)
assert len(chunks_before) > 0 assert len(chunks_before) > 0
@ -23,7 +23,7 @@ async def test_rebuild_full(qa_corpus: Dataset, temp_db_path):
# Verify DoclingDocument JSON is preserved after rebuild # Verify DoclingDocument JSON is preserved after rebuild
doc_after = await client.document_repository.get_by_id(doc.id) doc_after = await client.document_repository.get_by_id(doc.id)
assert doc_after is not None assert doc_after is not None
assert doc_after.docling_document_json is not None assert doc_after.docling_document is not None
assert doc_after.docling_version is not None assert doc_after.docling_version is not None
chunks_after = await client.chunk_repository.get_by_document_id(doc.id) chunks_after = await client.chunk_repository.get_by_document_id(doc.id)
@ -40,7 +40,7 @@ async def test_rebuild_embed_only(qa_corpus: Dataset, temp_db_path):
async with HaikuRAG(temp_db_path, create=True) as client: async with HaikuRAG(temp_db_path, create=True) as client:
doc = await client.create_document(content=qa_corpus["document_extracted"][0]) doc = await client.create_document(content=qa_corpus["document_extracted"][0])
assert doc.id is not None assert doc.id is not None
original_docling_json = doc.docling_document_json original_docling_json = doc.docling_document
chunks_before = await client.chunk_repository.get_by_document_id(doc.id) chunks_before = await client.chunk_repository.get_by_document_id(doc.id)
assert len(chunks_before) > 0 assert len(chunks_before) > 0
@ -57,7 +57,7 @@ async def test_rebuild_embed_only(qa_corpus: Dataset, temp_db_path):
# DoclingDocument JSON should be unchanged (embed-only doesn't touch documents) # DoclingDocument JSON should be unchanged (embed-only doesn't touch documents)
doc_after = await client.document_repository.get_by_id(doc.id) doc_after = await client.document_repository.get_by_id(doc.id)
assert doc_after is not None assert doc_after is not None
assert doc_after.docling_document_json == original_docling_json assert doc_after.docling_document == original_docling_json
chunks_after = await client.chunk_repository.get_by_document_id(doc.id) chunks_after = await client.chunk_repository.get_by_document_id(doc.id)
chunk_ids_after = {c.id for c in chunks_after} chunk_ids_after = {c.id for c in chunks_after}
@ -112,7 +112,7 @@ async def test_rebuild_rechunk(qa_corpus: Dataset, temp_db_path):
async with HaikuRAG(temp_db_path, create=True) as client: async with HaikuRAG(temp_db_path, create=True) as client:
doc = await client.create_document(content=qa_corpus["document_extracted"][0]) doc = await client.create_document(content=qa_corpus["document_extracted"][0])
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
# Set a fake URI to simulate a document that came from a file # Set a fake URI to simulate a document that came from a file
doc.uri = "file:///nonexistent/path.txt" doc.uri = "file:///nonexistent/path.txt"
@ -133,7 +133,7 @@ async def test_rebuild_rechunk(qa_corpus: Dataset, temp_db_path):
doc_after = await client.document_repository.get_by_id(doc.id) doc_after = await client.document_repository.get_by_id(doc.id)
assert doc_after is not None assert doc_after is not None
assert doc_after.content == content_before assert doc_after.content == content_before
assert doc_after.docling_document_json is not None assert doc_after.docling_document is not None
assert doc_after.docling_version is not None assert doc_after.docling_version is not None
chunks_after = await client.chunk_repository.get_by_document_id(doc.id) chunks_after = await client.chunk_repository.get_by_document_id(doc.id)