Simplify import_document, update_document
This commit is contained in:
parent
a1162f8020
commit
a56e1ba67c
6 changed files with 82 additions and 202 deletions
|
|
@ -29,9 +29,9 @@
|
||||||
- `contextualize()` - Prepend section headings to chunk content for embedding
|
- `contextualize()` - Prepend section headings to chunk content for embedding
|
||||||
- `embed_chunks()` - Generate embeddings for chunks
|
- `embed_chunks()` - Generate embeddings for chunks
|
||||||
- **New `import_document()` Method**: Import pre-processed documents with custom chunks
|
- **New `import_document()` Method**: Import pre-processed documents with custom chunks
|
||||||
|
- Accepts `DoclingDocument` directly for rich metadata (visual grounding, page numbers)
|
||||||
- Use when document conversion, chunking, or embedding were done externally
|
- Use when document conversion, chunking, or embedding were done externally
|
||||||
- Chunks without embeddings are automatically embedded
|
- Chunks without embeddings are automatically embedded
|
||||||
- Supports `docling_document_json` for rich metadata (visual grounding, page numbers)
|
|
||||||
- **Automatic Chunk Embedding**: `import_document()` and `update_document()` automatically embed chunks that don't have embeddings
|
- **Automatic Chunk Embedding**: `import_document()` and `update_document()` automatically embed chunks that don't have embeddings
|
||||||
- Pass chunks with or without embeddings - missing embeddings are generated
|
- Pass chunks with or without embeddings - missing embeddings are generated
|
||||||
- Chunks with pre-computed embeddings are stored as-is
|
- Chunks with pre-computed embeddings are stored as-is
|
||||||
|
|
@ -48,8 +48,8 @@
|
||||||
- Use `import_document()` for pre-processed documents with custom chunks
|
- Use `import_document()` for pre-processed documents with custom chunks
|
||||||
- **BREAKING: `update_document()` API**: Unified with `update_document_fields()`
|
- **BREAKING: `update_document()` API**: Unified with `update_document_fields()`
|
||||||
- Old: `update_document(document)` - pass modified Document object
|
- Old: `update_document(document)` - pass modified Document object
|
||||||
- New: `update_document(document_id, content=, metadata=, chunks=, title=, docling_document_json=, docling_version=)`
|
- New: `update_document(document_id, content=, metadata=, chunks=, title=, docling_document=)`
|
||||||
- `content` and `docling_document_json` are mutually exclusive
|
- `content` and `docling_document` are mutually exclusive
|
||||||
- **BREAKING: Chunker Interface**: `DocumentChunker.chunk()` now returns `list[Chunk]` instead of `list[str]`
|
- **BREAKING: Chunker Interface**: `DocumentChunker.chunk()` now returns `list[Chunk]` instead of `list[str]`
|
||||||
- Chunks include structured metadata (doc_item_refs, labels, headings, page_numbers)
|
- Chunks include structured metadata (doc_item_refs, labels, headings, page_numbers)
|
||||||
- **BREAKING: Config Renamed**: `context_chunk_radius` renamed to `text_context_radius`
|
- **BREAKING: Config Renamed**: `context_chunk_radius` renamed to `text_context_radius`
|
||||||
|
|
|
||||||
|
|
@ -33,9 +33,8 @@ async with HaikuRAG("database.lancedb", create=True) as client:
|
||||||
|
|
||||||
# 4. Store the document with chunks
|
# 4. Store the document with chunks
|
||||||
doc = await client.import_document(
|
doc = await client.import_document(
|
||||||
|
docling_document=docling_doc,
|
||||||
chunks=embedded_chunks,
|
chunks=embedded_chunks,
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
|
||||||
docling_version=docling_doc.version,
|
|
||||||
uri="file:///path/to/document.pdf",
|
uri="file:///path/to/document.pdf",
|
||||||
title="My Document",
|
title="My Document",
|
||||||
)
|
)
|
||||||
|
|
@ -181,9 +180,8 @@ async with HaikuRAG("database.lancedb", create=True) as client:
|
||||||
embedded_chunks = await embed_chunks(filtered)
|
embedded_chunks = await embed_chunks(filtered)
|
||||||
|
|
||||||
await client.import_document(
|
await client.import_document(
|
||||||
|
docling_document=docling_doc,
|
||||||
chunks=embedded_chunks,
|
chunks=embedded_chunks,
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
|
||||||
docling_version=docling_doc.version,
|
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
@ -221,9 +219,8 @@ async with HaikuRAG("database.lancedb", create=True) as client:
|
||||||
]
|
]
|
||||||
|
|
||||||
await client.import_document(
|
await client.import_document(
|
||||||
|
docling_document=docling_doc,
|
||||||
chunks=embedded_chunks,
|
chunks=embedded_chunks,
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
|
||||||
docling_version=docling_doc.version,
|
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -75,6 +75,9 @@ If you process documents externally or need custom processing, use `import_docum
|
||||||
```python
|
```python
|
||||||
from haiku.rag.store.models.chunk import Chunk
|
from haiku.rag.store.models.chunk import Chunk
|
||||||
|
|
||||||
|
# Convert your source to a DoclingDocument
|
||||||
|
docling_doc = await client.convert("path/to/document.pdf")
|
||||||
|
|
||||||
# Create chunks (embeddings optional - will be generated if missing)
|
# Create chunks (embeddings optional - will be generated if missing)
|
||||||
chunks = [
|
chunks = [
|
||||||
Chunk(
|
Chunk(
|
||||||
|
|
@ -92,7 +95,7 @@ chunks = [
|
||||||
|
|
||||||
# Import document with custom chunks
|
# Import document with custom chunks
|
||||||
doc = await client.import_document(
|
doc = await client.import_document(
|
||||||
content="Full document content",
|
docling_document=docling_doc,
|
||||||
chunks=chunks,
|
chunks=chunks,
|
||||||
uri="doc://custom",
|
uri="doc://custom",
|
||||||
title="Custom Document",
|
title="Custom Document",
|
||||||
|
|
@ -100,18 +103,7 @@ doc = await client.import_document(
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
With a DoclingDocument for rich metadata (visual grounding, page numbers):
|
The `docling_document` provides rich metadata for visual grounding, page numbers, and section headings. Content is automatically extracted from the DoclingDocument.
|
||||||
|
|
||||||
```python
|
|
||||||
doc = await client.import_document(
|
|
||||||
chunks=chunks,
|
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
|
||||||
docling_version=docling_doc.version,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
!!! note
|
|
||||||
Either `content` or `docling_document_json` must be provided. When `docling_document_json` is provided without `content`, the content is extracted automatically.
|
|
||||||
|
|
||||||
See [Custom Processing Pipelines](custom-pipelines.md) for building pipelines with `convert()`, `chunk()`, and `embed_chunks()`.
|
See [Custom Processing Pipelines](custom-pipelines.md) for building pipelines with `convert()`, `chunk()`, and `embed_chunks()`.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -380,13 +380,11 @@ class HaikuRAG:
|
||||||
|
|
||||||
async def import_document(
|
async def import_document(
|
||||||
self,
|
self,
|
||||||
|
docling_document: "DoclingDocument",
|
||||||
chunks: list[Chunk],
|
chunks: list[Chunk],
|
||||||
content: str | None = None,
|
|
||||||
uri: str | None = None,
|
uri: str | None = None,
|
||||||
title: str | None = None,
|
title: str | None = None,
|
||||||
metadata: dict | None = None,
|
metadata: dict | None = None,
|
||||||
docling_document_json: str | None = None,
|
|
||||||
docling_version: str | None = None,
|
|
||||||
) -> Document:
|
) -> Document:
|
||||||
"""Import a pre-processed document with chunks.
|
"""Import a pre-processed document with chunks.
|
||||||
|
|
||||||
|
|
@ -394,56 +392,23 @@ class HaikuRAG:
|
||||||
externally and you want to store the results in haiku.rag.
|
externally and you want to store the results in haiku.rag.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
chunks: Pre-created chunks.
|
docling_document: The DoclingDocument to import.
|
||||||
content: The document content. Optional if docling_document_json is provided.
|
chunks: Pre-created chunks. Chunks without embeddings will be
|
||||||
|
automatically embedded.
|
||||||
uri: Optional URI identifier for the document.
|
uri: Optional URI identifier for the document.
|
||||||
title: Optional title for the document.
|
title: Optional title for the document.
|
||||||
metadata: Optional metadata dictionary.
|
metadata: Optional metadata dictionary.
|
||||||
docling_document_json: Serialized DoclingDocument JSON. If provided without
|
|
||||||
content, content is extracted from the DoclingDocument.
|
|
||||||
docling_version: DoclingDocument schema version (required with docling_document_json).
|
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
The created Document instance.
|
The created Document instance.
|
||||||
|
|
||||||
Raises:
|
|
||||||
ValueError: If neither content nor docling_document_json is provided,
|
|
||||||
if docling_document_json is provided without docling_version,
|
|
||||||
or if the JSON is invalid.
|
|
||||||
"""
|
"""
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
|
||||||
|
|
||||||
# Validate docling parameters must be provided together
|
|
||||||
if (docling_document_json is None) != (docling_version is None):
|
|
||||||
raise ValueError(
|
|
||||||
"docling_document_json and docling_version must both be provided or both be None"
|
|
||||||
)
|
|
||||||
|
|
||||||
# Validate that we have at least one content source
|
|
||||||
if content is None and docling_document_json is None:
|
|
||||||
raise ValueError("Either content or docling_document_json must be provided")
|
|
||||||
|
|
||||||
# Parse and validate docling JSON if provided
|
|
||||||
docling_document: DoclingDocument | None = None
|
|
||||||
if docling_document_json is not None:
|
|
||||||
try:
|
|
||||||
docling_document = DoclingDocument.model_validate_json(
|
|
||||||
docling_document_json
|
|
||||||
)
|
|
||||||
except Exception as e:
|
|
||||||
raise ValueError(f"Invalid docling_document_json: {e}") from e
|
|
||||||
|
|
||||||
# Extract content from docling if not explicitly provided
|
|
||||||
if content is None and docling_document is not None:
|
|
||||||
content = docling_document.export_to_markdown()
|
|
||||||
|
|
||||||
document = Document(
|
document = Document(
|
||||||
content=content, # type: ignore[arg-type]
|
content=docling_document.export_to_markdown(),
|
||||||
uri=uri,
|
uri=uri,
|
||||||
title=title,
|
title=title,
|
||||||
metadata=metadata or {},
|
metadata=metadata or {},
|
||||||
docling_document_json=docling_document_json,
|
docling_document_json=docling_document.model_dump_json(),
|
||||||
docling_version=docling_version,
|
docling_version=docling_document.version,
|
||||||
)
|
)
|
||||||
|
|
||||||
return await self._store_document_with_chunks(document, chunks)
|
return await self._store_document_with_chunks(document, chunks)
|
||||||
|
|
@ -761,58 +726,38 @@ class HaikuRAG:
|
||||||
metadata: dict | None = None,
|
metadata: dict | None = None,
|
||||||
chunks: list[Chunk] | None = None,
|
chunks: list[Chunk] | None = None,
|
||||||
title: str | None = None,
|
title: str | None = None,
|
||||||
docling_document_json: str | None = None,
|
docling_document: "DoclingDocument | None" = None,
|
||||||
docling_version: str | None = None,
|
|
||||||
) -> Document:
|
) -> Document:
|
||||||
"""Update a document by ID.
|
"""Update a document by ID.
|
||||||
|
|
||||||
Updates specified fields. When content or docling_document_json is provided,
|
Updates specified fields. When content or docling_document is provided,
|
||||||
the document is rechunked and re-embedded. Updates to only metadata or title
|
the document is rechunked and re-embedded. Updates to only metadata or title
|
||||||
skip rechunking for efficiency.
|
skip rechunking for efficiency.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
document_id: The ID of the document to update.
|
document_id: The ID of the document to update.
|
||||||
content: New content (mutually exclusive with docling_document_json).
|
content: New content (mutually exclusive with docling_document).
|
||||||
metadata: New metadata dict.
|
metadata: New metadata dict.
|
||||||
chunks: Custom pre-embedded chunks (skips auto-chunking).
|
chunks: Custom chunks (will be embedded if missing embeddings).
|
||||||
title: New title.
|
title: New title.
|
||||||
docling_document_json: Serialized DoclingDocument JSON (mutually exclusive with content).
|
docling_document: DoclingDocument to replace content (mutually exclusive with content).
|
||||||
docling_version: DoclingDocument schema version (required with docling_document_json).
|
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
The updated Document instance.
|
The updated Document instance.
|
||||||
|
|
||||||
Raises:
|
Raises:
|
||||||
ValueError: If document not found, if both content and docling_document_json
|
ValueError: If document not found, or if both content and docling_document
|
||||||
are provided, or if docling_document_json is provided without docling_version.
|
are provided.
|
||||||
"""
|
"""
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
|
||||||
|
|
||||||
from haiku.rag.embeddings import embed_chunks
|
from haiku.rag.embeddings import embed_chunks
|
||||||
|
|
||||||
# Validate: content and docling_document_json are mutually exclusive
|
# Validate: content and docling_document are mutually exclusive
|
||||||
if content is not None and docling_document_json is not None:
|
if content is not None and docling_document is not None:
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
"content and docling_document_json are mutually exclusive. "
|
"content and docling_document are mutually exclusive. "
|
||||||
"Provide one or the other, not both."
|
"Provide one or the other, not both."
|
||||||
)
|
)
|
||||||
|
|
||||||
# Validate docling parameters must be provided together
|
|
||||||
if (docling_document_json is None) != (docling_version is None):
|
|
||||||
raise ValueError(
|
|
||||||
"docling_document_json and docling_version must both be provided or both be None"
|
|
||||||
)
|
|
||||||
|
|
||||||
# Parse and validate docling JSON if provided
|
|
||||||
docling_document: DoclingDocument | None = None
|
|
||||||
if docling_document_json is not None:
|
|
||||||
try:
|
|
||||||
docling_document = DoclingDocument.model_validate_json(
|
|
||||||
docling_document_json
|
|
||||||
)
|
|
||||||
except Exception as e:
|
|
||||||
raise ValueError(f"Invalid docling_document_json: {e}") from e
|
|
||||||
|
|
||||||
# Fetch the existing document
|
# Fetch the existing document
|
||||||
existing_doc = await self.get_document_by_id(document_id)
|
existing_doc = await self.get_document_by_id(document_id)
|
||||||
if existing_doc is None:
|
if existing_doc is None:
|
||||||
|
|
@ -828,27 +773,23 @@ class HaikuRAG:
|
||||||
if content is None and chunks is None and docling_document is None:
|
if content is None and chunks is None and docling_document is None:
|
||||||
return await self.document_repository.update(existing_doc)
|
return await self.document_repository.update(existing_doc)
|
||||||
|
|
||||||
# Custom chunks provided - use them as-is (pre-embedded)
|
# Custom chunks provided - use them as-is
|
||||||
if chunks is not None:
|
if chunks is not None:
|
||||||
# Update content field if provided
|
|
||||||
if content is not None:
|
|
||||||
existing_doc.content = content
|
|
||||||
|
|
||||||
# Store docling data if provided
|
# Store docling data if provided
|
||||||
if docling_document is not None:
|
if docling_document is not None:
|
||||||
existing_doc.docling_document_json = docling_document_json
|
existing_doc.content = docling_document.export_to_markdown()
|
||||||
existing_doc.docling_version = docling_version
|
existing_doc.docling_document_json = docling_document.model_dump_json()
|
||||||
# Extract content from docling if not explicitly provided
|
existing_doc.docling_version = docling_document.version
|
||||||
if content is None:
|
elif content is not None:
|
||||||
existing_doc.content = docling_document.export_to_markdown()
|
existing_doc.content = content
|
||||||
|
|
||||||
return await self._update_document_with_chunks(existing_doc, chunks)
|
return await self._update_document_with_chunks(existing_doc, chunks)
|
||||||
|
|
||||||
# DoclingDocument provided without chunks - chunk and embed using primitives
|
# DoclingDocument provided without chunks - chunk and embed using primitives
|
||||||
if docling_document is not None:
|
if docling_document is not None:
|
||||||
existing_doc.content = docling_document.export_to_markdown()
|
existing_doc.content = docling_document.export_to_markdown()
|
||||||
existing_doc.docling_document_json = docling_document_json
|
existing_doc.docling_document_json = docling_document.model_dump_json()
|
||||||
existing_doc.docling_version = docling_version
|
existing_doc.docling_version = docling_document.version
|
||||||
|
|
||||||
new_chunks = await self.chunk(docling_document)
|
new_chunks = await self.chunk(docling_document)
|
||||||
embedded_chunks = await embed_chunks(new_chunks, self._config)
|
embedded_chunks = await embed_chunks(new_chunks, self._config)
|
||||||
|
|
|
||||||
|
|
@ -689,7 +689,14 @@ async def test_client_async_context_manager(temp_db_path):
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_client_import_document_with_custom_chunks(temp_db_path):
|
async def test_client_import_document_with_custom_chunks(temp_db_path):
|
||||||
"""Test importing a document with pre-created chunks."""
|
"""Test importing a document with pre-created chunks."""
|
||||||
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
from docling_core.types.doc.labels import DocItemLabel
|
||||||
|
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
|
# Create a DoclingDocument
|
||||||
|
docling_doc = DoclingDocument(name="test")
|
||||||
|
docling_doc.add_text(label=DocItemLabel.TEXT, text="Full document content")
|
||||||
|
|
||||||
# Create some custom chunks with and without embeddings
|
# Create some custom chunks with and without embeddings
|
||||||
chunks = [
|
chunks = [
|
||||||
Chunk(
|
Chunk(
|
||||||
|
|
@ -712,11 +719,11 @@ async def test_client_import_document_with_custom_chunks(temp_db_path):
|
||||||
|
|
||||||
# Import document with custom chunks
|
# Import document with custom chunks
|
||||||
document = await client.import_document(
|
document = await client.import_document(
|
||||||
content="Full document content", chunks=chunks
|
docling_document=docling_doc, chunks=chunks
|
||||||
)
|
)
|
||||||
|
|
||||||
assert document.id is not None
|
assert document.id is not None
|
||||||
assert document.content == "Full document content"
|
assert "Full document content" in document.content
|
||||||
|
|
||||||
# Verify the chunks were created correctly
|
# Verify the chunks were created correctly
|
||||||
doc_chunks = await client.chunk_repository.get_by_document_id(document.id)
|
doc_chunks = await client.chunk_repository.get_by_document_id(document.id)
|
||||||
|
|
@ -797,59 +804,8 @@ async def test_client_create_document_stores_docling_json(temp_db_path):
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_client_import_document_without_docling(temp_db_path):
|
async def test_client_import_document_stores_docling_data(temp_db_path):
|
||||||
"""Test that import_document without docling params does not store docling JSON."""
|
"""Test that import_document stores DoclingDocument data correctly."""
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
|
||||||
custom_chunks = [Chunk(content="Custom chunk", order=0)]
|
|
||||||
|
|
||||||
doc = await client.import_document(content="Test content", chunks=custom_chunks)
|
|
||||||
|
|
||||||
assert doc.id is not None
|
|
||||||
# When no docling params provided, they remain None
|
|
||||||
assert doc.docling_document_json is None
|
|
||||||
assert doc.docling_version is None
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_client_import_document_validates_docling_params(temp_db_path):
|
|
||||||
"""Test that import_document validates docling parameters."""
|
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
|
||||||
custom_chunks = [Chunk(content="Custom chunk", order=0)]
|
|
||||||
|
|
||||||
# Should fail if only one docling param is provided
|
|
||||||
with pytest.raises(ValueError, match="must both be provided"):
|
|
||||||
await client.import_document(
|
|
||||||
content="Test content",
|
|
||||||
chunks=custom_chunks,
|
|
||||||
docling_document_json='{"some": "json"}',
|
|
||||||
# Missing docling_version
|
|
||||||
)
|
|
||||||
|
|
||||||
with pytest.raises(ValueError, match="must both be provided"):
|
|
||||||
await client.import_document(
|
|
||||||
content="Test content",
|
|
||||||
chunks=custom_chunks,
|
|
||||||
docling_version="1.0.0",
|
|
||||||
# Missing docling_document_json
|
|
||||||
)
|
|
||||||
|
|
||||||
# Should fail with invalid JSON
|
|
||||||
with pytest.raises(ValueError, match="Invalid docling_document_json"):
|
|
||||||
await client.import_document(
|
|
||||||
content="Test content",
|
|
||||||
chunks=custom_chunks,
|
|
||||||
docling_document_json='{"invalid": "not a docling document"}',
|
|
||||||
docling_version="1.0.0",
|
|
||||||
)
|
|
||||||
|
|
||||||
# Should fail if neither content nor docling_document_json provided
|
|
||||||
with pytest.raises(ValueError, match="Either content or docling_document_json"):
|
|
||||||
await client.import_document(chunks=custom_chunks)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_client_import_document_extracts_content_from_docling(temp_db_path):
|
|
||||||
"""Test that import_document extracts content from DoclingDocument when not provided."""
|
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
from docling_core.types.doc.labels import DocItemLabel
|
from docling_core.types.doc.labels import DocItemLabel
|
||||||
|
|
||||||
|
|
@ -862,16 +818,16 @@ async def test_client_import_document_extracts_content_from_docling(temp_db_path
|
||||||
|
|
||||||
custom_chunks = [Chunk(content="Chunk content", order=0)]
|
custom_chunks = [Chunk(content="Chunk content", order=0)]
|
||||||
|
|
||||||
# Import without content - should extract from docling
|
# Import with DoclingDocument
|
||||||
doc = await client.import_document(
|
doc = await client.import_document(
|
||||||
|
docling_document=docling_doc,
|
||||||
chunks=custom_chunks,
|
chunks=custom_chunks,
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
|
||||||
docling_version=docling_doc.version,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
assert doc.id is not None
|
assert doc.id is not None
|
||||||
assert "Content from docling document" in doc.content
|
assert "Content from docling document" in doc.content
|
||||||
assert doc.docling_document_json == docling_doc.model_dump_json()
|
assert doc.docling_document_json == docling_doc.model_dump_json()
|
||||||
|
assert doc.docling_version == docling_doc.version
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
|
|
@ -941,16 +897,15 @@ async def test_client_update_document_with_custom_chunks_no_docling_json(
|
||||||
async def test_client_update_document_content_docling_mutually_exclusive(
|
async def test_client_update_document_content_docling_mutually_exclusive(
|
||||||
temp_db_path,
|
temp_db_path,
|
||||||
):
|
):
|
||||||
"""Test that content and docling_document_json cannot both be provided."""
|
"""Test that content and docling_document cannot both be provided."""
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
from docling_core.types.doc.labels import DocItemLabel
|
||||||
|
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
doc = await client.create_document(content="Initial content")
|
doc = await client.create_document(content="Initial content")
|
||||||
assert doc.id is not None
|
assert doc.id is not None
|
||||||
|
|
||||||
# Create a docling document
|
# Create a docling document
|
||||||
from docling_core.types.doc.labels import DocItemLabel
|
|
||||||
|
|
||||||
docling_doc = DoclingDocument(name="test")
|
docling_doc = DoclingDocument(name="test")
|
||||||
docling_doc.add_text(label=DocItemLabel.TEXT, text="Some text")
|
docling_doc.add_text(label=DocItemLabel.TEXT, text="Some text")
|
||||||
|
|
||||||
|
|
@ -958,15 +913,15 @@ async def test_client_update_document_content_docling_mutually_exclusive(
|
||||||
await client.update_document(
|
await client.update_document(
|
||||||
document_id=doc.id,
|
document_id=doc.id,
|
||||||
content="New content",
|
content="New content",
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
docling_document=docling_doc,
|
||||||
docling_version=docling_doc.version,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_client_update_document_with_docling_rechunks(temp_db_path):
|
async def test_client_update_document_with_docling_rechunks(temp_db_path):
|
||||||
"""Test that providing docling_document_json without chunks triggers rechunk."""
|
"""Test that providing docling_document without chunks triggers rechunk."""
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
from docling_core.types.doc.labels import DocItemLabel
|
||||||
|
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
# Create initial document
|
# Create initial document
|
||||||
|
|
@ -975,8 +930,6 @@ async def test_client_update_document_with_docling_rechunks(temp_db_path):
|
||||||
original_chunks = await client.chunk_repository.get_by_document_id(doc.id)
|
original_chunks = await client.chunk_repository.get_by_document_id(doc.id)
|
||||||
|
|
||||||
# Create a new docling document with different content
|
# Create a new docling document with different content
|
||||||
from docling_core.types.doc.labels import DocItemLabel
|
|
||||||
|
|
||||||
docling_doc = DoclingDocument(name="updated")
|
docling_doc = DoclingDocument(name="updated")
|
||||||
docling_doc.add_text(
|
docling_doc.add_text(
|
||||||
label=DocItemLabel.TEXT,
|
label=DocItemLabel.TEXT,
|
||||||
|
|
@ -986,8 +939,7 @@ async def test_client_update_document_with_docling_rechunks(temp_db_path):
|
||||||
# Update with docling document only - should rechunk from it
|
# Update with docling document only - should rechunk from it
|
||||||
updated_doc = await client.update_document(
|
updated_doc = await client.update_document(
|
||||||
document_id=doc.id,
|
document_id=doc.id,
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
docling_document=docling_doc,
|
||||||
docling_version=docling_doc.version,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Content should be extracted from docling document
|
# Content should be extracted from docling document
|
||||||
|
|
@ -1004,8 +956,9 @@ async def test_client_update_document_with_docling_rechunks(temp_db_path):
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_client_update_document_docling_with_chunks(temp_db_path):
|
async def test_client_update_document_docling_with_chunks(temp_db_path):
|
||||||
"""Test that providing both docling_document_json and chunks stores both."""
|
"""Test that providing both docling_document and chunks stores both."""
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
from docling_core.types.doc.labels import DocItemLabel
|
||||||
|
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
# Create initial document
|
# Create initial document
|
||||||
|
|
@ -1013,8 +966,6 @@ async def test_client_update_document_docling_with_chunks(temp_db_path):
|
||||||
assert doc.id is not None
|
assert doc.id is not None
|
||||||
|
|
||||||
# Create a docling document
|
# Create a docling document
|
||||||
from docling_core.types.doc.labels import DocItemLabel
|
|
||||||
|
|
||||||
docling_doc = DoclingDocument(name="custom")
|
docling_doc = DoclingDocument(name="custom")
|
||||||
docling_doc.add_text(label=DocItemLabel.TEXT, text="Text from docling")
|
docling_doc.add_text(label=DocItemLabel.TEXT, text="Text from docling")
|
||||||
|
|
||||||
|
|
@ -1027,8 +978,7 @@ async def test_client_update_document_docling_with_chunks(temp_db_path):
|
||||||
updated_doc = await client.update_document(
|
updated_doc = await client.update_document(
|
||||||
document_id=doc.id,
|
document_id=doc.id,
|
||||||
chunks=custom_chunks,
|
chunks=custom_chunks,
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
docling_document=docling_doc,
|
||||||
docling_version=docling_doc.version,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Content should be extracted from docling (since content wasn't provided)
|
# Content should be extracted from docling (since content wasn't provided)
|
||||||
|
|
@ -1079,22 +1029,6 @@ async def test_client_visualize_chunk_no_document(temp_db_path):
|
||||||
assert images == []
|
assert images == []
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_client_visualize_chunk_no_docling_document(temp_db_path):
|
|
||||||
"""Test visualize_chunk returns empty list when document has no DoclingDocument."""
|
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
|
||||||
# Import document with custom chunks (no DoclingDocument)
|
|
||||||
custom_chunks = [Chunk(content="Custom chunk", order=0)]
|
|
||||||
doc = await client.import_document(content="Test content", chunks=custom_chunks)
|
|
||||||
|
|
||||||
assert doc.id is not None
|
|
||||||
chunks = await client.chunk_repository.get_by_document_id(doc.id)
|
|
||||||
assert len(chunks) == 1
|
|
||||||
|
|
||||||
images = await client.visualize_chunk(chunks[0])
|
|
||||||
assert images == []
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_client_visualize_chunk_no_bounding_boxes(temp_db_path):
|
async def test_client_visualize_chunk_no_bounding_boxes(temp_db_path):
|
||||||
"""Test visualize_chunk returns empty list when chunk has no bounding boxes."""
|
"""Test visualize_chunk returns empty list when chunk has no bounding boxes."""
|
||||||
|
|
@ -1330,7 +1264,16 @@ async def test_client_chunk_empty_document(temp_db_path):
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_import_document_embeds_chunks_without_embeddings(temp_db_path):
|
async def test_import_document_embeds_chunks_without_embeddings(temp_db_path):
|
||||||
"""Test that import_document embeds chunks that don't have embeddings."""
|
"""Test that import_document embeds chunks that don't have embeddings."""
|
||||||
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
from docling_core.types.doc.labels import DocItemLabel
|
||||||
|
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
|
# Create a DoclingDocument
|
||||||
|
docling_doc = DoclingDocument(name="test")
|
||||||
|
docling_doc.add_text(
|
||||||
|
label=DocItemLabel.TEXT, text="Document with unembedded chunks"
|
||||||
|
)
|
||||||
|
|
||||||
# Create chunks without embeddings
|
# Create chunks without embeddings
|
||||||
chunks = [
|
chunks = [
|
||||||
Chunk(content="First chunk without embedding", order=0),
|
Chunk(content="First chunk without embedding", order=0),
|
||||||
|
|
@ -1339,7 +1282,7 @@ async def test_import_document_embeds_chunks_without_embeddings(temp_db_path):
|
||||||
|
|
||||||
# Import document with chunks that have no embeddings
|
# Import document with chunks that have no embeddings
|
||||||
doc = await client.import_document(
|
doc = await client.import_document(
|
||||||
content="Document with unembedded chunks",
|
docling_document=docling_doc,
|
||||||
chunks=chunks,
|
chunks=chunks,
|
||||||
)
|
)
|
||||||
assert doc.id is not None
|
assert doc.id is not None
|
||||||
|
|
|
||||||
|
|
@ -15,10 +15,9 @@ async def create_document_with_docling(
|
||||||
chunks = await client.chunk(docling_doc)
|
chunks = await client.chunk(docling_doc)
|
||||||
embedded_chunks = await client._ensure_chunks_embedded(chunks)
|
embedded_chunks = await client._ensure_chunks_embedded(chunks)
|
||||||
return await client.import_document(
|
return await client.import_document(
|
||||||
|
docling_document=docling_doc,
|
||||||
chunks=embedded_chunks,
|
chunks=embedded_chunks,
|
||||||
title=title,
|
title=title,
|
||||||
docling_document_json=docling_doc.model_dump_json(),
|
|
||||||
docling_version=docling_doc.version,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -415,22 +414,26 @@ async def test_expand_context_multiple_documents(temp_db_path):
|
||||||
|
|
||||||
async with HaikuRAG(temp_db_path, config=config, create=True) as client:
|
async with HaikuRAG(temp_db_path, config=config, create=True) as client:
|
||||||
# Create first document with manual chunks
|
# Create first document with manual chunks
|
||||||
|
docling_doc1 = DoclingDocument(name="doc1")
|
||||||
|
docling_doc1.add_text(label=DocItemLabel.TEXT, text="Doc1 content")
|
||||||
doc1_chunks = [
|
doc1_chunks = [
|
||||||
Chunk(content="Doc1 Part A", order=0),
|
Chunk(content="Doc1 Part A", order=0),
|
||||||
Chunk(content="Doc1 Part B", order=1),
|
Chunk(content="Doc1 Part B", order=1),
|
||||||
Chunk(content="Doc1 Part C", order=2),
|
Chunk(content="Doc1 Part C", order=2),
|
||||||
]
|
]
|
||||||
doc1 = await client.import_document(
|
doc1 = await client.import_document(
|
||||||
content="Doc1 content", chunks=doc1_chunks, uri="doc1.txt"
|
docling_document=docling_doc1, chunks=doc1_chunks, uri="doc1.txt"
|
||||||
)
|
)
|
||||||
|
|
||||||
# Create second document with manual chunks
|
# Create second document with manual chunks
|
||||||
|
docling_doc2 = DoclingDocument(name="doc2")
|
||||||
|
docling_doc2.add_text(label=DocItemLabel.TEXT, text="Doc2 content")
|
||||||
doc2_chunks = [
|
doc2_chunks = [
|
||||||
Chunk(content="Doc2 Section X", order=0),
|
Chunk(content="Doc2 Section X", order=0),
|
||||||
Chunk(content="Doc2 Section Y", order=1),
|
Chunk(content="Doc2 Section Y", order=1),
|
||||||
]
|
]
|
||||||
doc2 = await client.import_document(
|
doc2 = await client.import_document(
|
||||||
content="Doc2 content", chunks=doc2_chunks, uri="doc2.txt"
|
docling_document=docling_doc2, chunks=doc2_chunks, uri="doc2.txt"
|
||||||
)
|
)
|
||||||
|
|
||||||
assert doc1.id is not None
|
assert doc1.id is not None
|
||||||
|
|
@ -472,6 +475,8 @@ async def test_expand_context_merges_overlapping_chunks(temp_db_path):
|
||||||
|
|
||||||
async with HaikuRAG(temp_db_path, config=config, create=True) as client:
|
async with HaikuRAG(temp_db_path, config=config, create=True) as client:
|
||||||
# Create document with 5 chunks
|
# Create document with 5 chunks
|
||||||
|
docling_doc = DoclingDocument(name="test")
|
||||||
|
docling_doc.add_text(label=DocItemLabel.TEXT, text="Full document content")
|
||||||
manual_chunks = [
|
manual_chunks = [
|
||||||
Chunk(content="Chunk 0", order=0),
|
Chunk(content="Chunk 0", order=0),
|
||||||
Chunk(content="Chunk 1", order=1),
|
Chunk(content="Chunk 1", order=1),
|
||||||
|
|
@ -481,7 +486,7 @@ async def test_expand_context_merges_overlapping_chunks(temp_db_path):
|
||||||
]
|
]
|
||||||
|
|
||||||
doc = await client.import_document(
|
doc = await client.import_document(
|
||||||
content="Full document content", chunks=manual_chunks
|
docling_document=docling_doc, chunks=manual_chunks
|
||||||
)
|
)
|
||||||
|
|
||||||
assert doc.id is not None
|
assert doc.id is not None
|
||||||
|
|
@ -525,6 +530,8 @@ async def test_expand_context_keeps_separate_non_overlapping(temp_db_path):
|
||||||
|
|
||||||
async with HaikuRAG(temp_db_path, config=config, create=True) as client:
|
async with HaikuRAG(temp_db_path, config=config, create=True) as client:
|
||||||
# Create document with chunks far apart
|
# Create document with chunks far apart
|
||||||
|
docling_doc = DoclingDocument(name="test")
|
||||||
|
docling_doc.add_text(label=DocItemLabel.TEXT, text="Full document content")
|
||||||
manual_chunks = [
|
manual_chunks = [
|
||||||
Chunk(content="Chunk 0", order=0),
|
Chunk(content="Chunk 0", order=0),
|
||||||
Chunk(content="Chunk 1", order=1),
|
Chunk(content="Chunk 1", order=1),
|
||||||
|
|
@ -535,7 +542,7 @@ async def test_expand_context_keeps_separate_non_overlapping(temp_db_path):
|
||||||
]
|
]
|
||||||
|
|
||||||
doc = await client.import_document(
|
doc = await client.import_document(
|
||||||
content="Full document content", chunks=manual_chunks
|
docling_document=docling_doc, chunks=manual_chunks
|
||||||
)
|
)
|
||||||
|
|
||||||
assert doc.id is not None
|
assert doc.id is not None
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue