Introduce import_document(), remove chunks from create_document()
This commit is contained in:
parent
fe551066d8
commit
8e30f67266
5 changed files with 181 additions and 75 deletions
|
|
@ -29,9 +29,17 @@
|
||||||
- **Visual Grounding CLI**: New `haiku-rag visualize <chunk_id>` command
|
- **Visual Grounding CLI**: New `haiku-rag visualize <chunk_id>` command
|
||||||
- Displays page images with highlighted bounding boxes for a chunk
|
- Displays page images with highlighted bounding boxes for a chunk
|
||||||
- Requires terminal with image support (iTerm2, Kitty, etc.)
|
- Requires terminal with image support (iTerm2, Kitty, etc.)
|
||||||
|
- **New `import_document()` Method**: Import pre-processed documents with custom chunks
|
||||||
|
- Use when document conversion, chunking, and embedding were done externally
|
||||||
|
- Accepts `content`, `chunks` (required), and optional `docling_document_json`/`docling_version`
|
||||||
|
- Validates that if docling parameters are provided, both must be present
|
||||||
|
- Validates docling JSON parses correctly if provided
|
||||||
|
|
||||||
### Changed
|
### Changed
|
||||||
|
|
||||||
|
- **BREAKING: `create_document()` API**: Removed `chunks` parameter
|
||||||
|
- `create_document()` now always processes content (converts, chunks, embeds)
|
||||||
|
- Use new `import_document()` for pre-processed documents with custom chunks
|
||||||
- **BREAKING: Chunker Interface**: `DocumentChunker.chunk()` now returns `list[ChunkWithMetadata]` instead of `list[str]`
|
- **BREAKING: Chunker Interface**: `DocumentChunker.chunk()` now returns `list[ChunkWithMetadata]` instead of `list[str]`
|
||||||
- `ChunkWithMetadata` combines chunk text with `ChunkMetadata` (refs, labels, headings, page_numbers)
|
- `ChunkWithMetadata` combines chunk text with `ChunkMetadata` (refs, labels, headings, page_numbers)
|
||||||
- All chunker implementations updated: `DoclingLocalChunker`, `DoclingServeChunker`
|
- All chunker implementations updated: `DoclingLocalChunker`, `DoclingServeChunker`
|
||||||
|
|
|
||||||
|
|
@ -36,31 +36,6 @@ doc = await client.create_document(
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
With custom externally generated chunks:
|
|
||||||
```python
|
|
||||||
from haiku.rag.store.models.chunk import Chunk
|
|
||||||
|
|
||||||
# Create custom chunks with optional embeddings
|
|
||||||
chunks = [
|
|
||||||
Chunk(
|
|
||||||
content="This is the first chunk",
|
|
||||||
metadata={"section": "intro"}
|
|
||||||
),
|
|
||||||
Chunk(
|
|
||||||
content="This is the second chunk",
|
|
||||||
metadata={"section": "body"},
|
|
||||||
embedding=[0.1] * 1024 # Optional pre-computed embedding
|
|
||||||
),
|
|
||||||
]
|
|
||||||
|
|
||||||
doc = await client.create_document(
|
|
||||||
content="Full document content",
|
|
||||||
uri="doc://custom",
|
|
||||||
metadata={"source": "manual"},
|
|
||||||
chunks=chunks # Use provided chunks instead of auto-generating
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
From file:
|
From file:
|
||||||
```python
|
```python
|
||||||
doc = await client.create_document_from_source(
|
doc = await client.create_document_from_source(
|
||||||
|
|
@ -75,6 +50,53 @@ doc = await client.create_document_from_source(
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Importing Pre-Processed Documents
|
||||||
|
|
||||||
|
If you process documents externally (conversion, chunking, embedding), use `import_document()` to store them:
|
||||||
|
|
||||||
|
```python
|
||||||
|
from haiku.rag.store.models.chunk import Chunk
|
||||||
|
|
||||||
|
# Create chunks with optional embeddings
|
||||||
|
chunks = [
|
||||||
|
Chunk(
|
||||||
|
content="This is the first chunk",
|
||||||
|
metadata={"section": "intro"},
|
||||||
|
order=0,
|
||||||
|
),
|
||||||
|
Chunk(
|
||||||
|
content="This is the second chunk",
|
||||||
|
metadata={"section": "body"},
|
||||||
|
embedding=[0.1] * 1024, # Pre-computed embedding
|
||||||
|
order=1,
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
# Import document with custom chunks
|
||||||
|
doc = await client.import_document(
|
||||||
|
content="Full document content",
|
||||||
|
chunks=chunks,
|
||||||
|
uri="doc://custom",
|
||||||
|
title="Custom Document",
|
||||||
|
metadata={"source": "external-pipeline"},
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
If you also have a DoclingDocument from your processing pipeline, include it for visual grounding support:
|
||||||
|
|
||||||
|
```python
|
||||||
|
doc = await client.import_document(
|
||||||
|
content="Full document content",
|
||||||
|
chunks=chunks,
|
||||||
|
uri="doc://custom",
|
||||||
|
docling_document_json=docling_doc.model_dump_json(),
|
||||||
|
docling_version=docling_doc.version,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
!!! note
|
||||||
|
When providing `docling_document_json`, you must also provide `docling_version`. The JSON is validated to ensure it's a valid DoclingDocument.
|
||||||
|
|
||||||
### Retrieving Documents
|
### Retrieving Documents
|
||||||
|
|
||||||
By ID:
|
By ID:
|
||||||
|
|
|
||||||
|
|
@ -118,49 +118,92 @@ class HaikuRAG:
|
||||||
uri: str | None = None,
|
uri: str | None = None,
|
||||||
title: str | None = None,
|
title: str | None = None,
|
||||||
metadata: dict | None = None,
|
metadata: dict | None = None,
|
||||||
chunks: list[Chunk] | None = None,
|
|
||||||
) -> Document:
|
) -> Document:
|
||||||
"""Create a new document with optional URI and metadata.
|
"""Create a new document from text content.
|
||||||
|
|
||||||
|
Converts the content, chunks it, and generates embeddings.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
content: The text content of the document.
|
content: The text content of the document.
|
||||||
uri: Optional URI identifier for the document.
|
uri: Optional URI identifier for the document.
|
||||||
|
title: Optional title for the document.
|
||||||
metadata: Optional metadata dictionary.
|
metadata: Optional metadata dictionary.
|
||||||
chunks: Optional list of pre-created chunks to use instead of generating new ones.
|
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
The created Document instance.
|
The created Document instance.
|
||||||
"""
|
"""
|
||||||
# Only create docling_document if we need to generate chunks
|
converter = get_converter(self._config)
|
||||||
if chunks is None:
|
docling_document = await converter.convert_text(content)
|
||||||
# Use converter to convert text
|
|
||||||
converter = get_converter(self._config)
|
|
||||||
docling_document = await converter.convert_text(content)
|
|
||||||
|
|
||||||
document = Document(
|
document = Document(
|
||||||
content=content,
|
content=content,
|
||||||
uri=uri,
|
uri=uri,
|
||||||
title=title,
|
title=title,
|
||||||
metadata=metadata or {},
|
metadata=metadata or {},
|
||||||
docling_document_json=docling_document.model_dump_json(),
|
docling_document_json=docling_document.model_dump_json(),
|
||||||
docling_version=docling_document.version,
|
docling_version=docling_document.version,
|
||||||
|
)
|
||||||
|
|
||||||
|
return await self.document_repository._create_and_chunk(
|
||||||
|
document, docling_document, None
|
||||||
|
)
|
||||||
|
|
||||||
|
async def import_document(
|
||||||
|
self,
|
||||||
|
content: str,
|
||||||
|
chunks: list[Chunk],
|
||||||
|
uri: str | None = None,
|
||||||
|
title: str | None = None,
|
||||||
|
metadata: dict | None = None,
|
||||||
|
docling_document_json: str | None = None,
|
||||||
|
docling_version: str | None = None,
|
||||||
|
) -> Document:
|
||||||
|
"""Import a pre-processed document with chunks.
|
||||||
|
|
||||||
|
Use this when document conversion, chunking, and embedding were done
|
||||||
|
externally and you want to store the results in haiku.rag.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content: The document content.
|
||||||
|
chunks: Pre-created chunks (must include embeddings).
|
||||||
|
uri: Optional URI identifier for the document.
|
||||||
|
title: Optional title for the document.
|
||||||
|
metadata: Optional metadata dictionary.
|
||||||
|
docling_document_json: Optional serialized DoclingDocument JSON.
|
||||||
|
docling_version: Optional DoclingDocument schema version.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The created Document instance.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If docling_document_json is provided without docling_version
|
||||||
|
or vice versa, or if the JSON is invalid.
|
||||||
|
"""
|
||||||
|
# Validate docling parameters
|
||||||
|
if (docling_document_json is None) != (docling_version is None):
|
||||||
|
raise ValueError(
|
||||||
|
"docling_document_json and docling_version must both be provided or both be None"
|
||||||
)
|
)
|
||||||
|
|
||||||
return await self.document_repository._create_and_chunk(
|
# Validate docling JSON parses if provided
|
||||||
document, docling_document, chunks
|
if docling_document_json is not None:
|
||||||
)
|
try:
|
||||||
else:
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
# Chunks already provided, no conversion needed
|
|
||||||
document = Document(
|
|
||||||
content=content,
|
|
||||||
uri=uri,
|
|
||||||
title=title,
|
|
||||||
metadata=metadata or {},
|
|
||||||
)
|
|
||||||
|
|
||||||
return await self.document_repository._create_and_chunk(
|
DoclingDocument.model_validate_json(docling_document_json)
|
||||||
document, None, chunks
|
except Exception as e:
|
||||||
)
|
raise ValueError(f"Invalid docling_document_json: {e}") from e
|
||||||
|
|
||||||
|
document = Document(
|
||||||
|
content=content,
|
||||||
|
uri=uri,
|
||||||
|
title=title,
|
||||||
|
metadata=metadata or {},
|
||||||
|
docling_document_json=docling_document_json,
|
||||||
|
docling_version=docling_version,
|
||||||
|
)
|
||||||
|
|
||||||
|
return await self.document_repository._create_and_chunk(document, None, chunks)
|
||||||
|
|
||||||
async def create_document_from_source(
|
async def create_document_from_source(
|
||||||
self, source: str | Path, title: str | None = None, metadata: dict | None = None
|
self, source: str | Path, title: str | None = None, metadata: dict | None = None
|
||||||
|
|
|
||||||
|
|
@ -686,8 +686,8 @@ async def test_client_async_context_manager(temp_db_path):
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_client_create_document_with_custom_chunks(temp_db_path):
|
async def test_client_import_document_with_custom_chunks(temp_db_path):
|
||||||
"""Test creating a document with pre-created chunks."""
|
"""Test importing a document with pre-created chunks."""
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
# Create some custom chunks with and without embeddings
|
# Create some custom chunks with and without embeddings
|
||||||
chunks = [
|
chunks = [
|
||||||
|
|
@ -709,8 +709,8 @@ async def test_client_create_document_with_custom_chunks(temp_db_path):
|
||||||
),
|
),
|
||||||
]
|
]
|
||||||
|
|
||||||
# Create document with custom chunks
|
# Import document with custom chunks
|
||||||
document = await client.create_document(
|
document = await client.import_document(
|
||||||
content="Full document content", chunks=chunks
|
content="Full document content", chunks=chunks
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -775,11 +775,11 @@ async def test_client_expand_context(temp_db_path):
|
||||||
Chunk(content="Chunk 4 content", order=4, embedding=z),
|
Chunk(content="Chunk 4 content", order=4, embedding=z),
|
||||||
]
|
]
|
||||||
|
|
||||||
doc = await client.create_document(
|
doc = await client.import_document(
|
||||||
content="Full document content",
|
content="Full document content",
|
||||||
|
chunks=manual_chunks,
|
||||||
uri="test_doc.txt",
|
uri="test_doc.txt",
|
||||||
title="test_doc_title",
|
title="test_doc_title",
|
||||||
chunks=manual_chunks,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Get all chunks for the document
|
# Get all chunks for the document
|
||||||
|
|
@ -844,8 +844,8 @@ async def test_client_expand_context_multiple_chunks(temp_db_path):
|
||||||
Chunk(content="Doc1 Part B", order=1),
|
Chunk(content="Doc1 Part B", order=1),
|
||||||
Chunk(content="Doc1 Part C", order=2),
|
Chunk(content="Doc1 Part C", order=2),
|
||||||
]
|
]
|
||||||
doc1 = await client.create_document(
|
doc1 = await client.import_document(
|
||||||
content="Doc1 content", uri="doc1.txt", chunks=doc1_chunks
|
content="Doc1 content", chunks=doc1_chunks, uri="doc1.txt"
|
||||||
)
|
)
|
||||||
|
|
||||||
# Create second document with manual chunks
|
# Create second document with manual chunks
|
||||||
|
|
@ -853,8 +853,8 @@ async def test_client_expand_context_multiple_chunks(temp_db_path):
|
||||||
Chunk(content="Doc2 Section X", order=0),
|
Chunk(content="Doc2 Section X", order=0),
|
||||||
Chunk(content="Doc2 Section Y", order=1),
|
Chunk(content="Doc2 Section Y", order=1),
|
||||||
]
|
]
|
||||||
doc2 = await client.create_document(
|
doc2 = await client.import_document(
|
||||||
content="Doc2 content", uri="doc2.txt", chunks=doc2_chunks
|
content="Doc2 content", chunks=doc2_chunks, uri="doc2.txt"
|
||||||
)
|
)
|
||||||
|
|
||||||
assert doc1.id is not None
|
assert doc1.id is not None
|
||||||
|
|
@ -903,7 +903,7 @@ async def test_client_expand_context_merges_overlapping_chunks(temp_db_path):
|
||||||
Chunk(content="Chunk 4", order=4),
|
Chunk(content="Chunk 4", order=4),
|
||||||
]
|
]
|
||||||
|
|
||||||
doc = await client.create_document(
|
doc = await client.import_document(
|
||||||
content="Full document content", chunks=manual_chunks
|
content="Full document content", chunks=manual_chunks
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -956,7 +956,7 @@ async def test_client_expand_context_keeps_separate_non_overlapping(temp_db_path
|
||||||
Chunk(content="Chunk 7", order=7),
|
Chunk(content="Chunk 7", order=7),
|
||||||
]
|
]
|
||||||
|
|
||||||
doc = await client.create_document(
|
doc = await client.import_document(
|
||||||
content="Full document content", chunks=manual_chunks
|
content="Full document content", chunks=manual_chunks
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -1161,19 +1161,52 @@ async def test_client_create_document_stores_docling_json(temp_db_path):
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_client_create_document_with_custom_chunks_no_docling_json(temp_db_path):
|
async def test_client_import_document_without_docling(temp_db_path):
|
||||||
"""Test that create_document with custom chunks does not store docling JSON."""
|
"""Test that import_document without docling params does not store docling JSON."""
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
custom_chunks = [Chunk(content="Custom chunk", order=0)]
|
custom_chunks = [Chunk(content="Custom chunk", order=0)]
|
||||||
|
|
||||||
doc = await client.create_document(content="Test content", chunks=custom_chunks)
|
doc = await client.import_document(content="Test content", chunks=custom_chunks)
|
||||||
|
|
||||||
assert doc.id is not None
|
assert doc.id is not None
|
||||||
# When custom chunks are provided, no conversion happens
|
# When no docling params provided, they remain None
|
||||||
assert doc.docling_document_json is None
|
assert doc.docling_document_json is None
|
||||||
assert doc.docling_version is None
|
assert doc.docling_version is None
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_client_import_document_validates_docling_params(temp_db_path):
|
||||||
|
"""Test that import_document validates docling parameters."""
|
||||||
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
|
custom_chunks = [Chunk(content="Custom chunk", order=0)]
|
||||||
|
|
||||||
|
# Should fail if only one docling param is provided
|
||||||
|
with pytest.raises(ValueError, match="must both be provided"):
|
||||||
|
await client.import_document(
|
||||||
|
content="Test content",
|
||||||
|
chunks=custom_chunks,
|
||||||
|
docling_document_json='{"some": "json"}',
|
||||||
|
# Missing docling_version
|
||||||
|
)
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="must both be provided"):
|
||||||
|
await client.import_document(
|
||||||
|
content="Test content",
|
||||||
|
chunks=custom_chunks,
|
||||||
|
docling_version="1.0.0",
|
||||||
|
# Missing docling_document_json
|
||||||
|
)
|
||||||
|
|
||||||
|
# Should fail with invalid JSON
|
||||||
|
with pytest.raises(ValueError, match="Invalid docling_document_json"):
|
||||||
|
await client.import_document(
|
||||||
|
content="Test content",
|
||||||
|
chunks=custom_chunks,
|
||||||
|
docling_document_json='{"invalid": "not a docling document"}',
|
||||||
|
docling_version="1.0.0",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_client_create_document_from_file_stores_docling_json(temp_db_path):
|
async def test_client_create_document_from_file_stores_docling_json(temp_db_path):
|
||||||
"""Test that create_document_from_source stores DoclingDocument JSON for files."""
|
"""Test that create_document_from_source stores DoclingDocument JSON for files."""
|
||||||
|
|
@ -1297,9 +1330,9 @@ async def test_client_visualize_chunk_no_document(temp_db_path):
|
||||||
async def test_client_visualize_chunk_no_docling_document(temp_db_path):
|
async def test_client_visualize_chunk_no_docling_document(temp_db_path):
|
||||||
"""Test visualize_chunk returns empty list when document has no DoclingDocument."""
|
"""Test visualize_chunk returns empty list when document has no DoclingDocument."""
|
||||||
async with HaikuRAG(temp_db_path, create=True) as client:
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
# Create document with custom chunks (no DoclingDocument)
|
# Import document with custom chunks (no DoclingDocument)
|
||||||
custom_chunks = [Chunk(content="Custom chunk", order=0)]
|
custom_chunks = [Chunk(content="Custom chunk", order=0)]
|
||||||
doc = await client.create_document(content="Test content", chunks=custom_chunks)
|
doc = await client.import_document(content="Test content", chunks=custom_chunks)
|
||||||
|
|
||||||
assert doc.id is not None
|
assert doc.id is not None
|
||||||
chunks = await client.chunk_repository.get_by_document_id(doc.id)
|
chunks = await client.chunk_repository.get_by_document_id(doc.id)
|
||||||
|
|
|
||||||
|
|
@ -221,14 +221,14 @@ async def test_search_graceful_degradation(temp_db_path):
|
||||||
|
|
||||||
client = HaikuRAG(db_path=temp_db_path, config=Config, create=True)
|
client = HaikuRAG(db_path=temp_db_path, config=Config, create=True)
|
||||||
|
|
||||||
# Create document with custom chunks (no docling document)
|
# Import document with custom chunks (no docling document)
|
||||||
custom_chunks = [
|
custom_chunks = [
|
||||||
Chunk(content="Custom chunk without docling metadata", metadata={}),
|
Chunk(content="Custom chunk without docling metadata", metadata={}),
|
||||||
]
|
]
|
||||||
await client.create_document(
|
await client.import_document(
|
||||||
content="Document with custom chunks",
|
content="Document with custom chunks",
|
||||||
uri="https://example.com/custom.html",
|
|
||||||
chunks=custom_chunks,
|
chunks=custom_chunks,
|
||||||
|
uri="https://example.com/custom.html",
|
||||||
)
|
)
|
||||||
|
|
||||||
results = await client.search("custom chunk", limit=3)
|
results = await client.search("custom chunk", limit=3)
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue