Merge pull request #146 from ggozad/feat/update-document-chunks

New `update_document_fields()` method for partial document updates
This commit is contained in:
Yiorgis Gozadinos 2025-11-19 11:38:54 +02:00 committed by GitHub
commit 1ecc7152d3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 181 additions and 0 deletions

View file

@ -1,6 +1,12 @@
# Changelog
## [Unreleased]
### Added
- **Document Update API**: New `update_document_fields()` method for partial document updates
- Update individual fields (content, metadata, title, chunks) without fetching full document
- Support for custom chunks or auto-generation from content
### Changed
- Updated core dependencies:

View file

@ -101,11 +101,53 @@ docs = await client.list_documents(
### Updating Documents
Update entire document:
```python
doc.content = "Updated content"
await client.update_document(doc)
```
Update specific fields:
```python
# Update only content (triggers re-chunking)
await client.update_document_fields(
document_id=doc.id,
content="New content"
)
# Update only metadata (no re-chunking)
await client.update_document_fields(
document_id=doc.id,
metadata={"version": "2.0", "updated_by": "admin"}
)
# Update only title (no re-chunking)
await client.update_document_fields(
document_id=doc.id,
title="New Title"
)
# Update multiple fields at once
await client.update_document_fields(
document_id=doc.id,
content="New content",
title="Updated Title",
metadata={"status": "final"}
)
# Use custom chunks instead of auto-generation
custom_chunks = [
Chunk(content="Custom chunk 1"),
Chunk(content="Custom chunk 2"),
]
await client.update_document_fields(
document_id=doc.id,
chunks=custom_chunks
)
```
**Performance Note:** Updates to only `metadata` or `title` skip re-chunking for efficiency. Updates to `content` or `chunks` will regenerate or replace the document's chunks.
### Deleting Documents
```python

View file

@ -417,6 +417,68 @@ class HaikuRAG:
document, docling_document
)
async def update_document_fields(
self,
document_id: str,
content: str | None = None,
metadata: dict | None = None,
chunks: list[Chunk] | None = None,
title: str | None = None,
) -> Document:
"""Update specific fields of a document by ID.
Args:
document_id: The ID of the document to update
content: New content for the document
metadata: New metadata for the document
chunks: Custom chunks to use instead of auto-generating
title: New title for the document
Returns:
The updated Document instance.
"""
# Fetch the existing document
existing_doc = await self.get_document_by_id(document_id)
if existing_doc is None:
raise ValueError(f"Document with ID {document_id} not found")
# Update only the provided fields
if content is not None:
existing_doc.content = content
if title is not None:
existing_doc.title = title
if metadata is not None:
existing_doc.metadata = metadata
# Determine if we need to rechunk
if content is not None or chunks is not None:
# Content changed or custom chunks provided - need to rechunk
if chunks is not None:
# Use custom chunks
# Delete existing chunks
await self.chunk_repository.delete_by_document_id(document_id)
# Update document metadata
await self.document_repository.update(existing_doc)
# Add new chunks
for order, chunk in enumerate(chunks):
chunk.document_id = document_id
chunk.order = order
await self.chunk_repository.create(chunk)
return existing_doc
else:
# Auto-generate chunks from content
converter = get_converter(self._config)
docling_document = converter.convert_text(existing_doc.content)
return await self.document_repository._update_and_rechunk(
existing_doc, docling_document
)
else:
# Only metadata/title changed - no rechunking needed
return await self.document_repository.update(existing_doc)
async def delete_document(self, document_id: str) -> bool:
"""Delete a document by its ID."""
return await self.document_repository.delete(document_id)

View file

@ -78,6 +78,77 @@ async def test_client_document_crud(qa_corpus: Dataset, temp_db_path):
assert deleted_again is False
@pytest.mark.asyncio
async def test_client_update_document_fields(qa_corpus: Dataset, temp_db_path):
"""Test updating document with individual parameters."""
async with HaikuRAG(temp_db_path) as client:
# Get test data
first_doc = qa_corpus[0]
document_text = first_doc["document_extracted"]
test_uri = "file:///path/to/test.txt"
test_metadata = {"source": "test", "topic": "testing"}
# Create a document
created_doc = await client.create_document(
content=document_text,
uri=test_uri,
title="Original Title",
metadata=test_metadata,
)
assert created_doc.id is not None
original_id = created_doc.id
# Test updating only content
updated_doc = await client.update_document_fields(
document_id=original_id, content="Updated content only"
)
assert updated_doc.id == original_id
assert updated_doc.content == "Updated content only"
assert updated_doc.title == "Original Title"
assert updated_doc.uri == test_uri
# Test updating only metadata
new_metadata = {"source": "updated", "version": "2.0"}
updated_doc = await client.update_document_fields(
document_id=original_id, metadata=new_metadata
)
assert updated_doc.metadata == new_metadata
assert (
updated_doc.content == "Updated content only"
) # Should keep previous update
# Test updating only title
updated_doc = await client.update_document_fields(
document_id=original_id, title="New Title"
)
assert updated_doc.title == "New Title"
assert updated_doc.content == "Updated content only"
assert updated_doc.metadata == new_metadata
# Test updating multiple fields at once
custom_chunks = [
Chunk(content="Custom chunk 1", order=0),
Chunk(content="Custom chunk 2", order=1),
]
updated_doc = await client.update_document_fields(
document_id=original_id,
content="Content with custom chunks",
title="Final Title",
metadata={"final": "true"},
chunks=custom_chunks,
)
assert updated_doc.id == original_id
assert updated_doc.content == "Content with custom chunks"
assert updated_doc.title == "Final Title"
assert updated_doc.metadata == {"final": "true"}
# Verify the custom chunks were created
doc_chunks = await client.chunk_repository.get_by_document_id(original_id)
assert len(doc_chunks) == 2
assert doc_chunks[0].content == "Custom chunk 1"
assert doc_chunks[1].content == "Custom chunk 2"
@pytest.mark.asyncio
async def test_client_create_document_from_source(temp_db_path):
"""Test creating a document from a file source."""