Merge pull request #551 from lawrenceakka/chunk-metadata

Expose all chunk metadata on search results and citations
This commit is contained in:
Yiorgis Gozadinos 2026-08-18 15:30:43 +03:00 committed by GitHub
commit d177b5883b
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
12 changed files with 253 additions and 4 deletions

View file

@ -5,6 +5,7 @@
- `evaluations run --filter/-f CLAUSE`: SQL `WHERE` clause over document columns, applied to the retrieval benchmark's searches and to every capability search during QA. Recorded as `document_filter` in experiment metadata.
- `mtrag_clapnq` / `mtrag_clapnq_rewrite` / `mtrag_clapnq_live` / `mtrag_clapnq_live_uncompacted` evaluation datasets: multi-turn QA with gold-prefix and live-session conversation replay, live arms with and without `EvidenceCompactionCapability`, Recall@k/nDCG@k retrieval metrics, eligibility-aware citation scoring, refusal precision/recall, per-turn tool-traffic attributes, and `citation_status` / `turn_citation_status` eval attributes.
- Raw chunk metadata is now exposed to search and citation results, through `SearchResult.chunk_meta` and `Citation.chunk_meta`. For context-expanded results, the metadata is that of the anchor chunk.
- BTree indexes on `chunks.id`, `chunks.document_id` and `documents.id`, and a Bitmap index on `document_items.label`. Existing databases need `haiku-rag migrate`.
- `lancedb.read_consistency_interval_seconds` (default 30), `lancedb.index_cache_size_bytes` and `lancedb.metadata_cache_size_bytes`. The LanceDB session is shared across connections in a process, so its index and metadata caches survive a connection being closed.

View file

@ -105,6 +105,8 @@ for chunk in chunks:
print(f"Headings: {meta.headings}")
print(f"Page numbers: {meta.page_numbers}")
print(f"Labels: {meta.labels}")
# Access raw metadata (including headings, page_numbers and labels)
print(f"Raw metadata: {chunk.metadata}")
```
Chunks are returned with:
@ -115,6 +117,8 @@ Chunks are returned with:
- `embedding` - `None` (not yet embedded)
- `document_id` - `None` (not yet stored)
A custom `DocumentChunker` can provide other keys and values in `metadata`. They will be stored with the chunk and are accessible when it is returned in a search result or citation, within `SearchResult.chunk_meta` / `Citation.chunk_meta`.
## Embed
`embed_chunks()` generates embeddings for chunks using the client's embedder. It automatically contextualizes chunks (prepends section headings) before embedding for better semantic search, without modifying the stored content:

View file

@ -196,7 +196,7 @@ for result in results:
print(f"Document ID: {result.document_id}")
```
Each result carries the parent document's metadata in `result.document_meta`. It is not shown to the model during QA.
Each result carries the parent document's metadata in `result.document_meta` and the relevant chunk's verbatim metadata in `result.chunk_meta`. Neither is shown to the model during QA.
Search with different search types:
```python
@ -348,7 +348,7 @@ answer, citations = await client.ask(
Images are passed to the model alongside the question. Retrieval stays text-based. The QA model must have `vision: true` in its configuration.
`client.ask` runs the [RAG capability](capabilities/rag.md) and returns `(answer_text, list[Citation])`. Citations include page numbers, section headings, document references, and the document's metadata (`document_meta`), so UIs can render metadata keys such as a public source URL alongside the citation.
`client.ask` runs the [RAG capability](capabilities/rag.md) and returns `(answer_text, list[Citation])`. Citations include page numbers, section headings, document references, the document's metadata (`document_meta`), and the cited chunk's raw, unparsed metadata (`chunk_meta`), so UIs can render metadata keys such as a public source URL alongside the citation.
The QA provider and model are configured in `haiku.rag.yaml` or can be passed directly to the client (see [Configuration](configuration/index.md)).

View file

@ -408,6 +408,7 @@ def _build_result(
score=max(r.score for r in original_results),
chunk_id=first.chunk_id,
chunk_ids=chunk_ids,
chunk_meta=first.chunk_meta,
document_id=first.document_id,
document_uri=first.document_uri,
document_title=first.document_title,

View file

@ -4,7 +4,7 @@ from textual.app import ComposeResult
from textual.containers import VerticalScroll
from textual.widgets import Markdown, Static
from haiku.rag.store.models import Chunk, Document, SearchResult
from haiku.rag.store.models import Chunk, ChunkMetadata, Document, SearchResult
class ProvenanceData(Protocol):
@ -50,6 +50,21 @@ class DetailView(VerticalScroll):
parts.append(f"**DocItem Refs:** `{refs_str}`")
return parts
def _format_extra_metadata(self, metadata: dict) -> list[str]:
"""Format raw metadata keys not already shown by `_format_provenance`.
`ChunkMetadata`'s own fields (page_numbers, headings, labels,
doc_item_refs) are excluded here since `_format_provenance` already
renders them, with its own truncation for long ref lists.
"""
extra = {
k: v for k, v in metadata.items() if k not in ChunkMetadata.model_fields
}
if not extra:
return []
metadata_str = "\n".join(f" - {k}: {v}" for k, v in extra.items())
return [f"**Metadata:**\n{metadata_str}"]
async def show_document(self, document: Document) -> None:
"""Display document details."""
title = document.title or document.uri or "Untitled Document"
@ -92,6 +107,7 @@ class DetailView(VerticalScroll):
chunk_meta = chunk.get_chunk_metadata()
content_parts.extend(self._format_provenance(chunk_meta))
content_parts.extend(self._format_extra_metadata(chunk.metadata))
if chunk.embedding:
content_parts.append(f"**Embedding:** {len(chunk.embedding)} dimensions")
@ -120,6 +136,7 @@ class DetailView(VerticalScroll):
content_parts.append(f"**Score:** {search_result.score:.4f}")
content_parts.extend(self._format_provenance(search_result))
content_parts.extend(self._format_extra_metadata(search_result.chunk_meta))
if chunk.embedding:
content_parts.append(f"**Embedding:** {len(chunk.embedding)} dimensions")

View file

@ -132,12 +132,17 @@ class SearchResult(BaseModel):
``document_meta`` carries the parent document's metadata for citation
consumers (UIs). Never part of ``format_for_agent`` output.
``chunk_meta`` is the anchor chunk's unparsed ``Chunk.metadata`` and does not
include the metadata of any other chunks merged with it. Never part of
``format_for_agent`` output.
"""
content: str
score: float
chunk_id: str | None = None
chunk_ids: list[str] = []
chunk_meta: dict = {}
document_id: str | None = None
document_uri: str | None = None
document_title: str | None = None
@ -172,6 +177,7 @@ class SearchResult(BaseModel):
page_numbers=meta.page_numbers,
headings=meta.headings,
labels=meta.labels,
chunk_meta=chunk.metadata,
image_data=image_data,
)

View file

@ -28,12 +28,18 @@ class Citation(BaseModel):
``picture_refs`` is the picture-labeled subset.
``document_meta`` carries the cited document's metadata for UIs.
``chunk_meta`` is the cited chunk's raw, unparsed ``Chunk.metadata``
dict lossless and independent of the typed fields above, so a
third-party chunker's own fields survive here even as this schema
evolves.
"""
index: int | None = None
document_id: str
chunk_id: str
chunk_ids: list[str] = Field(default_factory=list)
chunk_meta: dict = Field(default_factory=dict)
document_uri: str
document_title: str | None = None
document_meta: dict = Field(default_factory=dict)
@ -65,6 +71,7 @@ def resolve_citations(
document_id=r.document_id or "",
chunk_id=chunk_id,
chunk_ids=r.chunk_ids or [chunk_id],
chunk_meta=r.chunk_meta,
document_uri=r.document_uri or "",
document_title=r.document_title,
document_meta=r.document_meta,

View file

@ -6,6 +6,7 @@ def _result(
chunk_id: str,
chunk_ids: list[str] | None = None,
document_meta: dict | None = None,
chunk_meta: dict | None = None,
) -> SearchResult:
return SearchResult(
content="content",
@ -15,6 +16,7 @@ def _result(
document_id="doc-1",
document_uri="test://doc",
document_meta=document_meta or {},
chunk_meta=chunk_meta or {},
)
@ -54,3 +56,9 @@ def test_resolve_citations_copies_document_meta():
assert citations[0].document_meta == {
"source_url": "https://example.org/report/view"
}
def test_resolve_citations_copies_chunk_meta():
result = _result("c1", chunk_meta={"para_no": "12", "speaker": "MR SMITH"})
citations = resolve_citations(["c1"], [result])
assert citations[0].chunk_meta == {"para_no": "12", "speaker": "MR SMITH"}

View file

@ -222,6 +222,42 @@ def test_search_result_from_chunk_preserves_document_meta():
assert result.document_meta == {"source_url": "https://example.org/report/view"}
def test_search_result_from_chunk_preserves_chunk_meta():
"""Test flow through of unparsed chunk metadata from Chunk to SearchResult"""
chunk = Chunk(
id="chunk-1",
document_id="doc-1",
content="Some content.",
metadata={
"headings": ["Chapter 1"],
"para_no": "12",
"speaker": "MR SMITH",
},
)
result = SearchResult.from_chunk(chunk, score=0.9)
assert result.chunk_meta == {
"headings": ["Chapter 1"],
"para_no": "12",
"speaker": "MR SMITH",
}
def test_search_result_format_for_agent_omits_chunk_meta():
"""Test that chunk_meta is never shown to the model"""
result = SearchResult(
content="Some content.",
score=0.9,
chunk_id="chunk-1",
chunk_meta={"para_no": "12"},
)
formatted = result.format_for_agent(rank=1, total=1)
assert "para_no" not in formatted
def test_search_result_format_for_agent_omits_document_meta():
"""Document metadata is UI plumbing, never shown to the model."""
result = SearchResult(

View file

@ -799,6 +799,46 @@ class TestExpandWithItems:
"source_url": "https://example.org/report/view"
}
async def test_expanded_result_carries_anchor_chunk_meta(self, temp_db_path):
"""chunk_meta belongs only to the anchor chunk (ie whichever constituent chunk earned the result its rank)"""
from haiku.rag.client import HaikuRAG
async with HaikuRAG(temp_db_path, create=True) as rag:
items = [
DocumentItem(
document_id="doc-1",
position=i,
self_ref=f"#/texts/{i}",
label="text",
text=f"Paragraph {i}. " * 10,
)
for i in range(5)
]
await rag.document_item_repository.create_items("doc-1", items)
r1 = SearchResult(
content="Paragraph 1.",
score=0.9,
chunk_id="c1",
document_id="doc-1",
doc_item_refs=["#/texts/1"],
chunk_meta={"para_no": "12"},
)
r2 = SearchResult(
content="Paragraph 3.",
score=0.85,
chunk_id="c2",
document_id="doc-1",
doc_item_refs=["#/texts/3"],
chunk_meta={"para_no": "14"},
)
expanded = await expand_with_items(
rag.document_item_repository, "doc-1", [r1, r2], 5000
)
assert len(expanded) == 1
assert expanded[0].chunk_id == "c1"
assert expanded[0].chunk_meta == {"para_no": "12"}
async def test_merged_anchor_is_highest_scoring_constituent(self, temp_db_path):
"""A merged result's chunk_id anchors on the best-scoring constituent,
not whichever chunk sits earliest in the document."""

View file

@ -165,6 +165,100 @@ async def test_document_list_tracks_has_more():
assert doc_list.has_more is False
@pytest.mark.asyncio
async def test_detail_view_shows_chunk_metadata():
"""show_chunk renders metadata keys _format_provenance doesn't already
cover, but not a duplicate of the standard fields it does (headings,
page_numbers, labels, doc_item_refs)."""
from textual.app import App
from haiku.rag.inspector.widgets.detail_view import DetailView
chunk = Chunk(
id="chunk-1",
document_id="doc-1",
content="raw chunk text",
metadata={"headings": ["Chapter 1"], "para_no": "12"},
)
class TestApp(App):
def compose(self):
yield DetailView(id="detail")
app = TestApp()
async with app.run_test():
detail_view = app.query_one(DetailView)
await detail_view.show_chunk(chunk)
source = detail_view.content_widget.source
assert "**Metadata:**" in source
assert "para_no: 12" in source
assert "headings:" not in source # already shown as **Section:**
@pytest.mark.asyncio
async def test_detail_view_omits_metadata_block_when_only_standard_fields():
"""No **Metadata:** block at all when chunk.metadata holds nothing
beyond what _format_provenance already renders."""
from textual.app import App
from haiku.rag.inspector.widgets.detail_view import DetailView
chunk = Chunk(
id="chunk-1",
document_id="doc-1",
content="raw chunk text",
metadata={"headings": ["Chapter 1"], "page_numbers": [1]},
)
class TestApp(App):
def compose(self):
yield DetailView(id="detail")
app = TestApp()
async with app.run_test():
detail_view = app.query_one(DetailView)
await detail_view.show_chunk(chunk)
source = detail_view.content_widget.source
assert "**Metadata:**" not in source
@pytest.mark.asyncio
async def test_detail_view_shows_search_result_chunk_meta():
"""show_search_result renders SearchResult.chunk_meta's non-standard
keys, the anchor chunk's own metadata carried through search/expansion."""
from textual.app import App
from haiku.rag.inspector.widgets.detail_view import DetailView
chunk = Chunk(id="chunk-1", document_id="doc-1", content="raw chunk text")
search_result = SearchResult(
content="raw chunk text",
score=0.5,
chunk_id="chunk-1",
doc_item_refs=[f"#/texts/{i}" for i in range(7)],
chunk_meta={
"para_no": "12",
"doc_item_refs": [f"#/texts/{i}" for i in range(7)],
},
)
class TestApp(App):
def compose(self):
yield DetailView(id="detail")
app = TestApp()
async with app.run_test():
detail_view = app.query_one(DetailView)
await detail_view.show_search_result(chunk, search_result)
source = detail_view.content_widget.source
assert "**Metadata:**" in source
assert "para_no: 12" in source
# _format_provenance's own 5-item truncation for doc_item_refs is
# untouched by the filtered-out duplicate in chunk_meta.
assert "+2 more" in source
assert "doc_item_refs:" not in source
@pytest.mark.asyncio
async def test_context_modal_renders_pictures_when_vision_enabled():
"""ContextModal must mount one TextualImage per attached picture

View file

@ -2,7 +2,7 @@ import pytest
from haiku.rag.client import HaikuRAG
from haiku.rag.mcp import create_mcp_server
from haiku.rag.store.models import Document, SearchResult
from haiku.rag.store.models import Chunk, Document, SearchResult
from haiku.rag.tools.document import DocumentInfo
@ -69,6 +69,41 @@ class TestMCPReadTools:
results = await search(query="artificial intelligence", limit=1)
assert len(results) == 1
@pytest.mark.asyncio
@pytest.mark.filterwarnings("ignore:Found propagated trace context:RuntimeWarning")
async def test_search_documents_preserves_chunk_meta_through_serialization(
self, mcp_db
):
"""Chunk_meta must survive FastMCP's actual wire serialization.
Calling the tool function directly bypasses that serialization step entirely."""
from fastmcp import Client
async with HaikuRAG(mcp_db, create=True) as rag:
doc = await rag.get_document_by_uri("test://ai-overview")
embedding = (await rag.embedder.embed_documents(["x"]))[0]
await rag.chunk_repository.create(
Chunk(
document_id=doc.id,
content="Artificial intelligence is transforming industries worldwide.",
metadata={"fake-metadata-for-testing": "42"},
embedding=embedding,
)
)
await rag.chunk_repository._ensure_fts_index()
mcp = create_mcp_server(mcp_db, read_only=True)
async with Client(mcp) as client:
result = await client.call_tool(
"search_documents", {"query": "artificial intelligence"}
)
results = result.structured_content["result"]
assert results
assert any(
r["chunk_meta"] == {"fake-metadata-for-testing": "42"} for r in results
)
@pytest.mark.asyncio
async def test_get_document(self, mcp_db):
mcp = create_mcp_server(mcp_db, read_only=True)