Add order to SearchResult, add ChunkRepository.get_chunks_in_range(), to use them in _expand_with_chunks to fetch only nearby chunks
This commit is contained in:
parent
6fdb15e3b2
commit
68c3fa0f79
4 changed files with 86 additions and 26 deletions
|
|
@ -4,10 +4,14 @@
|
||||||
### Changed
|
### Changed
|
||||||
|
|
||||||
- **Dependency updates**: lancedb 0.30.2, pydantic-ai-slim ≥1.77.0, docling ≥2.84.0, docling-core ≥2.71.0, haiku.skills ≥0.13.0, cachetools ≥7.0.5, pydantic-monty ≥0.0.9, cohere ≥5.21.1, textual ≥8.2.1, ty ≥0.0.28, ruff ≥0.15.9
|
- **Dependency updates**: lancedb 0.30.2, pydantic-ai-slim ≥1.77.0, docling ≥2.84.0, docling-core ≥2.71.0, haiku.skills ≥0.13.0, cachetools ≥7.0.5, pydantic-monty ≥0.0.9, cohere ≥5.21.1, textual ≥8.2.1, ty ≥0.0.28, ruff ≥0.15.9
|
||||||
|
- **Search result model**: `SearchResult` now includes `order` field propagated from chunk order
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
- **Type checking**: Fix 37 new ty 0.0.28 diagnostics with proper None guards, assertions, and specific ignore codes
|
- **Type checking**: Fix 37 new ty 0.0.28 diagnostics with proper None guards, assertions, and specific ignore codes
|
||||||
|
- **Search performance**: Avoid loading full document blobs (docling_document, content) during search — use column projection to fetch only needed metadata (id, uri, title, metadata)
|
||||||
|
- **Context expansion performance**: Load only docling columns during expand_context (skip content blob), and only when doc_item_refs exist
|
||||||
|
- **Chunk expansion performance**: Fetch only chunks in the needed order range during context expansion instead of all chunks for a document
|
||||||
|
|
||||||
## [0.36.3] - 2026-04-01
|
## [0.36.3] - 2026-04-01
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1400,32 +1400,40 @@ class HaikuRAG:
|
||||||
radius: int,
|
radius: int,
|
||||||
) -> list[SearchResult]:
|
) -> list[SearchResult]:
|
||||||
"""Expand results using chunk-based adjacency."""
|
"""Expand results using chunk-based adjacency."""
|
||||||
all_chunks = await self.chunk_repository.get_by_document_id(doc_id)
|
# Build ranges from result orders
|
||||||
if not all_chunks:
|
|
||||||
return results
|
|
||||||
|
|
||||||
content_to_chunk = {c.content: c for c in all_chunks}
|
|
||||||
chunk_by_order = {c.order: c for c in all_chunks}
|
|
||||||
min_order, max_order = min(chunk_by_order.keys()), max(chunk_by_order.keys())
|
|
||||||
|
|
||||||
# Build ranges
|
|
||||||
ranges: list[tuple[int, int, SearchResult]] = []
|
ranges: list[tuple[int, int, SearchResult]] = []
|
||||||
passthrough: list[SearchResult] = []
|
passthrough: list[SearchResult] = []
|
||||||
|
|
||||||
for result in results:
|
for result in results:
|
||||||
chunk = content_to_chunk.get(result.content)
|
if result.chunk_id is None:
|
||||||
if chunk is None:
|
|
||||||
passthrough.append(result)
|
passthrough.append(result)
|
||||||
continue
|
continue
|
||||||
start = max(min_order, chunk.order - radius)
|
start = result.order - radius
|
||||||
end = min(max_order, chunk.order + radius)
|
end = result.order + radius
|
||||||
ranges.append((start, end, result))
|
ranges.append((start, end, result))
|
||||||
|
|
||||||
|
if not ranges:
|
||||||
|
return results
|
||||||
|
|
||||||
|
# Compute the full order range needed and fetch only those chunks
|
||||||
|
all_starts = [s for s, _, _ in ranges]
|
||||||
|
all_ends = [e for _, e, _ in ranges]
|
||||||
|
range_min = min(all_starts)
|
||||||
|
range_max = max(all_ends)
|
||||||
|
|
||||||
|
chunks_in_range = await self.chunk_repository.get_chunks_in_range(
|
||||||
|
doc_id, range_min, range_max
|
||||||
|
)
|
||||||
|
if not chunks_in_range:
|
||||||
|
return results
|
||||||
|
|
||||||
|
chunk_by_order = {c.order: c for c in chunks_in_range}
|
||||||
|
|
||||||
# Merge and build results
|
# Merge and build results
|
||||||
final_results: list[SearchResult] = []
|
final_results: list[SearchResult] = []
|
||||||
for min_idx, max_idx, original_results in self._merge_ranges(ranges):
|
for min_idx, max_idx, original_results in self._merge_ranges(ranges):
|
||||||
# Collect chunks in order
|
# Collect chunks in order
|
||||||
chunks_in_range = [
|
merged_chunks = [
|
||||||
chunk_by_order[o]
|
chunk_by_order[o]
|
||||||
for o in range(min_idx, max_idx + 1)
|
for o in range(min_idx, max_idx + 1)
|
||||||
if o in chunk_by_order
|
if o in chunk_by_order
|
||||||
|
|
@ -1433,7 +1441,7 @@ class HaikuRAG:
|
||||||
first = original_results[0]
|
first = original_results[0]
|
||||||
final_results.append(
|
final_results.append(
|
||||||
SearchResult(
|
SearchResult(
|
||||||
content="".join(c.content for c in chunks_in_range),
|
content="".join(c.content for c in merged_chunks),
|
||||||
score=max(r.score for r in original_results),
|
score=max(r.score for r in original_results),
|
||||||
chunk_id=first.chunk_id,
|
chunk_id=first.chunk_id,
|
||||||
document_id=first.document_id,
|
document_id=first.document_id,
|
||||||
|
|
|
||||||
|
|
@ -117,6 +117,7 @@ class SearchResult(BaseModel):
|
||||||
document_id: str | None = None
|
document_id: str | None = None
|
||||||
document_uri: str | None = None
|
document_uri: str | None = None
|
||||||
document_title: str | None = None
|
document_title: str | None = None
|
||||||
|
order: int = 0
|
||||||
doc_item_refs: list[str] = []
|
doc_item_refs: list[str] = []
|
||||||
page_numbers: list[int] = []
|
page_numbers: list[int] = []
|
||||||
headings: list[str] | None = None
|
headings: list[str] | None = None
|
||||||
|
|
@ -137,6 +138,7 @@ class SearchResult(BaseModel):
|
||||||
document_id=chunk.document_id,
|
document_id=chunk.document_id,
|
||||||
document_uri=chunk.document_uri,
|
document_uri=chunk.document_uri,
|
||||||
document_title=chunk.document_title,
|
document_title=chunk.document_title,
|
||||||
|
order=chunk.order,
|
||||||
doc_item_refs=meta.doc_item_refs,
|
doc_item_refs=meta.doc_item_refs,
|
||||||
page_numbers=meta.page_numbers,
|
page_numbers=meta.page_numbers,
|
||||||
headings=meta.headings,
|
headings=meta.headings,
|
||||||
|
|
|
||||||
|
|
@ -363,22 +363,68 @@ class ChunkRepository:
|
||||||
)
|
)
|
||||||
return len(df)
|
return len(df)
|
||||||
|
|
||||||
|
async def get_chunks_in_range(
|
||||||
|
self, document_id: str, min_order: int, max_order: int
|
||||||
|
) -> list[Chunk]:
|
||||||
|
"""Get chunks for a document within an order range.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
document_id: The document ID to get chunks for.
|
||||||
|
min_order: Minimum order value (inclusive).
|
||||||
|
max_order: Maximum order value (inclusive).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of chunks within the order range.
|
||||||
|
"""
|
||||||
|
where = (
|
||||||
|
f"document_id = '{document_id}'"
|
||||||
|
f" AND `order` >= {min_order}"
|
||||||
|
f" AND `order` <= {max_order}"
|
||||||
|
)
|
||||||
|
results = list(
|
||||||
|
self.store.chunks_table.search()
|
||||||
|
.where(where)
|
||||||
|
.to_pydantic(self.store.ChunkRecord)
|
||||||
|
)
|
||||||
|
return [
|
||||||
|
Chunk(
|
||||||
|
id=rec.id,
|
||||||
|
document_id=rec.document_id,
|
||||||
|
content=rec.content,
|
||||||
|
metadata=json.loads(rec.metadata),
|
||||||
|
order=rec.order,
|
||||||
|
)
|
||||||
|
for rec in results
|
||||||
|
]
|
||||||
|
|
||||||
async def get_adjacent_chunks(self, chunk: Chunk, num_adjacent: int) -> list[Chunk]:
|
async def get_adjacent_chunks(self, chunk: Chunk, num_adjacent: int) -> list[Chunk]:
|
||||||
"""Get adjacent chunks before and after the given chunk within the same document."""
|
"""Get adjacent chunks before and after the given chunk within the same document."""
|
||||||
assert chunk.document_id, "Document id is required for adjacent chunk finding"
|
assert chunk.document_id, "Document id is required for adjacent chunk finding"
|
||||||
|
|
||||||
chunk_order = chunk.order
|
min_order = chunk.order - num_adjacent
|
||||||
|
max_order = chunk.order + num_adjacent
|
||||||
|
|
||||||
# Fetch chunks for the same document and filter by order proximity
|
where = (
|
||||||
all_chunks = await self.get_by_document_id(chunk.document_id)
|
f"document_id = '{chunk.document_id}'"
|
||||||
|
f" AND `order` >= {min_order}"
|
||||||
adjacent_chunks: list[Chunk] = []
|
f" AND `order` <= {max_order}"
|
||||||
for c in all_chunks:
|
f" AND id != '{chunk.id}'"
|
||||||
c_order = c.order
|
)
|
||||||
if c.id != chunk.id and abs(c_order - chunk_order) <= num_adjacent:
|
results = list(
|
||||||
adjacent_chunks.append(c)
|
self.store.chunks_table.search()
|
||||||
|
.where(where)
|
||||||
return adjacent_chunks
|
.to_pydantic(self.store.ChunkRecord)
|
||||||
|
)
|
||||||
|
return [
|
||||||
|
Chunk(
|
||||||
|
id=rec.id,
|
||||||
|
document_id=rec.document_id,
|
||||||
|
content=rec.content,
|
||||||
|
metadata=json.loads(rec.metadata),
|
||||||
|
order=rec.order,
|
||||||
|
)
|
||||||
|
for rec in results
|
||||||
|
]
|
||||||
|
|
||||||
async def _process_search_results(
|
async def _process_search_results(
|
||||||
self, query_result: "pd.DataFrame | LanceQueryBuilder"
|
self, query_result: "pd.DataFrame | LanceQueryBuilder"
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue