diff --git a/CHANGELOG.md b/CHANGELOG.md index f7c93ca3..61c216b0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed - Context expansion no longer drops a retrieved result from a merged group when clipping to `search.max_context_chars`; groups whose clip would evict a constituent's evidence are returned as separate expanded results. +- Context expansion now fills missing page metadata from input search results when the referenced items survive in the expanded result. ## [0.65.1] - 2026-07-10 diff --git a/haiku_rag_slim/haiku/rag/context.py b/haiku_rag_slim/haiku/rag/context.py index c49e66d0..f6ce778f 100644 --- a/haiku_rag_slim/haiku/rag/context.py +++ b/haiku_rag_slim/haiku/rag/context.py @@ -284,6 +284,20 @@ def _group_lost_constituent(built: SearchResult, group: list[SearchResult]) -> b ) +def _add_input_pages_for_surviving_refs( + pages: set[int], refs: list[str], original_results: list[SearchResult] +) -> None: + """Fill missing item-table pages from inputs whose own refs all survived.""" + surviving = set(refs) + if not surviving: + return + for result in original_results: + if not result.page_numbers or not result.doc_item_refs: + continue + if set(result.doc_item_refs) <= surviving: + pages.update(result.page_numbers) + + def _build_result( range_start: int, range_end: int, @@ -374,6 +388,8 @@ def _build_result( else: pages, refs, labels = _collect_meta(base_spans) + _add_input_pages_for_surviving_refs(pages, refs, original_results) + # Carry image_data and picture_captions from the originally retrieved # chunks, but only for constituents whose refs survive the window — a # chunk clipped out of the budget must not still ship its image to the diff --git a/tests/test_context.py b/tests/test_context.py index 31f30c2b..87c9a5f0 100644 --- a/tests/test_context.py +++ b/tests/test_context.py @@ -1027,6 +1027,107 @@ class TestExpandWithItems: assert by_chunk["c2"].content == "B" * 400 assert by_chunk["c2"].page_numbers == [2] + async def test_surviving_refs_fill_missing_item_pages_from_input( + self, temp_db_path + ): + """When a visible item has missing page metadata, use the input + result's page_numbers as a floor for that surviving ref.""" + from haiku.rag.client import HaikuRAG + + async with HaikuRAG(temp_db_path, create=True) as rag: + items = [ + DocumentItem( + document_id="doc-1", + position=0, + self_ref="#/texts/0", + label="text", + text="Visible item with missing item-table pages.", + page_numbers=[], + ), + DocumentItem( + document_id="doc-1", + position=1, + self_ref="#/texts/1", + label="text", + text="Visible item with stored item-table pages.", + page_numbers=[8], + ), + ] + await rag.document_item_repository.create_items("doc-1", items) + + r_missing_item_page = SearchResult( + content=items[0].text, + score=0.9, + chunk_id="c-missing", + document_id="doc-1", + doc_item_refs=["#/texts/0"], + page_numbers=[7], + ) + r_with_item_page = SearchResult( + content=items[1].text, + score=0.8, + chunk_id="c-present", + document_id="doc-1", + doc_item_refs=["#/texts/1"], + page_numbers=[8], + ) + expanded = await expand_with_items( + rag.document_item_repository, + "doc-1", + [r_missing_item_page, r_with_item_page], + 5000, + ) + + assert len(expanded) == 1 + assert set(expanded[0].doc_item_refs) == {"#/texts/0", "#/texts/1"} + assert expanded[0].page_numbers == [7, 8] + + async def test_input_pages_not_added_for_clipped_out_refs(self, temp_db_path): + """Input page metadata is not blindly unioned when only some of a + constituent's refs survive the clip window.""" + from haiku.rag.client import HaikuRAG + + item0_text = "LEFTMARK " + "a" * 91 + item1_text = "RIGHTMARK " + "b" * 70 + async with HaikuRAG(temp_db_path, create=True) as rag: + items = [ + DocumentItem( + document_id="doc-1", + position=0, + self_ref="#/texts/0", + label="text", + text=item0_text, + page_numbers=[1], + ), + DocumentItem( + document_id="doc-1", + position=1, + self_ref="#/texts/1", + label="text", + text=item1_text, + page_numbers=[2], + ), + ] + await rag.document_item_repository.create_items("doc-1", items) + + result = SearchResult( + content=item0_text, + score=0.9, + chunk_id="c-both", + document_id="doc-1", + doc_item_refs=["#/texts/0", "#/texts/1"], + page_numbers=[1, 2], + ) + expanded = await expand_with_items( + rag.document_item_repository, "doc-1", [result], 100 + ) + + assert len(expanded) == 1 + assert "LEFTMARK" in expanded[0].content + assert "RIGHTMARK" not in expanded[0].content + assert expanded[0].doc_item_refs == ["#/texts/0"] + assert expanded[0].page_numbers == [1] + async def test_fuzzy_match_preserves_central_marker(self, temp_db_path): """The chunk's text need not be verbatim in the joined item text: a clean central marker is still located via the central-slice anchor."""