From 31fc8aba295c7ff93ae705c8c97b1549e47dc65c Mon Sep 17 00:00:00 2001 From: Yiorgis Gozadinos Date: Wed, 2 Sep 2026 12:20:33 +0300 Subject: [PATCH] Guard dedup against compaction and document search units A picture deduplicated within a burst must still reach the model once in its own question and, when cited, again via the capsule re-fetch after compaction; uncited it is dropped. Two wire tests pin both paths. The harness builds the rag capability with defer_loading=False: a deferred evidence capability the model has not loaded presents an empty record, so compaction computes boundary 0 and rewrites nothing. qa.md describes max_searches in units. --- docs/configuration/qa.md | 4 +- tests/capabilities/test_evidence_wire.py | 83 ++++++++++++++++++++++++ 2 files changed, 85 insertions(+), 2 deletions(-) diff --git a/docs/configuration/qa.md b/docs/configuration/qa.md index 9716c4d1..05e4c5a4 100644 --- a/docs/configuration/qa.md +++ b/docs/configuration/qa.md @@ -30,12 +30,12 @@ qa: enable_thinking: true temperature: 0.3 # Default: 0.3 vision: false # Set true for vision-capable models - max_searches: 5 # Maximum search tool calls per question + max_searches: 5 # Maximum search units per question ``` - **model**: LLM configuration (see [Providers](providers.md#model-settings)) - **model.vision**: Set to `true` for vision-capable models (`qwen2.5vl`, `qwen3.6`, `gpt-4o`, `claude-sonnet`, …). The capability's `search` tool only attaches picture bytes (`BinaryContent`) to its `ToolReturn` when this is `true`, otherwise picture bytes are withheld. See [Pictures × embedder × QA model](processing.md#pictures-embedder-qa-model-how-the-pieces-compose) for the full matrix. -- **max_searches**: Maximum number of search tool calls a capability can make per question (default: 5). Shared by the RAG and analysis capabilities. +- **max_searches**: Maximum number of search units a capability can spend per question (default: 5). Up to three searches emitted in the same model response share one unit, so a model that rephrases its query in one response spends one unit. A search in a later response starts a new unit, as does each further group of three within one response. Shared by the RAG and analysis capabilities. Searches in one response also deduplicate their returns: evidence a sibling search already showed collapses to a reference line, and each picture attaches once per response. !!! note "Thinking on vLLM" `enable_thinking` only applies to models with a pydantic-ai reasoning profile (o-series, gpt-5, gpt-oss). For other vLLM-served models such as Qwen3 or the Gemma family, the field is a silent no-op — set the chat template switch via [`extra_body`](providers.md#raw-provider-pass-through) instead. diff --git a/tests/capabilities/test_evidence_wire.py b/tests/capabilities/test_evidence_wire.py index 263a0e35..c189c2f6 100644 --- a/tests/capabilities/test_evidence_wire.py +++ b/tests/capabilities/test_evidence_wire.py @@ -517,6 +517,89 @@ async def test_a_picture_that_will_not_decode_emits_neither_image_nor_label( assert texts_of(wire[-1]) == [] +def _burst_result() -> SearchResult: + return SearchResult( + content="evidence", + score=1.0, + chunk_id="chunk-1", + document_id="doc-1", + source="main", + doc_item_refs=["#/pictures/0"], + image_data={"#/pictures/0": base64.b64encode(REAL_PNG).decode()}, + ) + + +async def _fanout_question_then_another(temp_db_path, cite: bool) -> list[list[Any]]: + """Question 1 fans out over one picture chunk; question 2 follows compacted.""" + rag = create_rag( + db_path=temp_db_path, config=AppConfig(), defer_loading=False, vision=True + ) + client = AsyncMock() + client.search.side_effect = [[_burst_result()], [_burst_result()]] + client.expand_context.side_effect = lambda results: results + client.source_names = ["main"] + rag.borrowed_rag = client + citing = ( + [[ToolCallPart("rag_cite", {"chunk_ids": ["chunk-1"]}, "call-3")]] + if cite + else [] + ) + calls = iter( + [ + [ + ToolCallPart("rag_search", {"query": "figure"}, "call-1"), + ToolCallPart("rag_search", {"query": "the figure"}, "call-2"), + ], + *citing, + [TextPart("first answer")], + [TextPart("second answer")], + ] + ) + wire: list[list[Any]] = [] + + async def model(messages, _info): + wire.append(list(messages)) + return ModelResponse(parts=next(calls)) + + agent = Agent( + FunctionModel(model), + deps_type=Deps, + capabilities=[rag, create_compaction()], + ) + deps = Deps() + + with patch.object( + RAGCapability, "get_picture_bytes", AsyncMock(return_value=REAL_PNG) + ): + first = await agent.run("what does the figure show?", deps=deps) + await agent.run( + "and what else?", deps=deps, message_history=first.all_messages() + ) + return wire + + +@pytest.mark.asyncio +async def test_a_burst_deduplicated_picture_survives_compaction_when_cited( + temp_db_path, +): + """Dedup attaches the picture once in its own question; the capsule re-fetches + it for the next. Neither pass may leave the model without it.""" + wire = await _fanout_question_then_another(temp_db_path, cite=True) + + assert [picture.data for picture in images_of(wire[1])] == [REAL_PNG] + assert [picture.data for picture in images_of(wire[-1])] == [REAL_PNG] + + +@pytest.mark.asyncio +async def test_a_burst_deduplicated_picture_is_dropped_by_compaction_uncited( + temp_db_path, +): + wire = await _fanout_question_then_another(temp_db_path, cite=False) + + assert [picture.data for picture in images_of(wire[1])] == [REAL_PNG] + assert images_of(wire[-1]) == [] + + @pytest.mark.asyncio async def test_the_capsule_is_built_once_per_request_and_again_for_the_next( temp_db_path,