Convert batched ingest content as text, not as a locator
`_ingest_batched` asserts its payload is inline content and then passed it to `HaikuRAG.convert`, which disambiguates a str by parsing it: anything whose scheme reads as http or https is fetched over the network instead of stored. `urlparse` strips leading whitespace, so a passage beginning with a newline and a URL qualifies. 187 passages across MTRAG's cloud and fiqa corpora start that way, which crashed the pooled build. No clapnq passage does, so mtrag_clapnq and every other existing dataset is unaffected. Now converts through the configured converter's text path, which is what create_document already does. Claude-Session: https://claude.ai/code/session_01WhudUtZm6qqiuv8Y1sbwSc
This commit is contained in:
parent
ae5c798af7
commit
058120c05a
3 changed files with 20 additions and 3 deletions
|
|
@ -20,6 +20,7 @@
|
||||||
and rerun `create-index`.
|
and rerun `create-index`.
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
|
- Batched evaluation ingest converts inline content as text instead of letting `HaikuRAG.convert` disambiguate it, so a passage beginning with a URL is stored rather than fetched over HTTP. 187 MTRAG cloud and fiqa passages start with one; no clapnq passage does, so no existing dataset's numbers change.
|
||||||
- `mtrag_federated` builds vacuum each collection after ingest and assert the chunks FTS index covers every row. Without the vacuum the index stays at zero rows, and full-text search returns near-arbitrary rows while still returning results.
|
- `mtrag_federated` builds vacuum each collection after ingest and assert the chunks FTS index covers every row. Without the vacuum the index stays at zero rows, and full-text search returns near-arbitrary rows while still returning results.
|
||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|
|
||||||
|
|
@ -39,6 +39,9 @@ async def _ingest_batched(
|
||||||
row["uri"]: row["id"] for row in uri_rows if row["id"] not in chunked_ids
|
row["uri"]: row["id"] for row in uri_rows if row["id"] not in chunked_ids
|
||||||
}
|
}
|
||||||
|
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
|
|
||||||
|
converter = get_converter(rag._config)
|
||||||
batch: list[DocumentImport] = []
|
batch: list[DocumentImport] = []
|
||||||
for doc in corpus:
|
for doc in corpus:
|
||||||
payload = spec.document_mapper(cast(Mapping[str, Any], doc))
|
payload = spec.document_mapper(cast(Mapping[str, Any], doc))
|
||||||
|
|
@ -48,7 +51,13 @@ async def _ingest_batched(
|
||||||
if payload.uri in chunkless:
|
if payload.uri in chunkless:
|
||||||
await rag.delete_document(chunkless[payload.uri])
|
await rag.delete_document(chunkless[payload.uri])
|
||||||
assert payload.content is not None, "batched ingest requires inline content"
|
assert payload.content is not None, "batched ingest requires inline content"
|
||||||
docling_document = await rag.convert(payload.content, format=payload.format)
|
# Convert as text explicitly. `rag.convert` disambiguates a str by
|
||||||
|
# parsing it, and a passage beginning with a URL (187 of them across
|
||||||
|
# MTRAG's cloud and fiqa corpora) is then fetched over HTTP instead of
|
||||||
|
# stored. Batched ingest has already asserted the content is inline.
|
||||||
|
docling_document = await converter.convert_text(
|
||||||
|
payload.content, format=payload.format
|
||||||
|
)
|
||||||
chunks = await rag.chunk(docling_document)
|
chunks = await rag.chunk(docling_document)
|
||||||
batch.append(
|
batch.append(
|
||||||
DocumentImport(
|
DocumentImport(
|
||||||
|
|
|
||||||
|
|
@ -888,7 +888,11 @@ class TestBatchedIngest:
|
||||||
rag.store.chunks_table = _table(
|
rag.store.chunks_table = _table(
|
||||||
[{"document_id": f"id-{uri}"} for uri in complete_uris]
|
[{"document_id": f"id-{uri}"} for uri in complete_uris]
|
||||||
)
|
)
|
||||||
rag.convert = AsyncMock(side_effect=lambda content, **kw: f"docling:{content}")
|
# Batched ingest converts text through the configured converter, the
|
||||||
|
# same path create_document uses, so the double needs a real config.
|
||||||
|
from haiku.rag.config.models import AppConfig
|
||||||
|
|
||||||
|
rag._config = AppConfig()
|
||||||
rag.chunk = AsyncMock(return_value=[])
|
rag.chunk = AsyncMock(return_value=[])
|
||||||
rag.import_documents = AsyncMock()
|
rag.import_documents = AsyncMock()
|
||||||
rag.delete_document = AsyncMock()
|
rag.delete_document = AsyncMock()
|
||||||
|
|
@ -933,8 +937,11 @@ class TestBatchedIngest:
|
||||||
await _ingest_batched(rag, self._spec(), corpus, batch_size=10)
|
await _ingest_batched(rag, self._spec(), corpus, batch_size=10)
|
||||||
|
|
||||||
(batch,), _ = rag.import_documents.call_args
|
(batch,), _ = rag.import_documents.call_args
|
||||||
|
# Conversion goes through the configured converter now, not rag.convert,
|
||||||
|
# so the batch contents are the assertion: exactly the incomplete uris,
|
||||||
|
# which is stricter than counting conversions.
|
||||||
assert [imp.uri for imp in batch] == ["u1", "u3"]
|
assert [imp.uri for imp in batch] == ["u1", "u3"]
|
||||||
assert rag.convert.await_count == 2
|
assert len(batch) == 2
|
||||||
rag.delete_document.assert_not_awaited()
|
rag.delete_document.assert_not_awaited()
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue