haiku.rag/tests/store/test_v0_45_0_migration.py
Lawrence Akka 5e6284fd02 Stop eagerly importing Store from haiku.rag.store
Import Store from haiku.rag.store.engine directly instead of re-exporting it
through the package __init__, and defer HaikuRAGApp's import in cli.py behind
TYPE_CHECKING/lazy imports, avoiding an eager import at CLI startup.
2026-08-11 17:26:51 +01:00

228 lines
7.8 KiB
Python

"""Tests for the v0.45.0 picture-data backfill migration.
The migration walks every document, decodes picture image URIs out of the
stored docling blob, writes the bytes to ``document_items.picture_data``,
and then strips the URIs from the blob (which lives separately on the
documents table).
"""
import base64
import json
import pyarrow as pa
import pytest
from haiku.rag.store.compression import compress_json, decompress_json
from haiku.rag.store.engine import DocumentItemRecord, DocumentRecord, Store
from haiku.rag.store.upgrades.v0_45_0 import _apply_extract_picture_bytes
PNG_BYTES = b"\x89PNG\r\n\x1a\nfake-picture-bytes"
PNG_DATA_URI = f"data:image/png;base64,{base64.b64encode(PNG_BYTES).decode('ascii')}"
def _docling_blob_with_picture(self_ref: str = "#/pictures/0") -> bytes:
"""Build a compressed docling-document blob carrying one picture with
an inline data URI — the shape v0.45.0 was written to backfill from."""
doc = {
"schema_name": "DoclingDocument",
"version": "1.10.0",
"name": "test",
"pictures": [
{
"self_ref": self_ref,
"image": {"mimetype": "image/png", "uri": PNG_DATA_URI},
}
],
"tables": [],
"texts": [],
"groups": [],
"body": {"self_ref": "#/body", "children": [], "label": "unspecified"},
"furniture": {
"self_ref": "#/furniture",
"children": [],
"label": "unspecified",
},
}
return compress_json(json.dumps(doc))
@pytest.mark.asyncio
async def test_migration_backfills_picture_bytes_and_strips_blob(temp_db_path):
"""Happy path: doc has a picture with bytes inline in the blob plus a
matching items row (with picture_data NULL). The migration writes the
bytes onto the items row and clears the URI from the blob."""
async with Store(temp_db_path, create=True) as store:
doc_id = "doc-1"
# Insert a doc whose blob carries a picture data URI.
await store.documents_table.add(
[
DocumentRecord(
id=doc_id,
content="x",
docling_document=_docling_blob_with_picture(),
docling_version="1.10.0",
)
]
)
# Insert a matching items row, picture_data deliberately empty.
await store.document_items_table.add(
[
DocumentItemRecord(
document_id=doc_id,
position=0,
self_ref="#/pictures/0",
label="picture",
text="",
page_numbers="[1]",
picture_data=None,
)
]
)
await _apply_extract_picture_bytes(store)
# picture_data is populated on the items row.
rows = await (
store.document_items_table.query()
.where(f"document_id = '{doc_id}' AND self_ref = '#/pictures/0'")
.to_list()
)
assert len(rows) == 1
assert rows[0]["picture_data"] == PNG_BYTES
# The docling blob no longer carries the data URI on the picture.
doc_rows = await (
store.documents_table.query()
.where(f"id = '{doc_id}'")
.select(["docling_document"])
.to_list()
)
blob = json.loads(decompress_json(doc_rows[0]["docling_document"]))
assert blob["pictures"][0]["image"] is None
@pytest.mark.asyncio
async def test_migration_runs_against_pre_v0_48_0_schema(temp_db_path):
"""Regression: v0.45.0 must not depend on columns introduced later.
Reproduces the client report where a DB on which v0.40.0 had already
run (and thus had a ``document_items`` table without ``picture_data``,
``heading_level`` or ``tree_depth``) failed v0.45.0 with
``Field 'heading_level' not found in target schema`` because the
migration was building rows from the current ``DocumentItemRecord``
Pydantic model — which now carries fields added in v0.48.0.
"""
legacy_items_schema = pa.schema(
[
pa.field("document_id", pa.string()),
pa.field("position", pa.int64()),
pa.field("self_ref", pa.string()),
pa.field("label", pa.string()),
pa.field("text", pa.string()),
pa.field("page_numbers", pa.string()),
]
)
async with Store(temp_db_path, create=True, skip_migration_check=True) as store:
# Replace the freshly-created (latest-schema) document_items table
# with one that only carries the v0.40.0 columns.
await store.db.drop_table("document_items")
legacy_items = await store.db.create_table(
"document_items", schema=legacy_items_schema
)
doc_id = "doc-1"
await store.documents_table.add(
[
DocumentRecord(
id=doc_id,
content="x",
docling_document=_docling_blob_with_picture(),
docling_version="1.10.0",
)
]
)
await legacy_items.add(
[
{
"document_id": doc_id,
"position": 0,
"self_ref": "#/pictures/0",
"label": "picture",
"text": "",
"page_numbers": "[1]",
}
]
)
await store.set_haiku_version("0.40.0")
# Re-open and reach for the table fresh so we use the legacy-schema handle.
async with Store(temp_db_path, skip_migration_check=True) as store:
await _apply_extract_picture_bytes(store)
rows = await (
store.document_items_table.query()
.where(f"document_id = '{doc_id}' AND self_ref = '#/pictures/0'")
.to_list()
)
assert len(rows) == 1
assert rows[0]["picture_data"] == PNG_BYTES
# And the blob is stripped.
doc_rows = await (
store.documents_table.query()
.where(f"id = '{doc_id}'")
.select(["docling_document"])
.to_list()
)
blob = json.loads(decompress_json(doc_rows[0]["docling_document"]))
assert blob["pictures"][0]["image"] is None
@pytest.mark.asyncio
async def test_migration_is_idempotent(temp_db_path):
"""Running the migration twice on the same DB is a no-op the second
time — the blob is already stripped."""
async with Store(temp_db_path, create=True) as store:
doc_id = "doc-1"
await store.documents_table.add(
[
DocumentRecord(
id=doc_id,
content="x",
docling_document=_docling_blob_with_picture(),
docling_version="1.10.0",
)
]
)
await store.document_items_table.add(
[
DocumentItemRecord(
document_id=doc_id,
position=0,
self_ref="#/pictures/0",
label="picture",
text="",
page_numbers="[1]",
)
]
)
await _apply_extract_picture_bytes(store)
# Capture state after first run.
rows_first = await (
store.document_items_table.query()
.where(f"document_id = '{doc_id}'")
.to_list()
)
first_bytes = rows_first[0]["picture_data"]
# Re-run.
await _apply_extract_picture_bytes(store)
rows_second = await (
store.document_items_table.query()
.where(f"document_id = '{doc_id}'")
.to_list()
)
assert rows_second[0]["picture_data"] == first_bytes