diff --git a/haiku_rag_slim/haiku/rag/store/upgrades/v0_45_0.py b/haiku_rag_slim/haiku/rag/store/upgrades/v0_45_0.py index cc6d69ac..b6bdf43f 100644 --- a/haiku_rag_slim/haiku/rag/store/upgrades/v0_45_0.py +++ b/haiku_rag_slim/haiku/rag/store/upgrades/v0_45_0.py @@ -11,7 +11,7 @@ from haiku.rag.utils import escape_sql_string logger = logging.getLogger(__name__) -BATCH_SIZE = 5 +PROGRESS_INTERVAL = 5 async def _ensure_picture_data_column(store: Store) -> None: @@ -29,7 +29,7 @@ async def _ensure_picture_data_column(store: Store) -> None: ) -async def _apply_extract_picture_bytes(store: Store) -> None: # pragma: no cover +async def _apply_extract_picture_bytes(store: Store) -> None: """Add the ``picture_data`` column to ``document_items`` (if missing), backfill it from existing docling blobs, and strip the inline picture URIs out of those blobs. @@ -59,8 +59,8 @@ async def _apply_extract_picture_bytes(store: Store) -> None: # pragma: no cove blob_only = 0 skipped = 0 - for batch_start in range(0, total, BATCH_SIZE): - batch_ids = ids[batch_start : batch_start + BATCH_SIZE] + for batch_start in range(0, total, PROGRESS_INTERVAL): + batch_ids = ids[batch_start : batch_start + PROGRESS_INTERVAL] for doc_id in batch_ids: safe_id = escape_sql_string(doc_id) rows = await ( @@ -120,8 +120,8 @@ async def _apply_extract_picture_bytes(store: Store) -> None: # pragma: no cove if updates: # Find the matching items rows so we can preserve their # position/label/text/page_numbers and just attach bytes. - # Earlier migrations (v0.40.0) populate items rows for every - # picture self_ref, so the lookup below is total. + # v0.40.0 runs before v0.45.0 and populates items rows for + # every picture self_ref, so the lookup below is total. self_refs = [u[0] for u in updates] ref_clause = ", ".join(f"'{escape_sql_string(r)}'" for r in self_refs) existing_items = await ( diff --git a/tests/store/test_v0_45_0_migration.py b/tests/store/test_v0_45_0_migration.py new file mode 100644 index 00000000..efdd12b3 --- /dev/null +++ b/tests/store/test_v0_45_0_migration.py @@ -0,0 +1,150 @@ +"""Tests for the v0.45.0 picture-data backfill migration. + +The migration walks every document, decodes picture image URIs out of the +stored docling blob, writes the bytes to ``document_items.picture_data``, +and then strips the URIs from the blob (which lives separately on the +documents table). +""" + +import base64 +import json + +import pytest + +from haiku.rag.store import Store +from haiku.rag.store.compression import compress_json, decompress_json +from haiku.rag.store.engine import DocumentItemRecord, DocumentRecord +from haiku.rag.store.upgrades.v0_45_0 import _apply_extract_picture_bytes + +PNG_BYTES = b"\x89PNG\r\n\x1a\nfake-picture-bytes" +PNG_DATA_URI = f"data:image/png;base64,{base64.b64encode(PNG_BYTES).decode('ascii')}" + + +def _docling_blob_with_picture(self_ref: str = "#/pictures/0") -> bytes: + """Build a compressed docling-document blob carrying one picture with + an inline data URI — the shape v0.45.0 was written to backfill from.""" + doc = { + "schema_name": "DoclingDocument", + "version": "1.10.0", + "name": "test", + "pictures": [ + { + "self_ref": self_ref, + "image": {"mimetype": "image/png", "uri": PNG_DATA_URI}, + } + ], + "tables": [], + "texts": [], + "groups": [], + "body": {"self_ref": "#/body", "children": [], "label": "unspecified"}, + "furniture": { + "self_ref": "#/furniture", + "children": [], + "label": "unspecified", + }, + } + return compress_json(json.dumps(doc)) + + +@pytest.mark.asyncio +async def test_migration_backfills_picture_bytes_and_strips_blob(temp_db_path): + """Happy path: doc has a picture with bytes inline in the blob plus a + matching items row (with picture_data NULL). The migration writes the + bytes onto the items row and clears the URI from the blob.""" + async with Store(temp_db_path, create=True) as store: + doc_id = "doc-1" + # Insert a doc whose blob carries a picture data URI. + await store.documents_table.add( + [ + DocumentRecord( + id=doc_id, + content="x", + docling_document=_docling_blob_with_picture(), + docling_version="1.10.0", + ) + ] + ) + # Insert a matching items row, picture_data deliberately empty. + await store.document_items_table.add( + [ + DocumentItemRecord( + document_id=doc_id, + position=0, + self_ref="#/pictures/0", + label="picture", + text="", + page_numbers="[1]", + picture_data=None, + ) + ] + ) + + await _apply_extract_picture_bytes(store) + + # picture_data is populated on the items row. + rows = await ( + store.document_items_table.query() + .where(f"document_id = '{doc_id}' AND self_ref = '#/pictures/0'") + .to_list() + ) + assert len(rows) == 1 + assert rows[0]["picture_data"] == PNG_BYTES + + # The docling blob no longer carries the data URI on the picture. + doc_rows = await ( + store.documents_table.query() + .where(f"id = '{doc_id}'") + .select(["docling_document"]) + .to_list() + ) + blob = json.loads(decompress_json(doc_rows[0]["docling_document"])) + assert blob["pictures"][0]["image"] is None + + +@pytest.mark.asyncio +async def test_migration_is_idempotent(temp_db_path): + """Running the migration twice on the same DB is a no-op the second + time — the blob is already stripped.""" + async with Store(temp_db_path, create=True) as store: + doc_id = "doc-1" + await store.documents_table.add( + [ + DocumentRecord( + id=doc_id, + content="x", + docling_document=_docling_blob_with_picture(), + docling_version="1.10.0", + ) + ] + ) + await store.document_items_table.add( + [ + DocumentItemRecord( + document_id=doc_id, + position=0, + self_ref="#/pictures/0", + label="picture", + text="", + page_numbers="[1]", + ) + ] + ) + + await _apply_extract_picture_bytes(store) + # Capture state after first run. + rows_first = await ( + store.document_items_table.query() + .where(f"document_id = '{doc_id}'") + .to_list() + ) + first_bytes = rows_first[0]["picture_data"] + + # Re-run. + await _apply_extract_picture_bytes(store) + + rows_second = await ( + store.document_items_table.query() + .where(f"document_id = '{doc_id}'") + .to_list() + ) + assert rows_second[0]["picture_data"] == first_bytes