Merge pull request #227 from ggozad/feat/docling-doc-compression

Fix large storage overflow by changing docling_document_json (str) to docling_document (compressed bytes)
This commit is contained in:
Yiorgis Gozadinos 2026-01-12 11:57:31 +02:00 committed by GitHub
commit 119ca84da8
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
16 changed files with 380 additions and 80 deletions

View file

@ -1,6 +1,19 @@
# Changelog # Changelog
## [Unreleased] ## [Unreleased]
### Fixed
- **Large Document Storage Overflow**: Fixed "byte array offset overflow" panic when vacuuming/rebuilding databases with many large PDF documents ([#225](https://github.com/ggozad/haiku.rag/issues/225))
- Root cause: Arrow's 32-bit string column offsets limited to ~2GB per fragment
- Changed `docling_document_json` (string) to `docling_document` (bytes) with `large_binary` Arrow type (64-bit offsets)
- Added gzip compression for DoclingDocument JSON (~1.4x compression ratio)
- Migration automatically compresses existing documents in batches to avoid memory issues
- **Breaking**: Migration is destructive - all table version history is lost after upgrade
### Changed
- **Dependencies**: Updated lancedb 0.26.0 → 0.26.1, docling 2.65.0 → 2.67.0
## [0.24.2] - 2026-01-08 ## [0.24.2] - 2026-01-08
### Fixed ### Fixed

View file

@ -17,6 +17,7 @@ import httpx
from haiku.rag.config import AppConfig, Config from haiku.rag.config import AppConfig, Config
from haiku.rag.converters import get_converter from haiku.rag.converters import get_converter
from haiku.rag.reranking import get_reranker from haiku.rag.reranking import get_reranker
from haiku.rag.store.compression import compress_json
from haiku.rag.store.engine import Store from haiku.rag.store.engine import Store
from haiku.rag.store.models.chunk import Chunk, SearchResult from haiku.rag.store.models.chunk import Chunk, SearchResult
from haiku.rag.store.models.document import Document from haiku.rag.store.models.document import Document
@ -372,7 +373,7 @@ class HaikuRAG:
embedded_chunks = await embed_chunks(chunks, self._config) embedded_chunks = await embed_chunks(chunks, self._config)
# Store markdown export as content for better display/readability # Store markdown export as content for better display/readability
# The original content is preserved in docling_document_json # The original content is preserved in docling_document
stored_content = docling_document.export_to_markdown() stored_content = docling_document.export_to_markdown()
# Create document model # Create document model
@ -381,7 +382,7 @@ class HaikuRAG:
uri=uri, uri=uri,
title=title, title=title,
metadata=metadata or {}, metadata=metadata or {},
docling_document_json=docling_document.model_dump_json(), docling_document=compress_json(docling_document.model_dump_json()),
docling_version=docling_document.version, docling_version=docling_document.version,
) )
@ -417,7 +418,7 @@ class HaikuRAG:
uri=uri, uri=uri,
title=title, title=title,
metadata=metadata or {}, metadata=metadata or {},
docling_document_json=docling_document.model_dump_json(), docling_document=compress_json(docling_document.model_dump_json()),
docling_version=docling_document.version, docling_version=docling_document.version,
) )
@ -550,7 +551,9 @@ class HaikuRAG:
# Update existing document and rechunk # Update existing document and rechunk
existing_doc.content = docling_document.export_to_markdown() existing_doc.content = docling_document.export_to_markdown()
existing_doc.metadata = metadata existing_doc.metadata = metadata
existing_doc.docling_document_json = docling_document.model_dump_json() existing_doc.docling_document = compress_json(
docling_document.model_dump_json()
)
existing_doc.docling_version = docling_document.version existing_doc.docling_version = docling_document.version
if title is not None: if title is not None:
existing_doc.title = title existing_doc.title = title
@ -564,7 +567,7 @@ class HaikuRAG:
uri=uri, uri=uri,
title=title, title=title,
metadata=metadata, metadata=metadata,
docling_document_json=docling_document.model_dump_json(), docling_document=compress_json(docling_document.model_dump_json()),
docling_version=docling_document.version, docling_version=docling_document.version,
) )
return await self._store_document_with_chunks(document, embedded_chunks) return await self._store_document_with_chunks(document, embedded_chunks)
@ -657,7 +660,9 @@ class HaikuRAG:
# Update existing document and rechunk # Update existing document and rechunk
existing_doc.content = docling_document.export_to_markdown() existing_doc.content = docling_document.export_to_markdown()
existing_doc.metadata = metadata existing_doc.metadata = metadata
existing_doc.docling_document_json = docling_document.model_dump_json() existing_doc.docling_document = compress_json(
docling_document.model_dump_json()
)
existing_doc.docling_version = docling_document.version existing_doc.docling_version = docling_document.version
if title is not None: if title is not None:
existing_doc.title = title existing_doc.title = title
@ -671,7 +676,7 @@ class HaikuRAG:
uri=url, uri=url,
title=title, title=title,
metadata=metadata, metadata=metadata,
docling_document_json=docling_document.model_dump_json(), docling_document=compress_json(docling_document.model_dump_json()),
docling_version=docling_document.version, docling_version=docling_document.version,
) )
return await self._store_document_with_chunks(document, embedded_chunks) return await self._store_document_with_chunks(document, embedded_chunks)
@ -788,7 +793,9 @@ class HaikuRAG:
# Store docling data if provided # Store docling data if provided
if docling_document is not None: if docling_document is not None:
existing_doc.content = docling_document.export_to_markdown() existing_doc.content = docling_document.export_to_markdown()
existing_doc.docling_document_json = docling_document.model_dump_json() existing_doc.docling_document = compress_json(
docling_document.model_dump_json()
)
existing_doc.docling_version = docling_document.version existing_doc.docling_version = docling_document.version
elif content is not None: elif content is not None:
existing_doc.content = content existing_doc.content = content
@ -798,7 +805,9 @@ class HaikuRAG:
# DoclingDocument provided without chunks - chunk and embed using primitives # DoclingDocument provided without chunks - chunk and embed using primitives
if docling_document is not None: if docling_document is not None:
existing_doc.content = docling_document.export_to_markdown() existing_doc.content = docling_document.export_to_markdown()
existing_doc.docling_document_json = docling_document.model_dump_json() existing_doc.docling_document = compress_json(
docling_document.model_dump_json()
)
existing_doc.docling_version = docling_document.version existing_doc.docling_version = docling_document.version
new_chunks = await self.chunk(docling_document) new_chunks = await self.chunk(docling_document)
@ -810,7 +819,9 @@ class HaikuRAG:
# Content provided without chunks - convert, chunk, and embed using primitives # Content provided without chunks - convert, chunk, and embed using primitives
existing_doc.content = content # type: ignore[assignment] existing_doc.content = content # type: ignore[assignment]
converted_docling = await self.convert(existing_doc.content) converted_docling = await self.convert(existing_doc.content)
existing_doc.docling_document_json = converted_docling.model_dump_json() existing_doc.docling_document = compress_json(
converted_docling.model_dump_json()
)
existing_doc.docling_version = converted_docling.version existing_doc.docling_version = converted_docling.version
new_chunks = await self.chunk(converted_docling) new_chunks = await self.chunk(converted_docling)
@ -1485,7 +1496,7 @@ class HaikuRAG:
uri=doc.uri, uri=doc.uri,
title=doc.title, title=doc.title,
metadata=json.dumps(doc.metadata), metadata=json.dumps(doc.metadata),
docling_document_json=doc.docling_document_json, docling_document=doc.docling_document,
docling_version=doc.docling_version, docling_version=doc.docling_version,
created_at=doc.created_at.isoformat() if doc.created_at else now, created_at=doc.created_at.isoformat() if doc.created_at else now,
updated_at=now, updated_at=now,
@ -1523,7 +1534,7 @@ class HaikuRAG:
embedded_chunks = await embed_chunks(chunks, self._config) embedded_chunks = await embed_chunks(chunks, self._config)
# Update document fields # Update document fields
doc.docling_document_json = docling_document.model_dump_json() doc.docling_document = compress_json(docling_document.model_dump_json())
doc.docling_version = docling_document.version doc.docling_version = docling_document.version
# Prepare chunks with document_id and order # Prepare chunks with document_id and order
@ -1602,7 +1613,7 @@ class HaikuRAG:
chunks = await self.chunk(docling_document) chunks = await self.chunk(docling_document)
embedded_chunks = await embed_chunks(chunks, self._config) embedded_chunks = await embed_chunks(chunks, self._config)
doc.docling_document_json = docling_document.model_dump_json() doc.docling_document = compress_json(docling_document.model_dump_json())
doc.docling_version = docling_document.version doc.docling_version = docling_document.version
# Prepare chunks with document_id and order # Prepare chunks with document_id and order

View file

@ -0,0 +1,11 @@
import gzip
def compress_json(json_str: str) -> bytes:
"""Compress a JSON string with gzip."""
return gzip.compress(json_str.encode("utf-8"))
def decompress_json(data: bytes) -> str:
"""Decompress gzip-compressed data to a JSON string."""
return gzip.decompress(data).decode("utf-8")

View file

@ -8,6 +8,7 @@ from typing import Any
from uuid import uuid4 from uuid import uuid4
import lancedb import lancedb
import pyarrow as pa
from lancedb.pydantic import LanceModel, Vector from lancedb.pydantic import LanceModel, Vector
from pydantic import Field from pydantic import Field
@ -24,12 +25,33 @@ class DocumentRecord(LanceModel):
uri: str | None = None uri: str | None = None
title: str | None = None title: str | None = None
metadata: str = Field(default="{}") metadata: str = Field(default="{}")
docling_document_json: str | None = None docling_document: bytes | None = None
docling_version: str | None = None docling_version: str | None = None
created_at: str = Field(default_factory=lambda: "") created_at: str = Field(default_factory=lambda: "")
updated_at: str = Field(default_factory=lambda: "") updated_at: str = Field(default_factory=lambda: "")
def get_documents_arrow_schema() -> pa.Schema:
"""Generate Arrow schema for documents table with large_binary for docling_document.
LanceDB maps Python `bytes` to Arrow's `binary` type, which uses 32-bit offsets
and is limited to ~2GB per column in a fragment. When many large documents
(with embedded page images) are grouped in a single fragment, this limit is
exceeded, causing "byte array offset overflow" panics.
This function overrides the default mapping to use `large_binary` instead,
which has 64-bit offsets and no practical size limit.
"""
base_schema = DocumentRecord.to_arrow_schema()
fields = []
for field in base_schema:
if field.name == "docling_document":
fields.append(pa.field("docling_document", pa.large_binary()))
else:
fields.append(field)
return pa.schema(fields)
def create_chunk_model(vector_dim: int): def create_chunk_model(vector_dim: int):
"""Create a ChunkRecord model with the specified vector dimension. """Create a ChunkRecord model with the specified vector dimension.
@ -281,7 +303,7 @@ class Store:
self.documents_table = self.db.open_table("documents") self.documents_table = self.db.open_table("documents")
else: else:
self.documents_table = self.db.create_table( self.documents_table = self.db.create_table(
"documents", schema=DocumentRecord "documents", schema=get_documents_arrow_schema()
) )
# Create or get chunks table # Create or get chunks table

View file

@ -4,6 +4,8 @@ from typing import TYPE_CHECKING
from cachetools import LRUCache from cachetools import LRUCache
from pydantic import BaseModel, Field from pydantic import BaseModel, Field
from haiku.rag.store.compression import decompress_json
if TYPE_CHECKING: if TYPE_CHECKING:
from docling_core.types.doc.document import DoclingDocument from docling_core.types.doc.document import DoclingDocument
@ -11,13 +13,16 @@ if TYPE_CHECKING:
_docling_document_cache: LRUCache[str, "DoclingDocument"] = LRUCache(maxsize=100) _docling_document_cache: LRUCache[str, "DoclingDocument"] = LRUCache(maxsize=100)
def _get_cached_docling_document(document_id: str, json_str: str) -> "DoclingDocument": def _get_cached_docling_document(
document_id: str, compressed_data: bytes
) -> "DoclingDocument":
"""Get or parse DoclingDocument with LRU caching by document ID.""" """Get or parse DoclingDocument with LRU caching by document ID."""
if document_id in _docling_document_cache: if document_id in _docling_document_cache:
return _docling_document_cache[document_id] return _docling_document_cache[document_id]
from docling_core.types.doc.document import DoclingDocument from docling_core.types.doc.document import DoclingDocument
json_str = decompress_json(compressed_data)
doc = DoclingDocument.model_validate_json(json_str) doc = DoclingDocument.model_validate_json(json_str)
_docling_document_cache[document_id] = doc _docling_document_cache[document_id] = doc
return doc return doc
@ -38,7 +43,7 @@ class Document(BaseModel):
uri: str | None = None uri: str | None = None
title: str | None = None title: str | None = None
metadata: dict = {} metadata: dict = {}
docling_document_json: str | None = None docling_document: bytes | None = None
docling_version: str | None = None docling_version: str | None = None
created_at: datetime = Field(default_factory=datetime.now) created_at: datetime = Field(default_factory=datetime.now)
updated_at: datetime = Field(default_factory=datetime.now) updated_at: datetime = Field(default_factory=datetime.now)
@ -51,13 +56,14 @@ class Document(BaseModel):
Returns: Returns:
The parsed DoclingDocument, or None if not stored or no ID. The parsed DoclingDocument, or None if not stored or no ID.
""" """
if self.docling_document_json is None: if self.docling_document is None:
return None return None
# No caching for documents without ID # No caching for documents without ID
if self.id is None: if self.id is None:
from docling_core.types.doc.document import DoclingDocument from docling_core.types.doc.document import DoclingDocument
return DoclingDocument.model_validate_json(self.docling_document_json) json_str = decompress_json(self.docling_document)
return DoclingDocument.model_validate_json(json_str)
return _get_cached_docling_document(self.id, self.docling_document_json) return _get_cached_docling_document(self.id, self.docling_document)

View file

@ -2,7 +2,7 @@ import json
from datetime import datetime from datetime import datetime
from uuid import uuid4 from uuid import uuid4
from haiku.rag.store.engine import DocumentRecord, Store from haiku.rag.store.engine import DocumentRecord, Store, get_documents_arrow_schema
from haiku.rag.store.models.document import Document from haiku.rag.store.models.document import Document
@ -35,7 +35,7 @@ class DocumentRepository:
uri=record.uri, uri=record.uri,
title=record.title, title=record.title,
metadata=json.loads(record.metadata), metadata=json.loads(record.metadata),
docling_document_json=record.docling_document_json, docling_document=record.docling_document,
docling_version=record.docling_version, docling_version=record.docling_version,
created_at=datetime.fromisoformat(record.created_at) created_at=datetime.fromisoformat(record.created_at)
if record.created_at if record.created_at
@ -61,7 +61,7 @@ class DocumentRepository:
uri=entity.uri, uri=entity.uri,
title=entity.title, title=entity.title,
metadata=json.dumps(entity.metadata), metadata=json.dumps(entity.metadata),
docling_document_json=entity.docling_document_json, docling_document=entity.docling_document,
docling_version=entity.docling_version, docling_version=entity.docling_version,
created_at=now, created_at=now,
updated_at=now, updated_at=now,
@ -111,7 +111,7 @@ class DocumentRepository:
"uri": entity.uri, "uri": entity.uri,
"title": entity.title, "title": entity.title,
"metadata": json.dumps(entity.metadata), "metadata": json.dumps(entity.metadata),
"docling_document_json": entity.docling_document_json, "docling_document": entity.docling_document,
"docling_version": entity.docling_version, "docling_version": entity.docling_version,
"updated_at": now, "updated_at": now,
}, },
@ -198,5 +198,5 @@ class DocumentRepository:
# Drop and recreate table to clear all data # Drop and recreate table to clear all data
self.store.db.drop_table("documents") self.store.db.drop_table("documents")
self.store.documents_table = self.store.db.create_table( self.store.documents_table = self.store.db.create_table(
"documents", schema=DocumentRecord "documents", schema=get_documents_arrow_schema()
) )

View file

@ -69,6 +69,9 @@ from haiku.rag.store.upgrades.v0_20_0 import (
from haiku.rag.store.upgrades.v0_23_1 import ( from haiku.rag.store.upgrades.v0_23_1 import (
upgrade_contextualize_chunks as upgrade_0_23_1_contextualize, upgrade_contextualize_chunks as upgrade_0_23_1_contextualize,
) )
from haiku.rag.store.upgrades.v0_25_0 import (
upgrade_compress_docling_document as upgrade_0_25_0_compress,
)
upgrades.append(upgrade_0_9_3_order) upgrades.append(upgrade_0_9_3_order)
upgrades.append(upgrade_0_9_3_fts) upgrades.append(upgrade_0_9_3_fts)
@ -76,3 +79,4 @@ upgrades.append(upgrade_0_10_1_add_title)
upgrades.append(upgrade_0_19_6_embeddings) upgrades.append(upgrade_0_19_6_embeddings)
upgrades.append(upgrade_0_20_0_docling) upgrades.append(upgrade_0_20_0_docling)
upgrades.append(upgrade_0_23_1_contextualize) upgrades.append(upgrade_0_23_1_contextualize)
upgrades.append(upgrade_0_25_0_compress)

View file

@ -0,0 +1,200 @@
import json
import logging
from datetime import timedelta
import pyarrow as pa
from lancedb.pydantic import LanceModel
from pydantic import Field
from haiku.rag.store.compression import compress_json, decompress_json
from haiku.rag.store.engine import Store
from haiku.rag.store.upgrades import Upgrade
logger = logging.getLogger(__name__)
BATCH_SIZE = 10
def _apply_compress_docling_document(store: Store) -> None: # pragma: no cover
"""Migrate docling_document_json (str) to docling_document (compressed bytes)."""
class DocumentRecordV4(LanceModel):
id: str
content: str
uri: str | None = None
title: str | None = None
metadata: str = Field(default="{}")
docling_document: bytes | None = None
docling_version: str | None = None
created_at: str = Field(default_factory=lambda: "")
updated_at: str = Field(default_factory=lambda: "")
def get_documents_arrow_schema_v4() -> pa.Schema:
"""Generate Arrow schema with large_binary for docling_document."""
base_schema = DocumentRecordV4.to_arrow_schema()
fields = []
for field in base_schema:
if field.name == "docling_document":
fields.append(pa.field("docling_document", pa.large_binary()))
else:
fields.append(field)
return pa.schema(fields)
def migrate_row(row: dict) -> DocumentRecordV4:
"""Migrate a single row, compressing docling_document."""
docling_json = row.get("docling_document_json") or row.get("docling_document")
docling_bytes: bytes | None = None
if docling_json:
if isinstance(docling_json, str):
docling_bytes = compress_json(docling_json)
elif isinstance(docling_json, bytes):
try:
decompress_json(docling_json)
docling_bytes = docling_json # Already compressed
except Exception:
docling_bytes = compress_json(docling_json.decode("utf-8"))
metadata_raw = row.get("metadata")
metadata_str = (
metadata_raw
if isinstance(metadata_raw, str)
else json.dumps(metadata_raw or {})
)
return DocumentRecordV4(
id=row.get("id") or "",
content=row.get("content", ""),
uri=row.get("uri"),
title=row.get("title"),
metadata=metadata_str,
docling_document=docling_bytes,
docling_version=row.get("docling_version"),
created_at=row.get("created_at", ""),
updated_at=row.get("updated_at", ""),
)
# First pass: collect document IDs to process
try:
ids = [
row["id"]
for row in store.documents_table.search()
.select(["id"])
.to_arrow()
.to_pylist()
]
except Exception:
ids = []
if not ids:
# No documents, just recreate table with new schema
try:
store.db.drop_table("documents")
except Exception:
pass
store.documents_table = store.db.create_table(
"documents", schema=get_documents_arrow_schema_v4()
)
return
# Create staging table with new schema
try:
store.db.drop_table("documents_v4_staging")
except Exception:
pass
staging_table = store.db.create_table(
"documents_v4_staging", schema=get_documents_arrow_schema_v4()
)
# Migrate in batches: read from old, compress, write to staging
total_docs = len(ids)
total_batches = (total_docs + BATCH_SIZE - 1) // BATCH_SIZE
logger.info("Compressing %d documents in %d batches", total_docs, total_batches)
for batch_num, i in enumerate(range(0, len(ids), BATCH_SIZE), 1):
batch_ids = ids[i : i + BATCH_SIZE]
id_list = ", ".join(f"'{id}'" for id in batch_ids)
batch = (
store.documents_table.search()
.where(f"id IN ({id_list})")
.to_arrow()
.to_pylist()
)
migrated_batch = [migrate_row(row) for row in batch]
if migrated_batch:
staging_table.add(migrated_batch)
logger.info(
"Compressed batch %d/%d (%d documents)",
batch_num,
total_batches,
len(migrated_batch),
)
# Replace old table with staging table
try:
store.db.drop_table("documents")
except Exception:
pass
store.documents_table = store.db.create_table(
"documents", schema=get_documents_arrow_schema_v4()
)
# Copy from staging to final table in batches
staging_ids = [
row["id"]
for row in staging_table.search().select(["id"]).to_arrow().to_pylist()
]
logger.info("Copying %d documents to new table", len(staging_ids))
for batch_num, i in enumerate(range(0, len(staging_ids), BATCH_SIZE), 1):
batch_ids = staging_ids[i : i + BATCH_SIZE]
id_list = ", ".join(f"'{id}'" for id in batch_ids)
batch = (
staging_table.search().where(f"id IN ({id_list})").to_arrow().to_pylist()
)
records = [
DocumentRecordV4(
id=row["id"],
content=row["content"],
uri=row["uri"],
title=row["title"],
metadata=row["metadata"],
docling_document=row["docling_document"],
docling_version=row["docling_version"],
created_at=row["created_at"],
updated_at=row["updated_at"],
)
for row in batch
]
if records:
store.documents_table.add(records)
logger.info("Copied batch %d/%d", batch_num, total_batches)
# Cleanup staging table
try:
store.db.drop_table("documents_v4_staging")
except Exception:
pass
# Vacuum all tables (destructive migration, no history preserved)
logger.info("Vacuuming database")
for table in [store.documents_table, store.chunks_table, store.settings_table]:
try:
table.optimize(cleanup_older_than=timedelta(seconds=0))
except Exception:
pass
logger.info("Migration complete")
upgrade_compress_docling_document = Upgrade(
version="0.25.0",
apply=_apply_compress_docling_document,
description="Compress docling_document with gzip and use large_binary type",
)

View file

@ -2,7 +2,7 @@
name = "haiku.rag-slim" name = "haiku.rag-slim"
description = "Opinionated agentic RAG powered by LanceDB, Pydantic AI, and Docling - Minimal dependencies" description = "Opinionated agentic RAG powered by LanceDB, Pydantic AI, and Docling - Minimal dependencies"
version = "0.24.2" version = "0.25.0"
authors = [{ name = "Yiorgis Gozadinos", email = "ggozadinos@gmail.com" }] authors = [{ name = "Yiorgis Gozadinos", email = "ggozadinos@gmail.com" }]
license = { text = "MIT" } license = { text = "MIT" }
readme = { file = "README.md", content-type = "text/markdown" } readme = { file = "README.md", content-type = "text/markdown" }
@ -24,7 +24,7 @@ classifiers = [
dependencies = [ dependencies = [
"docling-core==2.57.0", "docling-core==2.57.0",
"httpx>=0.28.1", "httpx>=0.28.1",
"lancedb==0.26.0", "lancedb==0.26.1",
"pathspec>=0.12.1", "pathspec>=0.12.1",
"pydantic>=2.12.5", "pydantic>=2.12.5",
"pydantic-ai-slim[openai,fastmcp,logfire,ag-ui]>=1.39.0", "pydantic-ai-slim[openai,fastmcp,logfire,ag-ui]>=1.39.0",
@ -37,7 +37,7 @@ dependencies = [
[project.optional-dependencies] [project.optional-dependencies]
# Document processing # Document processing
docling = ["docling==2.65.0", "opencv-python-headless>=4.12.0.88"] docling = ["docling==2.67.0", "opencv-python-headless>=4.12.0.88"]
# Embedding providers # Embedding providers
voyageai = ["voyageai>=0.3.7"] voyageai = ["voyageai>=0.3.7"]
# Rerankers # Rerankers

View file

@ -0,0 +1,15 @@
from haiku.rag.store.compression import compress_json, decompress_json
class TestJsonCompression:
def test_compress_json_roundtrip(self):
json_str = '{"key": "value", "number": 42, "nested": {"a": 1}}'
compressed = compress_json(json_str)
decompressed = decompress_json(compressed)
assert decompressed == json_str
def test_compress_json_with_unicode(self):
json_str = '{"message": "Hello, 世界! 🌍"}'
compressed = compress_json(json_str)
decompressed = decompress_json(compressed)
assert decompressed == json_str

View file

@ -8,6 +8,7 @@ from datasets import Dataset
from haiku.rag.client import HaikuRAG from haiku.rag.client import HaikuRAG
from haiku.rag.config import Config from haiku.rag.config import Config
from haiku.rag.store.compression import decompress_json
from haiku.rag.store.models.chunk import Chunk from haiku.rag.store.models.chunk import Chunk
from haiku.rag.store.models.document import Document from haiku.rag.store.models.document import Document
@ -790,13 +791,15 @@ async def test_client_create_document_stores_docling_json(temp_db_path):
) )
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
assert doc.docling_version is not None assert doc.docling_version is not None
# Verify JSON is valid and can be parsed # Verify JSON is valid and can be parsed
import json import json
parsed = json.loads(doc.docling_document_json) from haiku.rag.store.compression import decompress_json
parsed = json.loads(decompress_json(doc.docling_document))
assert "version" in parsed assert "version" in parsed
assert parsed["version"] == doc.docling_version assert parsed["version"] == doc.docling_version
@ -824,7 +827,8 @@ async def test_client_import_document_stores_docling_data(temp_db_path):
assert doc.id is not None assert doc.id is not None
assert "Content from docling document" in doc.content assert "Content from docling document" in doc.content
assert doc.docling_document_json == docling_doc.model_dump_json() assert doc.docling_document is not None
assert decompress_json(doc.docling_document) == docling_doc.model_dump_json()
assert doc.docling_version == docling_doc.version assert doc.docling_version == docling_doc.version
@ -840,13 +844,13 @@ async def test_client_create_document_from_file_stores_docling_json(temp_db_path
assert isinstance(doc, Document) assert isinstance(doc, Document)
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
assert doc.docling_version is not None assert doc.docling_version is not None
# Verify the stored document also has the JSON # Verify the stored document also has the JSON
retrieved = await client.get_document_by_id(doc.id) retrieved = await client.get_document_by_id(doc.id)
assert retrieved is not None assert retrieved is not None
assert retrieved.docling_document_json == doc.docling_document_json assert retrieved.docling_document == doc.docling_document
assert retrieved.docling_version == doc.docling_version assert retrieved.docling_version == doc.docling_version
@ -857,17 +861,17 @@ async def test_client_update_document_stores_docling_json(temp_db_path):
# Create initial document # Create initial document
doc = await client.create_document(content="Initial content") doc = await client.create_document(content="Initial content")
assert doc.id is not None assert doc.id is not None
original_json = doc.docling_document_json original_json = doc.docling_document
# Update content via update_document # Update content via update_document
updated_doc = await client.update_document( updated_doc = await client.update_document(
document_id=doc.id, content="New content via fields update" document_id=doc.id, content="New content via fields update"
) )
assert updated_doc.docling_document_json is not None assert updated_doc.docling_document is not None
assert updated_doc.docling_version is not None assert updated_doc.docling_version is not None
# JSON should be different because content changed # JSON should be different because content changed
assert updated_doc.docling_document_json != original_json assert updated_doc.docling_document != original_json
@pytest.mark.vcr() @pytest.mark.vcr()
@ -879,7 +883,7 @@ async def test_client_update_document_with_custom_chunks_no_docling_json(
# Create initial document # Create initial document
doc = await client.create_document(content="Initial content") doc = await client.create_document(content="Initial content")
assert doc.id is not None assert doc.id is not None
original_json = doc.docling_document_json original_json = doc.docling_document
# Update with custom chunks # Update with custom chunks
custom_chunks = [Chunk(content="Custom chunk", order=0)] custom_chunks = [Chunk(content="Custom chunk", order=0)]
@ -888,7 +892,7 @@ async def test_client_update_document_with_custom_chunks_no_docling_json(
) )
# Docling JSON should remain unchanged (no conversion when custom chunks provided) # Docling JSON should remain unchanged (no conversion when custom chunks provided)
assert updated_doc.docling_document_json == original_json assert updated_doc.docling_document == original_json
@pytest.mark.vcr() @pytest.mark.vcr()
@ -942,7 +946,11 @@ async def test_client_update_document_with_docling_rechunks(temp_db_path):
# Content should be extracted from docling document # Content should be extracted from docling document
assert "Completely different text" in updated_doc.content assert "Completely different text" in updated_doc.content
assert updated_doc.docling_document_json == docling_doc.model_dump_json() assert updated_doc.docling_document is not None
assert (
decompress_json(updated_doc.docling_document)
== docling_doc.model_dump_json()
)
assert updated_doc.docling_version == docling_doc.version assert updated_doc.docling_version == docling_doc.version
# Chunks should be regenerated # Chunks should be regenerated
@ -981,7 +989,11 @@ async def test_client_update_document_docling_with_chunks(temp_db_path):
# Content should be extracted from docling (since content wasn't provided) # Content should be extracted from docling (since content wasn't provided)
assert "Text from docling" in updated_doc.content assert "Text from docling" in updated_doc.content
assert updated_doc.docling_document_json == docling_doc.model_dump_json() assert updated_doc.docling_document is not None
assert (
decompress_json(updated_doc.docling_document)
== docling_doc.model_dump_json()
)
# Custom chunks should be used (not rechunked from docling) # Custom chunks should be used (not rechunked from docling)
chunks = await client.chunk_repository.get_by_document_id(doc.id) chunks = await client.chunk_repository.get_by_document_id(doc.id)
@ -1001,7 +1013,7 @@ async def test_client_file_update_stores_docling_json(temp_db_path):
# Create initial document # Create initial document
doc1 = await client.create_document_from_source(temp_path) doc1 = await client.create_document_from_source(temp_path)
assert isinstance(doc1, Document) assert isinstance(doc1, Document)
original_json = doc1.docling_document_json original_json = doc1.docling_document
original_version = doc1.docling_version original_version = doc1.docling_version
# Modify file # Modify file
@ -1013,8 +1025,8 @@ async def test_client_file_update_stores_docling_json(temp_db_path):
assert doc2.id == doc1.id # Same document assert doc2.id == doc1.id # Same document
# Docling JSON should be updated # Docling JSON should be updated
assert doc2.docling_document_json is not None assert doc2.docling_document is not None
assert doc2.docling_document_json != original_json assert doc2.docling_document != original_json
assert doc2.docling_version == original_version # Version stays same assert doc2.docling_version == original_version # Version stays same
@ -1038,7 +1050,7 @@ async def test_client_visualize_chunk_no_bounding_boxes(temp_db_path):
) )
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
chunks = await client.chunk_repository.get_by_document_id(doc.id) chunks = await client.chunk_repository.get_by_document_id(doc.id)
assert len(chunks) >= 1 assert len(chunks) >= 1
@ -1095,7 +1107,7 @@ async def test_client_visualize_chunk_with_pdf(temp_db_path):
doc = await client.create_document_from_source(pdf_path) doc = await client.create_document_from_source(pdf_path)
assert isinstance(doc, Document) assert isinstance(doc, Document)
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
chunks = await client.chunk_repository.get_by_document_id(doc.id) chunks = await client.chunk_repository.get_by_document_id(doc.id)
assert len(chunks) > 0 assert len(chunks) > 0
@ -1345,7 +1357,7 @@ async def test_client_create_document_with_html_format(temp_db_path):
) )
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
# Verify the DoclingDocument has proper structure # Verify the DoclingDocument has proper structure
docling_doc = doc.get_docling_document() docling_doc = doc.get_docling_document()

View file

@ -613,7 +613,7 @@ This is paragraph four about topic C.
) )
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
# Get chunks which should have doc_item_refs # Get chunks which should have doc_item_refs
chunks = await client.chunk_repository.get_by_document_id(doc.id) chunks = await client.chunk_repository.get_by_document_id(doc.id)

View file

@ -20,7 +20,7 @@ def vcr_cassette_dir():
return str(Path(__file__).parent / "cassettes" / "test_converters") return str(Path(__file__).parent / "cassettes" / "test_converters")
def create_mock_docling_document_json(name: str = "test") -> dict: def create_mock_docling_document(name: str = "test") -> dict:
"""Create a minimal valid DoclingDocument JSON structure for mocking.""" """Create a minimal valid DoclingDocument JSON structure for mocking."""
return { return {
"schema_name": "DoclingDocument", "schema_name": "DoclingDocument",
@ -476,7 +476,7 @@ class TestDoclingServeConverter:
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_convert_text_success(self, converter): async def test_convert_text_success(self, converter):
"""Test successful text conversion via docling-serve async workflow.""" """Test successful text conversion via docling-serve async workflow."""
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -498,7 +498,7 @@ class TestDoclingServeConverter:
config.providers.docling_serve.api_key = "test-key" config.providers.docling_serve.api_key = "test-key"
converter = DoclingServeConverter(config) converter = DoclingServeConverter(config)
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -527,7 +527,7 @@ class TestDoclingServeConverter:
config.processing.conversion_options.images_scale = 3.0 config.processing.conversion_options.images_scale = 3.0
converter = DoclingServeConverter(config) converter = DoclingServeConverter(config)
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -622,7 +622,7 @@ class TestDoclingServeConverter:
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_convert_file_pdf(self, converter): async def test_convert_file_pdf(self, converter):
"""Test converting PDF file via docling-serve async workflow.""" """Test converting PDF file via docling-serve async workflow."""
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -645,7 +645,7 @@ class TestDoclingServeConverter:
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_convert_file_text(self, converter): async def test_convert_file_text(self, converter):
"""Test converting text file (reads locally, sends to docling-serve).""" """Test converting text file (reads locally, sends to docling-serve)."""
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -734,7 +734,7 @@ class TestDoclingServeConverterPictureDescription:
config.prompts.picture_description = "Test prompt for picture description" config.prompts.picture_description = "Test prompt for picture description"
converter = DoclingServeConverter(config) converter = DoclingServeConverter(config)
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:
@ -767,7 +767,7 @@ class TestDoclingServeConverterPictureDescription:
"""Test that picture description is disabled by default.""" """Test that picture description is disabled by default."""
converter = DoclingServeConverter(config) converter = DoclingServeConverter(config)
doc_json = create_mock_docling_document_json("test") doc_json = create_mock_docling_document("test")
submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json) submit_resp, poll_resp, result_resp = create_async_workflow_mocks(doc_json)
with patch("httpx.AsyncClient") as mock_client_class: with patch("httpx.AsyncClient") as mock_client_class:

View file

@ -70,9 +70,11 @@ def test_document_get_docling_document():
import json import json
from haiku.rag.store.compression import compress_json
document = Document( document = Document(
content="Test content", content="Test content",
docling_document_json=json.dumps(doc_json), docling_document=compress_json(json.dumps(doc_json)),
docling_version="1.3.0", docling_version="1.3.0",
) )
@ -88,7 +90,7 @@ def test_document_get_docling_document_none():
"""Test get_docling_document returns None when not stored.""" """Test get_docling_document returns None when not stored."""
document = Document(content="Test content") document = Document(content="Test content")
assert document.docling_document_json is None assert document.docling_document is None
assert document.get_docling_document() is None assert document.get_docling_document() is None
@ -118,13 +120,15 @@ def test_document_get_docling_document_caching():
import json import json
json_str = json.dumps(doc_json) from haiku.rag.store.compression import compress_json
compressed = compress_json(json.dumps(doc_json))
# Clear cache to get clean state # Clear cache to get clean state
_docling_document_cache.clear() _docling_document_cache.clear()
document = Document( document = Document(
id="test-doc-id", content="Test content", docling_document_json=json_str id="test-doc-id", content="Test content", docling_document=compressed
) )
# First call - not in cache # First call - not in cache
@ -157,13 +161,15 @@ def test_document_get_docling_document_no_id_no_cache():
import json import json
json_str = json.dumps(doc_json) from haiku.rag.store.compression import compress_json
compressed = compress_json(json.dumps(doc_json))
# Clear cache # Clear cache
_docling_document_cache.clear() _docling_document_cache.clear()
# Document without ID # Document without ID
document = Document(content="Test content", docling_document_json=json_str) document = Document(content="Test content", docling_document=compressed)
doc1 = document.get_docling_document() doc1 = document.get_docling_document()
doc2 = document.get_docling_document() doc2 = document.get_docling_document()

View file

@ -10,7 +10,7 @@ async def test_rebuild_full(qa_corpus: Dataset, temp_db_path):
async with HaikuRAG(temp_db_path, create=True) as client: async with HaikuRAG(temp_db_path, create=True) as client:
doc = await client.create_document(content=qa_corpus["document_extracted"][0]) doc = await client.create_document(content=qa_corpus["document_extracted"][0])
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
chunks_before = await client.chunk_repository.get_by_document_id(doc.id) chunks_before = await client.chunk_repository.get_by_document_id(doc.id)
assert len(chunks_before) > 0 assert len(chunks_before) > 0
@ -23,7 +23,7 @@ async def test_rebuild_full(qa_corpus: Dataset, temp_db_path):
# Verify DoclingDocument JSON is preserved after rebuild # Verify DoclingDocument JSON is preserved after rebuild
doc_after = await client.document_repository.get_by_id(doc.id) doc_after = await client.document_repository.get_by_id(doc.id)
assert doc_after is not None assert doc_after is not None
assert doc_after.docling_document_json is not None assert doc_after.docling_document is not None
assert doc_after.docling_version is not None assert doc_after.docling_version is not None
chunks_after = await client.chunk_repository.get_by_document_id(doc.id) chunks_after = await client.chunk_repository.get_by_document_id(doc.id)
@ -40,7 +40,7 @@ async def test_rebuild_embed_only(qa_corpus: Dataset, temp_db_path):
async with HaikuRAG(temp_db_path, create=True) as client: async with HaikuRAG(temp_db_path, create=True) as client:
doc = await client.create_document(content=qa_corpus["document_extracted"][0]) doc = await client.create_document(content=qa_corpus["document_extracted"][0])
assert doc.id is not None assert doc.id is not None
original_docling_json = doc.docling_document_json original_docling_json = doc.docling_document
chunks_before = await client.chunk_repository.get_by_document_id(doc.id) chunks_before = await client.chunk_repository.get_by_document_id(doc.id)
assert len(chunks_before) > 0 assert len(chunks_before) > 0
@ -57,7 +57,7 @@ async def test_rebuild_embed_only(qa_corpus: Dataset, temp_db_path):
# DoclingDocument JSON should be unchanged (embed-only doesn't touch documents) # DoclingDocument JSON should be unchanged (embed-only doesn't touch documents)
doc_after = await client.document_repository.get_by_id(doc.id) doc_after = await client.document_repository.get_by_id(doc.id)
assert doc_after is not None assert doc_after is not None
assert doc_after.docling_document_json == original_docling_json assert doc_after.docling_document == original_docling_json
chunks_after = await client.chunk_repository.get_by_document_id(doc.id) chunks_after = await client.chunk_repository.get_by_document_id(doc.id)
chunk_ids_after = {c.id for c in chunks_after} chunk_ids_after = {c.id for c in chunks_after}
@ -112,7 +112,7 @@ async def test_rebuild_rechunk(qa_corpus: Dataset, temp_db_path):
async with HaikuRAG(temp_db_path, create=True) as client: async with HaikuRAG(temp_db_path, create=True) as client:
doc = await client.create_document(content=qa_corpus["document_extracted"][0]) doc = await client.create_document(content=qa_corpus["document_extracted"][0])
assert doc.id is not None assert doc.id is not None
assert doc.docling_document_json is not None assert doc.docling_document is not None
# Set a fake URI to simulate a document that came from a file # Set a fake URI to simulate a document that came from a file
doc.uri = "file:///nonexistent/path.txt" doc.uri = "file:///nonexistent/path.txt"
@ -133,7 +133,7 @@ async def test_rebuild_rechunk(qa_corpus: Dataset, temp_db_path):
doc_after = await client.document_repository.get_by_id(doc.id) doc_after = await client.document_repository.get_by_id(doc.id)
assert doc_after is not None assert doc_after is not None
assert doc_after.content == content_before assert doc_after.content == content_before
assert doc_after.docling_document_json is not None assert doc_after.docling_document is not None
assert doc_after.docling_version is not None assert doc_after.docling_version is not None
chunks_after = await client.chunk_repository.get_by_document_id(doc.id) chunks_after = await client.chunk_repository.get_by_document_id(doc.id)

26
uv.lock
View file

@ -741,7 +741,7 @@ wheels = [
[[package]] [[package]]
name = "docling" name = "docling"
version = "2.65.0" version = "2.67.0"
source = { registry = "https://pypi.org/simple" } source = { registry = "https://pypi.org/simple" }
dependencies = [ dependencies = [
{ name = "accelerate" }, { name = "accelerate" },
@ -773,9 +773,9 @@ dependencies = [
{ name = "tqdm" }, { name = "tqdm" },
{ name = "typer" }, { name = "typer" },
] ]
sdist = { url = "https://files.pythonhosted.org/packages/bb/dc/d42fb4dbe39578ff81d9c52b7a1faaf115c014a38a6e335e6db7ced80658/docling-2.65.0.tar.gz", hash = "sha256:64c6feccf808e19f6100aa9e2a6c71c992cd66e897cdfb33ccf15d0b62873b50", size = 257037, upload-time = "2025-12-15T16:56:16.609Z" } sdist = { url = "https://files.pythonhosted.org/packages/a7/eb/1900eaf576ab935d5b630a5ba6eb0e7063a2e1d36542df2f9ba572505d57/docling-2.67.0.tar.gz", hash = "sha256:d8c1992bbc090cee8c3f602fa63ac0c3127a65e3cabdf509d23ebb4e14feaf38", size = 259923, upload-time = "2026-01-09T08:37:37.984Z" }
wheels = [ wheels = [
{ url = "https://files.pythonhosted.org/packages/94/71/ba084670cc4cc3679f7ab650a32ff3ddbf61ad7a1eb559104ee819c028cf/docling-2.65.0-py3-none-any.whl", hash = "sha256:02b00096e2785c08924c9fa89034b918d90f0ee6625aacf20435d1d847c89d09", size = 275389, upload-time = "2025-12-15T16:56:15.003Z" }, { url = "https://files.pythonhosted.org/packages/8d/99/fa9ac2cffaa77cf46b6414d0a32490b2580c8f3753ab1c65d983ee27f3a0/docling-2.67.0-py3-none-any.whl", hash = "sha256:d3ea4c0d7a64fce4b8f8b38415e075f6637384eaadb27acbf5621673e0da7669", size = 277794, upload-time = "2026-01-09T08:37:36.52Z" },
] ]
[[package]] [[package]]
@ -1362,7 +1362,7 @@ requires-dist = [
[[package]] [[package]]
name = "haiku-rag-slim" name = "haiku-rag-slim"
version = "0.24.2" version = "0.25.0"
source = { editable = "haiku_rag_slim" } source = { editable = "haiku_rag_slim" }
dependencies = [ dependencies = [
{ name = "docling-core" }, { name = "docling-core" },
@ -1421,10 +1421,10 @@ zeroentropy = [
[package.metadata] [package.metadata]
requires-dist = [ requires-dist = [
{ name = "cohere", marker = "extra == 'cohere'", specifier = ">=5.20.1" }, { name = "cohere", marker = "extra == 'cohere'", specifier = ">=5.20.1" },
{ name = "docling", marker = "extra == 'docling'", specifier = "==2.65.0" }, { name = "docling", marker = "extra == 'docling'", specifier = "==2.67.0" },
{ name = "docling-core", specifier = "==2.57.0" }, { name = "docling-core", specifier = "==2.57.0" },
{ name = "httpx", specifier = ">=0.28.1" }, { name = "httpx", specifier = ">=0.28.1" },
{ name = "lancedb", specifier = "==0.26.0" }, { name = "lancedb", specifier = "==0.26.1" },
{ name = "mxbai-rerank", marker = "extra == 'mxbai'", specifier = ">=0.1.6" }, { name = "mxbai-rerank", marker = "extra == 'mxbai'", specifier = ">=0.1.6" },
{ name = "opencv-python-headless", marker = "extra == 'docling'", specifier = ">=4.12.0.88" }, { name = "opencv-python-headless", marker = "extra == 'docling'", specifier = ">=4.12.0.88" },
{ name = "pathspec", specifier = ">=0.12.1" }, { name = "pathspec", specifier = ">=0.12.1" },
@ -1842,7 +1842,7 @@ wheels = [
[[package]] [[package]]
name = "lancedb" name = "lancedb"
version = "0.26.0" version = "0.26.1"
source = { registry = "https://pypi.org/simple" } source = { registry = "https://pypi.org/simple" }
dependencies = [ dependencies = [
{ name = "deprecation" }, { name = "deprecation" },
@ -1854,12 +1854,12 @@ dependencies = [
{ name = "tqdm" }, { name = "tqdm" },
] ]
wheels = [ wheels = [
{ url = "https://files.pythonhosted.org/packages/a8/91/fe585b2181bd61efc65e1da410ae8ab7b29a26f156e4ca7d7d616b1234de/lancedb-0.26.0-cp39-abi3-macosx_11_0_arm64.whl", hash = "sha256:3a0d435fff1392f056c173f695f71d495c691c555daa9802c056ea23f6a3900e", size = 41174270, upload-time = "2025-12-16T17:16:30.699Z" }, { url = "https://files.pythonhosted.org/packages/45/b5/110651418ceb1fa4ff2eb74ce4bad911ecf49dc765b134f0201d5564aab8/lancedb-0.26.1-cp39-abi3-macosx_11_0_arm64.whl", hash = "sha256:b1c4389134ede49e4be0497b9719f573f447e627426bb9e6fc1b642db11fb22d", size = 43416143, upload-time = "2026-01-02T17:57:07.232Z" },
{ url = "https://files.pythonhosted.org/packages/ce/fc/e47e092f4fc97a8810b37dbee07996689bca42f0817f3f3c38d7fb51dd9d/lancedb-0.26.0-cp39-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a2206320fd0f33c01e264960afd768987646133cf152c4d3a8b7faf81b3017bf", size = 42936720, upload-time = "2025-12-16T17:24:43.527Z" }, { url = "https://files.pythonhosted.org/packages/81/8a/b48a14281d7875e5bfccf22d911d9e1fa019c1fe7b805d290a4449e3cf60/lancedb-0.26.1-cp39-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:07abd18e0aa4730442d0361bab4491ad469de14f9087c3542e56ca6d7fcda473", size = 45302392, upload-time = "2026-01-02T18:04:55.963Z" },
{ url = "https://files.pythonhosted.org/packages/b5/d7/323897d22a7c00ef1dc4f5b76df1a11df549fe887d8e05d689c2224e47b8/lancedb-0.26.0-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:7ca0322cb4b62d526748f6f29e5b43cce4251c7f693e111897eb1f77e7f1ec2b", size = 45846184, upload-time = "2025-12-16T17:27:33.802Z" }, { url = "https://files.pythonhosted.org/packages/4b/d0/8f6bc531f290206c7a0061236928710506598a2591ff1fcaea477fc52e7f/lancedb-0.26.1-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:df8eb631519c6ede9975099bea187ea25a09e4617a421fe19e5e1613651cd62f", size = 48372676, upload-time = "2026-01-02T18:08:12.373Z" },
{ url = "https://files.pythonhosted.org/packages/3a/0b/7671c94b27a5aa267b9f1d6db759c9e08070cb8f783828ade04da9dc7d79/lancedb-0.26.0-cp39-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:7f2b8d69a647265b8753576b501354333c3edfd47d12ec9f47e665e8574c92fe", size = 42954293, upload-time = "2025-12-16T17:24:30.335Z" }, { url = "https://files.pythonhosted.org/packages/f5/13/d8db83335ddf28afe1fb814ca995da7f67826f337d547e54471d7d425dd1/lancedb-0.26.1-cp39-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:2941c9f8aa22244002307c4da5d19f12ab77dcb0569eb4f8a48b60e9c4fdee79", size = 45318771, upload-time = "2026-01-02T18:04:26.429Z" },
{ url = "https://files.pythonhosted.org/packages/52/2e/9f720d6ae7bd3a94d096f320a0ec2f277735423af9d16cf5c61c4a70e6ca/lancedb-0.26.0-cp39-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:8e5cc334686a389cf2f28d1c239d13a205098ed98f3914226d3966858e58b957", size = 45896935, upload-time = "2025-12-16T17:27:30.156Z" }, { url = "https://files.pythonhosted.org/packages/08/94/10e9d4b5ba49eeba72024d310dc42e0c24feb8d5676f48e989198121a8a0/lancedb-0.26.1-cp39-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:0d5f125b98836a49095c492085f5ecf3a78906fafcab59c367d9347eb372a4cc", size = 48425627, upload-time = "2026-01-02T18:11:57.443Z" },
{ url = "https://files.pythonhosted.org/packages/00/0e/4b292c24a9e25ee2cd081d2da930fcdc672ee0eea531fc453c19c73addb5/lancedb-0.26.0-cp39-abi3-win_amd64.whl", hash = "sha256:2fc9b48a11f526de87388002eb3838329db7279241eefb3166c1c6c3b194a3cf", size = 50615000, upload-time = "2025-12-16T17:53:34.409Z" }, { url = "https://files.pythonhosted.org/packages/17/5d/d7a834ce8dd9c5e6ef7a0e308c7de5f87bb8f04c0944a1bea617d9d42dc7/lancedb-0.26.1-cp39-abi3-win_amd64.whl", hash = "sha256:9338d34c6e7472c97e49fd6b2638b29d3d087e8b002d92cafdbb46a8b0b1480e", size = 53214501, upload-time = "2026-01-02T22:34:06.836Z" },
] ]
[[package]] [[package]]