Throttle background auto-vacuum to at most once per 5 minutes
This commit is contained in:
parent
5d5d87d44c
commit
f2a5ac4246
2 changed files with 48 additions and 2 deletions
|
|
@ -323,7 +323,12 @@ async def _refresh_doc_metadata(
|
||||||
updated = True
|
updated = True
|
||||||
|
|
||||||
if updated:
|
if updated:
|
||||||
return await client.document_repository.update_meta(doc)
|
result = await client.document_repository.update_meta(doc)
|
||||||
|
# Reclaim the document_meta churn from rolling source_revision sweeps.
|
||||||
|
# The vacuum is debounced, and document_meta is tiny, so this is cheap.
|
||||||
|
if client._config.storage.auto_vacuum:
|
||||||
|
client._schedule_vacuum()
|
||||||
|
return result
|
||||||
return doc
|
return doc
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -730,7 +735,10 @@ async def update_document(
|
||||||
existing_doc.metadata = metadata
|
existing_doc.metadata = metadata
|
||||||
|
|
||||||
if content is None and chunks is None and docling_document is None:
|
if content is None and chunks is None and docling_document is None:
|
||||||
return await client.document_repository.update_meta(existing_doc)
|
updated = await client.document_repository.update_meta(existing_doc)
|
||||||
|
if client._config.storage.auto_vacuum:
|
||||||
|
client._schedule_vacuum()
|
||||||
|
return updated
|
||||||
|
|
||||||
if chunks is not None:
|
if chunks is not None:
|
||||||
if docling_document is not None:
|
if docling_document is not None:
|
||||||
|
|
|
||||||
|
|
@ -4,6 +4,18 @@ import pytest
|
||||||
|
|
||||||
import haiku.rag.client as client_mod
|
import haiku.rag.client as client_mod
|
||||||
from haiku.rag.client import HaikuRAG
|
from haiku.rag.client import HaikuRAG
|
||||||
|
from haiku.rag.client.documents import _refresh_doc_metadata
|
||||||
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.store.models.chunk import Chunk
|
||||||
|
|
||||||
|
|
||||||
|
def _docling_doc(name: str, text: str):
|
||||||
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
from docling_core.types.doc.labels import DocItemLabel
|
||||||
|
|
||||||
|
doc = DoclingDocument(name=name)
|
||||||
|
doc.add_text(label=DocItemLabel.TEXT, text=text)
|
||||||
|
return doc
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
|
|
@ -54,3 +66,29 @@ async def test_debounced_writes_still_collapse_on_close(temp_db_path, monkeypatc
|
||||||
# one scheduled background pass + one final collapse on drain
|
# one scheduled background pass + one final collapse on drain
|
||||||
assert len(calls) == 2
|
assert len(calls) == 2
|
||||||
assert client._vacuum_dirty is False
|
assert client._vacuum_dirty is False
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_metadata_refresh_sweep_schedules_vacuum(temp_db_path):
|
||||||
|
"""A source re-sweep that only rolls source_revision (MD5/revision
|
||||||
|
short-circuit) writes document_meta and must still schedule the (debounced)
|
||||||
|
vacuum, so that tiny churn gets reclaimed instead of accumulating."""
|
||||||
|
dim = Config.embeddings.model.vector_dim
|
||||||
|
async with HaikuRAG(temp_db_path, create=True) as client:
|
||||||
|
doc = await client.import_document(
|
||||||
|
_docling_doc("d", "body"),
|
||||||
|
[Chunk(content="body", embedding=[0.1] * dim, order=0)],
|
||||||
|
uri="mem://sweep",
|
||||||
|
metadata={"source_revision": "r1"},
|
||||||
|
)
|
||||||
|
# Isolate the refresh: the import already scheduled a vacuum.
|
||||||
|
client._vacuum_dirty = False
|
||||||
|
|
||||||
|
await _refresh_doc_metadata(
|
||||||
|
client,
|
||||||
|
doc,
|
||||||
|
title=None,
|
||||||
|
user_metadata={},
|
||||||
|
source_metadata={"source_revision": "r2", "md5": "same"},
|
||||||
|
)
|
||||||
|
assert client._vacuum_dirty is True
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue