chore: add flag for expansion method
fix: remove gzip config
This commit is contained in:
parent
9bbe43adf4
commit
e26a452da5
3 changed files with 25 additions and 43 deletions
|
|
@ -103,13 +103,6 @@ class HaikuRAG:
|
||||||
"""Whether the client is in read-only mode."""
|
"""Whether the client is in read-only mode."""
|
||||||
return self.store.is_read_only
|
return self.store.is_read_only
|
||||||
|
|
||||||
def _compress_docling(self, json_str: str) -> bytes:
|
|
||||||
"""Compress a docling JSON string, respecting config."""
|
|
||||||
return compress_json(
|
|
||||||
json_str,
|
|
||||||
enabled=self._config.storage.compress_docling,
|
|
||||||
)
|
|
||||||
|
|
||||||
async def __aenter__(self):
|
async def __aenter__(self):
|
||||||
"""Async context manager entry."""
|
"""Async context manager entry."""
|
||||||
return self
|
return self
|
||||||
|
|
@ -502,7 +495,7 @@ class HaikuRAG:
|
||||||
uri=uri,
|
uri=uri,
|
||||||
title=title,
|
title=title,
|
||||||
metadata=metadata or {},
|
metadata=metadata or {},
|
||||||
docling_document=self._compress_docling(docling_document.model_dump_json()),
|
docling_document=compress_json(docling_document.model_dump_json()),
|
||||||
docling_version=docling_document.version,
|
docling_version=docling_document.version,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -542,7 +535,7 @@ class HaikuRAG:
|
||||||
uri=uri,
|
uri=uri,
|
||||||
title=title,
|
title=title,
|
||||||
metadata=metadata or {},
|
metadata=metadata or {},
|
||||||
docling_document=self._compress_docling(docling_document.model_dump_json()),
|
docling_document=compress_json(docling_document.model_dump_json()),
|
||||||
docling_version=docling_document.version,
|
docling_version=docling_document.version,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -679,7 +672,7 @@ class HaikuRAG:
|
||||||
# Update existing document and rechunk
|
# Update existing document and rechunk
|
||||||
existing_doc.content = stored_content
|
existing_doc.content = stored_content
|
||||||
existing_doc.metadata = metadata
|
existing_doc.metadata = metadata
|
||||||
existing_doc.docling_document = self._compress_docling(
|
existing_doc.docling_document = compress_json(
|
||||||
docling_document.model_dump_json()
|
docling_document.model_dump_json()
|
||||||
)
|
)
|
||||||
existing_doc.docling_version = docling_document.version
|
existing_doc.docling_version = docling_document.version
|
||||||
|
|
@ -701,7 +694,7 @@ class HaikuRAG:
|
||||||
uri=uri,
|
uri=uri,
|
||||||
title=title,
|
title=title,
|
||||||
metadata=metadata,
|
metadata=metadata,
|
||||||
docling_document=self._compress_docling(docling_document.model_dump_json()),
|
docling_document=compress_json(docling_document.model_dump_json()),
|
||||||
docling_version=docling_document.version,
|
docling_version=docling_document.version,
|
||||||
)
|
)
|
||||||
return await self._store_document_with_chunks(document, embedded_chunks)
|
return await self._store_document_with_chunks(document, embedded_chunks)
|
||||||
|
|
@ -796,7 +789,7 @@ class HaikuRAG:
|
||||||
# Update existing document and rechunk
|
# Update existing document and rechunk
|
||||||
existing_doc.content = stored_content
|
existing_doc.content = stored_content
|
||||||
existing_doc.metadata = metadata
|
existing_doc.metadata = metadata
|
||||||
existing_doc.docling_document = self._compress_docling(
|
existing_doc.docling_document = compress_json(
|
||||||
docling_document.model_dump_json()
|
docling_document.model_dump_json()
|
||||||
)
|
)
|
||||||
existing_doc.docling_version = docling_document.version
|
existing_doc.docling_version = docling_document.version
|
||||||
|
|
@ -818,7 +811,7 @@ class HaikuRAG:
|
||||||
uri=url,
|
uri=url,
|
||||||
title=title,
|
title=title,
|
||||||
metadata=metadata,
|
metadata=metadata,
|
||||||
docling_document=self._compress_docling(docling_document.model_dump_json()),
|
docling_document=compress_json(docling_document.model_dump_json()),
|
||||||
docling_version=docling_document.version,
|
docling_version=docling_document.version,
|
||||||
)
|
)
|
||||||
return await self._store_document_with_chunks(document, embedded_chunks)
|
return await self._store_document_with_chunks(document, embedded_chunks)
|
||||||
|
|
@ -978,7 +971,7 @@ class HaikuRAG:
|
||||||
# Store docling data if provided
|
# Store docling data if provided
|
||||||
if docling_document is not None:
|
if docling_document is not None:
|
||||||
existing_doc.content = docling_document.export_to_markdown()
|
existing_doc.content = docling_document.export_to_markdown()
|
||||||
existing_doc.docling_document = self._compress_docling(
|
existing_doc.docling_document = compress_json(
|
||||||
docling_document.model_dump_json()
|
docling_document.model_dump_json()
|
||||||
)
|
)
|
||||||
existing_doc.docling_version = docling_document.version
|
existing_doc.docling_version = docling_document.version
|
||||||
|
|
@ -990,7 +983,7 @@ class HaikuRAG:
|
||||||
# DoclingDocument provided without chunks - chunk and embed using primitives
|
# DoclingDocument provided without chunks - chunk and embed using primitives
|
||||||
if docling_document is not None:
|
if docling_document is not None:
|
||||||
existing_doc.content = docling_document.export_to_markdown()
|
existing_doc.content = docling_document.export_to_markdown()
|
||||||
existing_doc.docling_document = self._compress_docling(
|
existing_doc.docling_document = compress_json(
|
||||||
docling_document.model_dump_json()
|
docling_document.model_dump_json()
|
||||||
)
|
)
|
||||||
existing_doc.docling_version = docling_document.version
|
existing_doc.docling_version = docling_document.version
|
||||||
|
|
@ -1005,7 +998,7 @@ class HaikuRAG:
|
||||||
assert content is not None
|
assert content is not None
|
||||||
existing_doc.content = content
|
existing_doc.content = content
|
||||||
converted_docling = await self.convert(existing_doc.content)
|
converted_docling = await self.convert(existing_doc.content)
|
||||||
existing_doc.docling_document = self._compress_docling(
|
existing_doc.docling_document = compress_json(
|
||||||
converted_docling.model_dump_json()
|
converted_docling.model_dump_json()
|
||||||
)
|
)
|
||||||
existing_doc.docling_version = converted_docling.version
|
existing_doc.docling_version = converted_docling.version
|
||||||
|
|
@ -1182,10 +1175,15 @@ class HaikuRAG:
|
||||||
|
|
||||||
expand_start = time.perf_counter()
|
expand_start = time.perf_counter()
|
||||||
|
|
||||||
|
mode = self._config.search.context_expansion_mode
|
||||||
radius = self._config.search.context_radius
|
radius = self._config.search.context_radius
|
||||||
max_items = self._config.search.max_context_items
|
max_items = self._config.search.max_context_items
|
||||||
max_chars = self._config.search.max_context_chars
|
max_chars = self._config.search.max_context_chars
|
||||||
|
|
||||||
|
if mode == "disabled":
|
||||||
|
logger.info("expand.disabled, skipping")
|
||||||
|
return search_results
|
||||||
|
|
||||||
# Group by document_id for efficient processing
|
# Group by document_id for efficient processing
|
||||||
document_groups: dict[str | None, list[SearchResult]] = {}
|
document_groups: dict[str | None, list[SearchResult]] = {}
|
||||||
for result in search_results:
|
for result in search_results:
|
||||||
|
|
@ -1195,9 +1193,10 @@ class HaikuRAG:
|
||||||
document_groups[doc_id].append(result)
|
document_groups[doc_id].append(result)
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
"expand.groups docs=%d results=%d",
|
"expand.groups docs=%d results=%d mode=%s",
|
||||||
len(document_groups),
|
len(document_groups),
|
||||||
len(search_results),
|
len(search_results),
|
||||||
|
mode,
|
||||||
)
|
)
|
||||||
|
|
||||||
expanded_results = []
|
expanded_results = []
|
||||||
|
|
@ -1210,7 +1209,7 @@ class HaikuRAG:
|
||||||
has_refs = any(r.doc_item_refs for r in doc_results)
|
has_refs = any(r.doc_item_refs for r in doc_results)
|
||||||
docling_doc = None
|
docling_doc = None
|
||||||
|
|
||||||
if has_refs:
|
if mode == "auto" and has_refs:
|
||||||
# Only fetch docling data (skip content blob)
|
# Only fetch docling data (skip content blob)
|
||||||
t0 = time.perf_counter()
|
t0 = time.perf_counter()
|
||||||
doc = await self.document_repository.get_docling_data(doc_id)
|
doc = await self.document_repository.get_docling_data(doc_id)
|
||||||
|
|
@ -1973,7 +1972,7 @@ class HaikuRAG:
|
||||||
embedded_chunks = await embed_chunks(chunks, self._config)
|
embedded_chunks = await embed_chunks(chunks, self._config)
|
||||||
|
|
||||||
# Update document fields
|
# Update document fields
|
||||||
doc.docling_document = self._compress_docling(docling_document.model_dump_json())
|
doc.docling_document = compress_json(docling_document.model_dump_json())
|
||||||
doc.docling_version = docling_document.version
|
doc.docling_version = docling_document.version
|
||||||
|
|
||||||
# Prepare chunks with document_id and order
|
# Prepare chunks with document_id and order
|
||||||
|
|
@ -2052,7 +2051,7 @@ class HaikuRAG:
|
||||||
chunks = await self.chunk(docling_document)
|
chunks = await self.chunk(docling_document)
|
||||||
embedded_chunks = await embed_chunks(chunks, self._config)
|
embedded_chunks = await embed_chunks(chunks, self._config)
|
||||||
|
|
||||||
doc.docling_document = self._compress_docling(docling_document.model_dump_json())
|
doc.docling_document = compress_json(docling_document.model_dump_json())
|
||||||
doc.docling_version = docling_document.version
|
doc.docling_version = docling_document.version
|
||||||
|
|
||||||
# Prepare chunks with document_id and order
|
# Prepare chunks with document_id and order
|
||||||
|
|
|
||||||
|
|
@ -48,7 +48,6 @@ class StorageConfig(BaseModel):
|
||||||
auto_vacuum: bool = True
|
auto_vacuum: bool = True
|
||||||
vacuum_retention_seconds: int = 86400
|
vacuum_retention_seconds: int = 86400
|
||||||
docling_cache_size: int = 100
|
docling_cache_size: int = 100
|
||||||
compress_docling: bool = True
|
|
||||||
|
|
||||||
|
|
||||||
class MonitorConfig(BaseModel):
|
class MonitorConfig(BaseModel):
|
||||||
|
|
@ -176,6 +175,7 @@ class ProcessingConfig(BaseModel):
|
||||||
class SearchConfig(BaseModel):
|
class SearchConfig(BaseModel):
|
||||||
limit: int = 10
|
limit: int = 10
|
||||||
context_radius: int = 0
|
context_radius: int = 0
|
||||||
|
context_expansion_mode: Literal["auto", "chunks", "disabled"] = "auto"
|
||||||
max_context_items: int = 10
|
max_context_items: int = 10
|
||||||
max_context_chars: int = 10000
|
max_context_chars: int = 10000
|
||||||
vector_index_metric: Literal["cosine", "l2", "dot"] = "cosine"
|
vector_index_metric: Literal["cosine", "l2", "dot"] = "cosine"
|
||||||
|
|
|
||||||
|
|
@ -1,28 +1,11 @@
|
||||||
import gzip
|
import gzip
|
||||||
|
|
||||||
# Gzip magic number: first two bytes of any gzip stream
|
|
||||||
_GZIP_MAGIC = b"\x1f\x8b"
|
|
||||||
|
|
||||||
|
def compress_json(json_str: str) -> bytes:
|
||||||
def compress_json(json_str: str, *, enabled: bool = True) -> bytes:
|
"""Compress a JSON string with gzip."""
|
||||||
"""Compress a JSON string, optionally with gzip.
|
return gzip.compress(json_str.encode("utf-8"))
|
||||||
|
|
||||||
Args:
|
|
||||||
json_str: The JSON string to compress.
|
|
||||||
enabled: If False, returns raw UTF-8 bytes without compression.
|
|
||||||
"""
|
|
||||||
data = json_str.encode("utf-8")
|
|
||||||
if enabled:
|
|
||||||
return gzip.compress(data)
|
|
||||||
return data
|
|
||||||
|
|
||||||
|
|
||||||
def decompress_json(data: bytes) -> str:
|
def decompress_json(data: bytes) -> str:
|
||||||
"""Decompress data to a JSON string.
|
"""Decompress gzip-compressed data to a JSON string."""
|
||||||
|
return gzip.decompress(data).decode("utf-8")
|
||||||
Automatically detects gzip-compressed data via magic bytes,
|
|
||||||
so it handles both compressed and uncompressed storage.
|
|
||||||
"""
|
|
||||||
if data[:2] == _GZIP_MAGIC:
|
|
||||||
return gzip.decompress(data).decode("utf-8")
|
|
||||||
return data.decode("utf-8")
|
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue