Remove pre-processor, no longer needed
This commit is contained in:
parent
69b1afa534
commit
17c7147a49
7 changed files with 7 additions and 189 deletions
|
|
@ -70,6 +70,12 @@
|
||||||
- Cleaner separation: LLM outputs chunk IDs, citations resolved programmatically
|
- Cleaner separation: LLM outputs chunk IDs, citations resolved programmatically
|
||||||
- `Citation` fields are now required (no defaults) for type safety
|
- `Citation` fields are now required (no defaults) for type safety
|
||||||
|
|
||||||
|
### Removed
|
||||||
|
|
||||||
|
- **BREAKING: `markdown_preprocessor` Config Option**: Removed the `processing.markdown_preprocessor` configuration option
|
||||||
|
- Use `convert()`, `chunk()`, and `embed_chunks()` primitives for custom processing pipelines
|
||||||
|
- Transform content at any stage before calling `import_document()`
|
||||||
|
|
||||||
### Migration
|
### Migration
|
||||||
|
|
||||||
This release requires a database rebuild to populate the new DoclingDocument fields:
|
This release requires a database rebuild to populate the new DoclingDocument fields:
|
||||||
|
|
|
||||||
|
|
@ -119,7 +119,6 @@ processing:
|
||||||
chunking_tokenizer: "Qwen/Qwen3-Embedding-0.6B"
|
chunking_tokenizer: "Qwen/Qwen3-Embedding-0.6B"
|
||||||
chunking_merge_peers: true
|
chunking_merge_peers: true
|
||||||
chunking_use_markdown_tables: false
|
chunking_use_markdown_tables: false
|
||||||
markdown_preprocessor: ""
|
|
||||||
conversion_options:
|
conversion_options:
|
||||||
do_ocr: true
|
do_ocr: true
|
||||||
force_ocr: false
|
force_ocr: false
|
||||||
|
|
|
||||||
|
|
@ -11,7 +11,6 @@ processing:
|
||||||
# Chunking configuration
|
# Chunking configuration
|
||||||
chunk_size: 256 # Maximum tokens per chunk
|
chunk_size: 256 # Maximum tokens per chunk
|
||||||
context_chunk_radius: 0 # Context radius for chunk expansion
|
context_chunk_radius: 0 # Context radius for chunk expansion
|
||||||
markdown_preprocessor: "" # Optional preprocessor script
|
|
||||||
|
|
||||||
# Converter selection
|
# Converter selection
|
||||||
converter: docling-local # docling-local or docling-serve
|
converter: docling-local # docling-local or docling-serve
|
||||||
|
|
@ -148,39 +147,6 @@ processing:
|
||||||
context_chunk_radius: 0
|
context_chunk_radius: 0
|
||||||
```
|
```
|
||||||
|
|
||||||
### Markdown Preprocessor
|
|
||||||
|
|
||||||
Optionally preprocess Markdown before chunking by pointing to a callable that receives and returns Markdown text. This is useful for normalizing content, stripping boilerplate, or applying custom transformations before chunk boundaries are computed.
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
processing:
|
|
||||||
# A callable path in one of these formats:
|
|
||||||
# - package.module:func
|
|
||||||
# - package.module.func
|
|
||||||
# - /abs/or/relative/path/to/file.py:func
|
|
||||||
markdown_preprocessor: my_pkg.preprocess:clean_md
|
|
||||||
```
|
|
||||||
|
|
||||||
!!! note
|
|
||||||
- The function signature should be `def clean_md(text: str) -> str` or `async def clean_md(text: str) -> str`.
|
|
||||||
- If the function raises or returns a non-string, haiku.rag logs a warning and proceeds without preprocessing.
|
|
||||||
- The preprocessor affects only the chunking pipeline. The stored document content remains unchanged.
|
|
||||||
|
|
||||||
Example implementation:
|
|
||||||
|
|
||||||
```python
|
|
||||||
# my_pkg/preprocess.py
|
|
||||||
def clean_md(text: str) -> str:
|
|
||||||
# strip HTML comments and collapse multiple blank lines
|
|
||||||
lines = [line for line in text.splitlines() if not line.strip().startswith("<!--")]
|
|
||||||
out = []
|
|
||||||
for line in lines:
|
|
||||||
if line.strip() == "" and (out and out[-1] == ""):
|
|
||||||
continue
|
|
||||||
out.append(line)
|
|
||||||
return "\n".join(out)
|
|
||||||
```
|
|
||||||
|
|
||||||
## File Monitoring
|
## File Monitoring
|
||||||
|
|
||||||
Set directories to monitor for automatic indexing:
|
Set directories to monitor for automatic indexing:
|
||||||
|
|
|
||||||
|
|
@ -111,7 +111,6 @@ class ConversionOptions(BaseModel):
|
||||||
class ProcessingConfig(BaseModel):
|
class ProcessingConfig(BaseModel):
|
||||||
chunk_size: int = 256
|
chunk_size: int = 256
|
||||||
context_chunk_radius: int = 0
|
context_chunk_radius: int = 0
|
||||||
markdown_preprocessor: str = ""
|
|
||||||
converter: str = "docling-local"
|
converter: str = "docling-local"
|
||||||
chunker: str = "docling-local"
|
chunker: str = "docling-local"
|
||||||
chunker_type: str = "hybrid"
|
chunker_type: str = "hybrid"
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,3 @@
|
||||||
import inspect
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
from typing import TYPE_CHECKING, cast
|
from typing import TYPE_CHECKING, cast
|
||||||
|
|
@ -16,7 +15,6 @@ from lancedb.rerankers import RRFReranker
|
||||||
|
|
||||||
from haiku.rag.store.engine import DocumentRecord, Store
|
from haiku.rag.store.engine import DocumentRecord, Store
|
||||||
from haiku.rag.store.models.chunk import Chunk
|
from haiku.rag.store.models.chunk import Chunk
|
||||||
from haiku.rag.utils import load_callable
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
|
@ -196,36 +194,10 @@ class ChunkRepository:
|
||||||
self, document_id: str, document: "DoclingDocument"
|
self, document_id: str, document: "DoclingDocument"
|
||||||
) -> list[Chunk]:
|
) -> list[Chunk]:
|
||||||
"""Create chunks and embeddings for a document from DoclingDocument."""
|
"""Create chunks and embeddings for a document from DoclingDocument."""
|
||||||
# Lazy imports to avoid loading docling during module import
|
|
||||||
from haiku.rag.chunkers import get_chunker
|
from haiku.rag.chunkers import get_chunker
|
||||||
from haiku.rag.converters import get_converter
|
|
||||||
|
|
||||||
chunker = get_chunker(self.store._config)
|
chunker = get_chunker(self.store._config)
|
||||||
|
chunks = await chunker.chunk(document)
|
||||||
# Optionally preprocess markdown before chunking
|
|
||||||
processed_document = document
|
|
||||||
preprocessor_path = self.store._config.processing.markdown_preprocessor
|
|
||||||
if preprocessor_path:
|
|
||||||
try:
|
|
||||||
pre_fn = load_callable(preprocessor_path)
|
|
||||||
markdown = document.export_to_markdown()
|
|
||||||
result = pre_fn(markdown)
|
|
||||||
if inspect.isawaitable(result):
|
|
||||||
result = await result # type: ignore[assignment]
|
|
||||||
processed_markdown = result
|
|
||||||
if not isinstance(processed_markdown, str):
|
|
||||||
raise ValueError("Preprocessor must return a markdown string")
|
|
||||||
converter = get_converter(self.store._config)
|
|
||||||
processed_document = await converter.convert_text(
|
|
||||||
processed_markdown, name="content.md"
|
|
||||||
)
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(
|
|
||||||
f"Failed to apply MARKDOWN_PREPROCESSOR '{preprocessor_path}': {e}. Proceeding without preprocessing."
|
|
||||||
)
|
|
||||||
raise e
|
|
||||||
|
|
||||||
chunks = await chunker.chunk(processed_document)
|
|
||||||
|
|
||||||
# Build embedding texts with headings prepended for better semantic search
|
# Build embedding texts with headings prepended for better semantic search
|
||||||
# The stored content stays raw, but embeddings capture section context
|
# The stored content stays raw, but embeddings capture section context
|
||||||
|
|
|
||||||
|
|
@ -1,9 +1,6 @@
|
||||||
import importlib
|
|
||||||
import importlib.util
|
|
||||||
import sys
|
import sys
|
||||||
from importlib import metadata
|
from importlib import metadata
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from types import ModuleType
|
|
||||||
from typing import TYPE_CHECKING, Any
|
from typing import TYPE_CHECKING, Any
|
||||||
|
|
||||||
from packaging.version import Version, parse
|
from packaging.version import Version, parse
|
||||||
|
|
@ -350,52 +347,3 @@ async def is_up_to_date() -> tuple[bool, Version, Version]:
|
||||||
return running_version >= pypi_version, running_version, pypi_version
|
return running_version >= pypi_version, running_version, pypi_version
|
||||||
|
|
||||||
|
|
||||||
def load_callable(path: str):
|
|
||||||
"""Load a callable from a dotted path or file path.
|
|
||||||
|
|
||||||
Supported formats:
|
|
||||||
- "package.module:func" or "package.module.func"
|
|
||||||
- "path/to/file.py:func"
|
|
||||||
|
|
||||||
Returns the loaded callable. Raises ValueError on failure.
|
|
||||||
"""
|
|
||||||
if not path:
|
|
||||||
raise ValueError("Empty callable path provided")
|
|
||||||
|
|
||||||
module_part = None
|
|
||||||
func_name = None
|
|
||||||
|
|
||||||
if ":" in path:
|
|
||||||
module_part, func_name = path.split(":", 1)
|
|
||||||
else:
|
|
||||||
# split by last dot for module.attr
|
|
||||||
if "." in path:
|
|
||||||
module_part, func_name = path.rsplit(".", 1)
|
|
||||||
else:
|
|
||||||
raise ValueError(
|
|
||||||
"Invalid callable path format. Use 'module:func' or 'module.func' or 'file.py:func'."
|
|
||||||
)
|
|
||||||
|
|
||||||
# Try file path first
|
|
||||||
mod: ModuleType | None = None
|
|
||||||
module_path = Path(module_part)
|
|
||||||
if module_path.suffix == ".py" and module_path.exists():
|
|
||||||
spec = importlib.util.spec_from_file_location(module_path.stem, module_path)
|
|
||||||
if spec and spec.loader:
|
|
||||||
mod = importlib.util.module_from_spec(spec)
|
|
||||||
spec.loader.exec_module(mod)
|
|
||||||
else:
|
|
||||||
# Import as a module path
|
|
||||||
try:
|
|
||||||
mod = importlib.import_module(module_part)
|
|
||||||
except Exception as e:
|
|
||||||
raise ValueError(f"Failed to import module '{module_part}': {e}")
|
|
||||||
|
|
||||||
if not hasattr(mod, func_name):
|
|
||||||
raise ValueError(f"Callable '{func_name}' not found in module '{module_part}'")
|
|
||||||
func = getattr(mod, func_name)
|
|
||||||
if not callable(func):
|
|
||||||
raise ValueError(
|
|
||||||
f"Attribute '{func_name}' in module '{module_part}' is not callable"
|
|
||||||
)
|
|
||||||
return func
|
|
||||||
|
|
|
||||||
|
|
@ -1,72 +0,0 @@
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
|
|
||||||
from haiku.rag.config import Config
|
|
||||||
from haiku.rag.converters import get_converter
|
|
||||||
from haiku.rag.store.engine import Store
|
|
||||||
from haiku.rag.store.models.document import Document
|
|
||||||
from haiku.rag.store.repositories.chunk import ChunkRepository
|
|
||||||
from haiku.rag.store.repositories.document import DocumentRepository
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
"is_async, marker",
|
|
||||||
[
|
|
||||||
(False, "MARKER_LINE"),
|
|
||||||
(True, "ASYNC_MARKER"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_markdown_preprocessor_applied_parametrized(
|
|
||||||
is_async: bool, marker: str, tmp_path: Path, temp_db_path: Path
|
|
||||||
):
|
|
||||||
"""Ensure MARKDOWN_PREPROCESSOR (sync or async) transforms markdown before chunking."""
|
|
||||||
pre_file = tmp_path / ("pre_async.py" if is_async else "pre.py")
|
|
||||||
if is_async:
|
|
||||||
pre_file.write_text(
|
|
||||||
"""
|
|
||||||
import asyncio
|
|
||||||
|
|
||||||
async def add_marker(text: str) -> str:
|
|
||||||
await asyncio.sleep(0)
|
|
||||||
return text + "\\n\\nASYNC_MARKER\\n"
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
pre_file.write_text(
|
|
||||||
"""
|
|
||||||
def add_marker(text: str) -> str:
|
|
||||||
return text + "\\n\\nMARKER_LINE\\n"
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
|
|
||||||
original_pre = Config.processing.markdown_preprocessor
|
|
||||||
try:
|
|
||||||
Config.processing.markdown_preprocessor = f"{pre_file}:add_marker"
|
|
||||||
|
|
||||||
store = Store(temp_db_path, create=True)
|
|
||||||
chunk_repo = ChunkRepository(store)
|
|
||||||
doc_repo = DocumentRepository(store)
|
|
||||||
|
|
||||||
document = Document(content="Hello world")
|
|
||||||
created_doc = await doc_repo.create(document)
|
|
||||||
assert created_doc.id is not None
|
|
||||||
|
|
||||||
# Stub embeddings to avoid network
|
|
||||||
dim = chunk_repo.embedder._vector_dim
|
|
||||||
|
|
||||||
async def fake_embed(x): # type: ignore[override]
|
|
||||||
if isinstance(x, list):
|
|
||||||
return [[0.0] * dim for _ in x]
|
|
||||||
return [0.0] * dim
|
|
||||||
|
|
||||||
chunk_repo.embedder.embed = fake_embed # type: ignore[assignment]
|
|
||||||
|
|
||||||
converter = get_converter(Config)
|
|
||||||
docling = await converter.convert_text(document.content, name="test.md")
|
|
||||||
chunks = await chunk_repo.create_chunks_for_document(created_doc.id, docling)
|
|
||||||
|
|
||||||
assert any(marker in c.content for c in chunks)
|
|
||||||
finally:
|
|
||||||
Config.processing.markdown_preprocessor = original_pre
|
|
||||||
Loading…
Reference in a new issue