Introduce converters for supporting more than local docling document conversion. Transform existing FileReader and utils to "docling-local" converter
This commit is contained in:
parent
636d5261e6
commit
7ba1c2376e
17 changed files with 390 additions and 245 deletions
|
|
@ -9,6 +9,7 @@ from urllib.parse import urlparse
|
||||||
import httpx
|
import httpx
|
||||||
|
|
||||||
from haiku.rag.config import AppConfig, Config
|
from haiku.rag.config import AppConfig, Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
from haiku.rag.reranking import get_reranker
|
from haiku.rag.reranking import get_reranker
|
||||||
from haiku.rag.store.engine import Store
|
from haiku.rag.store.engine import Store
|
||||||
from haiku.rag.store.models.chunk import Chunk
|
from haiku.rag.store.models.chunk import Chunk
|
||||||
|
|
@ -111,10 +112,9 @@ class HaikuRAG:
|
||||||
|
|
||||||
# Only create docling_document if we need to generate chunks
|
# Only create docling_document if we need to generate chunks
|
||||||
if chunks is None:
|
if chunks is None:
|
||||||
# Lazy import to avoid loading docling
|
# Use converter to convert text
|
||||||
from haiku.rag.utils import text_to_docling_document
|
converter = get_converter(self._config)
|
||||||
|
docling_document = converter.convert_text(content)
|
||||||
docling_document = text_to_docling_document(content)
|
|
||||||
else:
|
else:
|
||||||
# Chunks already provided, no conversion needed
|
# Chunks already provided, no conversion needed
|
||||||
docling_document = None
|
docling_document = None
|
||||||
|
|
@ -201,12 +201,10 @@ class HaikuRAG:
|
||||||
Raises:
|
Raises:
|
||||||
ValueError: If the file cannot be parsed or doesn't exist
|
ValueError: If the file cannot be parsed or doesn't exist
|
||||||
"""
|
"""
|
||||||
# Lazy import to avoid loading docling
|
|
||||||
from haiku.rag.reader import FileReader
|
|
||||||
|
|
||||||
metadata = metadata or {}
|
metadata = metadata or {}
|
||||||
|
|
||||||
if source_path.suffix.lower() not in FileReader.extensions:
|
converter = get_converter(self._config)
|
||||||
|
if source_path.suffix.lower() not in converter.supported_extensions:
|
||||||
raise ValueError(f"Unsupported file extension: {source_path.suffix}")
|
raise ValueError(f"Unsupported file extension: {source_path.suffix}")
|
||||||
|
|
||||||
if not source_path.exists():
|
if not source_path.exists():
|
||||||
|
|
@ -242,7 +240,8 @@ class HaikuRAG:
|
||||||
return existing_doc
|
return existing_doc
|
||||||
|
|
||||||
# Parse file only when content changed or new document
|
# Parse file only when content changed or new document
|
||||||
docling_document = FileReader.parse_file(source_path)
|
converter = get_converter(self._config)
|
||||||
|
docling_document = converter.convert_file(source_path)
|
||||||
|
|
||||||
if existing_doc:
|
if existing_doc:
|
||||||
# Update existing document
|
# Update existing document
|
||||||
|
|
@ -283,11 +282,11 @@ class HaikuRAG:
|
||||||
ValueError: If the content cannot be parsed
|
ValueError: If the content cannot be parsed
|
||||||
httpx.RequestError: If URL request fails
|
httpx.RequestError: If URL request fails
|
||||||
"""
|
"""
|
||||||
# Lazy import to avoid loading docling
|
|
||||||
from haiku.rag.reader import FileReader
|
|
||||||
|
|
||||||
metadata = metadata or {}
|
metadata = metadata or {}
|
||||||
|
|
||||||
|
converter = get_converter(self._config)
|
||||||
|
supported_extensions = converter.supported_extensions
|
||||||
|
|
||||||
async with httpx.AsyncClient() as client:
|
async with httpx.AsyncClient() as client:
|
||||||
response = await client.get(url)
|
response = await client.get(url)
|
||||||
response.raise_for_status()
|
response.raise_for_status()
|
||||||
|
|
@ -320,7 +319,7 @@ class HaikuRAG:
|
||||||
url, content_type
|
url, content_type
|
||||||
)
|
)
|
||||||
|
|
||||||
if file_extension not in FileReader.extensions:
|
if file_extension not in supported_extensions:
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
f"Unsupported content type/extension: {content_type}/{file_extension}"
|
f"Unsupported content type/extension: {content_type}/{file_extension}"
|
||||||
)
|
)
|
||||||
|
|
@ -333,8 +332,8 @@ class HaikuRAG:
|
||||||
temp_file.flush() # Ensure content is written to disk
|
temp_file.flush() # Ensure content is written to disk
|
||||||
temp_path = Path(temp_file.name)
|
temp_path = Path(temp_file.name)
|
||||||
|
|
||||||
# Parse the content using FileReader
|
# Parse the content using converter
|
||||||
docling_document = FileReader.parse_file(temp_path)
|
docling_document = converter.convert_file(temp_path)
|
||||||
|
|
||||||
# Merge metadata with contentType and md5
|
# Merge metadata with contentType and md5
|
||||||
metadata.update({"contentType": content_type, "md5": md5_hash})
|
metadata.update({"contentType": content_type, "md5": md5_hash})
|
||||||
|
|
@ -410,11 +409,9 @@ class HaikuRAG:
|
||||||
|
|
||||||
async def update_document(self, document: Document) -> Document:
|
async def update_document(self, document: Document) -> Document:
|
||||||
"""Update an existing document."""
|
"""Update an existing document."""
|
||||||
# Lazy import to avoid loading docling
|
|
||||||
from haiku.rag.utils import text_to_docling_document
|
|
||||||
|
|
||||||
# Convert content to DoclingDocument
|
# Convert content to DoclingDocument
|
||||||
docling_document = text_to_docling_document(document.content)
|
converter = get_converter(self._config)
|
||||||
|
docling_document = converter.convert_text(document.content)
|
||||||
|
|
||||||
return await self.document_repository._update_and_rechunk(
|
return await self.document_repository._update_and_rechunk(
|
||||||
document, docling_document
|
document, docling_document
|
||||||
|
|
@ -646,12 +643,11 @@ class HaikuRAG:
|
||||||
Yields:
|
Yields:
|
||||||
int: The ID of the document currently being processed
|
int: The ID of the document currently being processed
|
||||||
"""
|
"""
|
||||||
# Lazy import to avoid loading docling
|
|
||||||
from haiku.rag.utils import text_to_docling_document
|
|
||||||
|
|
||||||
await self.chunk_repository.delete_all()
|
await self.chunk_repository.delete_all()
|
||||||
self.store.recreate_embeddings_table()
|
self.store.recreate_embeddings_table()
|
||||||
|
|
||||||
|
converter = get_converter(self._config)
|
||||||
|
|
||||||
# Update settings to current config
|
# Update settings to current config
|
||||||
settings_repo = SettingsRepository(self.store)
|
settings_repo = SettingsRepository(self.store)
|
||||||
settings_repo.save_current_settings()
|
settings_repo.save_current_settings()
|
||||||
|
|
@ -703,14 +699,14 @@ class HaikuRAG:
|
||||||
logger.warning(
|
logger.warning(
|
||||||
"Source missing for %s, re-embedding from content", doc.uri
|
"Source missing for %s, re-embedding from content", doc.uri
|
||||||
)
|
)
|
||||||
docling_document = text_to_docling_document(doc.content)
|
docling_document = converter.convert_text(doc.content)
|
||||||
await self.chunk_repository.create_chunks_for_document(
|
await self.chunk_repository.create_chunks_for_document(
|
||||||
doc.id, docling_document
|
doc.id, docling_document
|
||||||
)
|
)
|
||||||
yield doc.id
|
yield doc.id
|
||||||
else:
|
else:
|
||||||
# Document without URI - re-create chunks from existing content
|
# Document without URI - re-create chunks from existing content
|
||||||
docling_document = text_to_docling_document(doc.content)
|
docling_document = converter.convert_text(doc.content)
|
||||||
await self.chunk_repository.create_chunks_for_document(
|
await self.chunk_repository.create_chunks_for_document(
|
||||||
doc.id, docling_document
|
doc.id, docling_document
|
||||||
)
|
)
|
||||||
|
|
|
||||||
|
|
@ -54,6 +54,7 @@ class ProcessingConfig(BaseModel):
|
||||||
chunk_size: int = 256
|
chunk_size: int = 256
|
||||||
context_chunk_radius: int = 0
|
context_chunk_radius: int = 0
|
||||||
markdown_preprocessor: str = ""
|
markdown_preprocessor: str = ""
|
||||||
|
converter: str = "docling-local"
|
||||||
|
|
||||||
|
|
||||||
class OllamaConfig(BaseModel):
|
class OllamaConfig(BaseModel):
|
||||||
|
|
@ -71,9 +72,16 @@ class VLLMConfig(BaseModel):
|
||||||
research_base_url: str = ""
|
research_base_url: str = ""
|
||||||
|
|
||||||
|
|
||||||
|
class DoclingServeConfig(BaseModel):
|
||||||
|
base_url: str = "http://localhost:5001"
|
||||||
|
api_key: str = ""
|
||||||
|
timeout: int = 300
|
||||||
|
|
||||||
|
|
||||||
class ProvidersConfig(BaseModel):
|
class ProvidersConfig(BaseModel):
|
||||||
ollama: OllamaConfig = Field(default_factory=OllamaConfig)
|
ollama: OllamaConfig = Field(default_factory=OllamaConfig)
|
||||||
vllm: VLLMConfig = Field(default_factory=VLLMConfig)
|
vllm: VLLMConfig = Field(default_factory=VLLMConfig)
|
||||||
|
docling_serve: DoclingServeConfig = Field(default_factory=DoclingServeConfig)
|
||||||
|
|
||||||
|
|
||||||
class AGUIConfig(BaseModel):
|
class AGUIConfig(BaseModel):
|
||||||
|
|
|
||||||
|
|
@ -23,11 +23,18 @@ class FileFilter(DefaultFilter):
|
||||||
*,
|
*,
|
||||||
ignore_patterns: list[str] | None = None,
|
ignore_patterns: list[str] | None = None,
|
||||||
include_patterns: list[str] | None = None,
|
include_patterns: list[str] | None = None,
|
||||||
|
supported_extensions: list[str] | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
# Lazy import to avoid loading docling
|
if supported_extensions is None:
|
||||||
from haiku.rag.reader import FileReader
|
# Default to docling-local extensions if not provided
|
||||||
|
from haiku.rag.converters.docling_local import DoclingLocalConverter
|
||||||
|
|
||||||
self.extensions = tuple(FileReader.extensions)
|
supported_extensions = (
|
||||||
|
DoclingLocalConverter.docling_extensions
|
||||||
|
+ DoclingLocalConverter.text_extensions
|
||||||
|
)
|
||||||
|
|
||||||
|
self.extensions = tuple(supported_extensions)
|
||||||
self.ignore_spec = (
|
self.ignore_spec = (
|
||||||
pathspec.PathSpec.from_lines(GitWildMatchPattern, ignore_patterns)
|
pathspec.PathSpec.from_lines(GitWildMatchPattern, ignore_patterns)
|
||||||
if ignore_patterns
|
if ignore_patterns
|
||||||
|
|
@ -72,16 +79,21 @@ class FileWatcher:
|
||||||
client: HaikuRAG,
|
client: HaikuRAG,
|
||||||
config: AppConfig = Config,
|
config: AppConfig = Config,
|
||||||
):
|
):
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
|
|
||||||
self.paths = config.monitor.directories
|
self.paths = config.monitor.directories
|
||||||
self.client = client
|
self.client = client
|
||||||
self.ignore_patterns = config.monitor.ignore_patterns or None
|
self.ignore_patterns = config.monitor.ignore_patterns or None
|
||||||
self.include_patterns = config.monitor.include_patterns or None
|
self.include_patterns = config.monitor.include_patterns or None
|
||||||
self.delete_orphans = config.monitor.delete_orphans
|
self.delete_orphans = config.monitor.delete_orphans
|
||||||
|
self.supported_extensions = get_converter(config).supported_extensions
|
||||||
|
|
||||||
async def observe(self):
|
async def observe(self):
|
||||||
logger.info(f"Watching files in {self.paths}")
|
logger.info(f"Watching files in {self.paths}")
|
||||||
filter = FileFilter(
|
filter = FileFilter(
|
||||||
ignore_patterns=self.ignore_patterns, include_patterns=self.include_patterns
|
ignore_patterns=self.ignore_patterns,
|
||||||
|
include_patterns=self.include_patterns,
|
||||||
|
supported_extensions=self.supported_extensions,
|
||||||
)
|
)
|
||||||
await self.refresh()
|
await self.refresh()
|
||||||
|
|
||||||
|
|
@ -96,9 +108,6 @@ class FileWatcher:
|
||||||
await self._delete_document(Path(path))
|
await self._delete_document(Path(path))
|
||||||
|
|
||||||
async def refresh(self):
|
async def refresh(self):
|
||||||
# Lazy import to avoid loading docling
|
|
||||||
from haiku.rag.reader import FileReader
|
|
||||||
|
|
||||||
# Delete orphaned documents in background if enabled
|
# Delete orphaned documents in background if enabled
|
||||||
if self.delete_orphans:
|
if self.delete_orphans:
|
||||||
logger.info("Starting orphan cleanup in background")
|
logger.info("Starting orphan cleanup in background")
|
||||||
|
|
@ -106,12 +115,14 @@ class FileWatcher:
|
||||||
|
|
||||||
# Create filter to apply same logic as observe()
|
# Create filter to apply same logic as observe()
|
||||||
filter = FileFilter(
|
filter = FileFilter(
|
||||||
ignore_patterns=self.ignore_patterns, include_patterns=self.include_patterns
|
ignore_patterns=self.ignore_patterns,
|
||||||
|
include_patterns=self.include_patterns,
|
||||||
|
supported_extensions=self.supported_extensions,
|
||||||
)
|
)
|
||||||
|
|
||||||
for path in self.paths:
|
for path in self.paths:
|
||||||
for f in Path(path).rglob("**/*"):
|
for f in Path(path).rglob("**/*"):
|
||||||
if f.is_file() and f.suffix in FileReader.extensions:
|
if f.is_file() and f.suffix in self.supported_extensions:
|
||||||
# Apply pattern filters
|
# Apply pattern filters
|
||||||
if filter(Change.added, str(f)):
|
if filter(Change.added, str(f)):
|
||||||
await self._upsert_document(f)
|
await self._upsert_document(f)
|
||||||
|
|
|
||||||
|
|
@ -1,135 +0,0 @@
|
||||||
from pathlib import Path
|
|
||||||
from typing import ClassVar
|
|
||||||
|
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
|
||||||
|
|
||||||
from haiku.rag.utils import text_to_docling_document
|
|
||||||
|
|
||||||
# Check if docling is available
|
|
||||||
try:
|
|
||||||
import docling # noqa: F401
|
|
||||||
|
|
||||||
DOCLING_AVAILABLE = True
|
|
||||||
except ImportError:
|
|
||||||
DOCLING_AVAILABLE = False
|
|
||||||
|
|
||||||
|
|
||||||
class FileReader:
|
|
||||||
# Extensions supported by docling
|
|
||||||
docling_extensions: ClassVar[list[str]] = [
|
|
||||||
".adoc",
|
|
||||||
".asc",
|
|
||||||
".asciidoc",
|
|
||||||
".bmp",
|
|
||||||
".csv",
|
|
||||||
".docx",
|
|
||||||
".html",
|
|
||||||
".xhtml",
|
|
||||||
".jpeg",
|
|
||||||
".jpg",
|
|
||||||
".md",
|
|
||||||
".pdf",
|
|
||||||
".png",
|
|
||||||
".pptx",
|
|
||||||
".tiff",
|
|
||||||
".xlsx",
|
|
||||||
".xml",
|
|
||||||
".webp",
|
|
||||||
]
|
|
||||||
|
|
||||||
# Plain text extensions that we'll read directly
|
|
||||||
text_extensions: ClassVar[list[str]] = [
|
|
||||||
".astro",
|
|
||||||
".c",
|
|
||||||
".cpp",
|
|
||||||
".css",
|
|
||||||
".go",
|
|
||||||
".h",
|
|
||||||
".hpp",
|
|
||||||
".java",
|
|
||||||
".js",
|
|
||||||
".json",
|
|
||||||
".kt",
|
|
||||||
".mdx",
|
|
||||||
".mjs",
|
|
||||||
".php",
|
|
||||||
".py",
|
|
||||||
".rb",
|
|
||||||
".rs",
|
|
||||||
".svelte",
|
|
||||||
".swift",
|
|
||||||
".ts",
|
|
||||||
".tsx",
|
|
||||||
".txt",
|
|
||||||
".vue",
|
|
||||||
".yaml",
|
|
||||||
".yml",
|
|
||||||
]
|
|
||||||
|
|
||||||
# Code file extensions with their markdown language identifiers for syntax highlighting
|
|
||||||
code_markdown_identifier: ClassVar[dict[str, str]] = {
|
|
||||||
".astro": "astro",
|
|
||||||
".c": "c",
|
|
||||||
".cpp": "cpp",
|
|
||||||
".css": "css",
|
|
||||||
".go": "go",
|
|
||||||
".h": "c",
|
|
||||||
".hpp": "cpp",
|
|
||||||
".java": "java",
|
|
||||||
".js": "javascript",
|
|
||||||
".json": "json",
|
|
||||||
".kt": "kotlin",
|
|
||||||
".mjs": "javascript",
|
|
||||||
".php": "php",
|
|
||||||
".py": "python",
|
|
||||||
".rb": "ruby",
|
|
||||||
".rs": "rust",
|
|
||||||
".svelte": "svelte",
|
|
||||||
".swift": "swift",
|
|
||||||
".ts": "typescript",
|
|
||||||
".tsx": "tsx",
|
|
||||||
".vue": "vue",
|
|
||||||
".yaml": "yaml",
|
|
||||||
".yml": "yaml",
|
|
||||||
}
|
|
||||||
|
|
||||||
extensions: ClassVar[list[str]] = docling_extensions + text_extensions
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def parse_file(path: Path) -> DoclingDocument:
|
|
||||||
try:
|
|
||||||
file_extension = path.suffix.lower()
|
|
||||||
|
|
||||||
if file_extension in FileReader.docling_extensions:
|
|
||||||
# Use docling for complex document formats
|
|
||||||
if not DOCLING_AVAILABLE:
|
|
||||||
raise ImportError(
|
|
||||||
"Docling is required for processing this file type. "
|
|
||||||
"Install with: pip install haiku.rag-slim[docling]"
|
|
||||||
)
|
|
||||||
from docling.document_converter import DocumentConverter
|
|
||||||
|
|
||||||
converter = DocumentConverter()
|
|
||||||
result = converter.convert(path)
|
|
||||||
return result.document
|
|
||||||
elif file_extension in FileReader.text_extensions:
|
|
||||||
# Read plain text files directly
|
|
||||||
content = path.read_text(encoding="utf-8")
|
|
||||||
|
|
||||||
# Wrap code files (but not plain txt) in markdown code blocks for better presentation
|
|
||||||
if file_extension in FileReader.code_markdown_identifier:
|
|
||||||
language = FileReader.code_markdown_identifier[file_extension]
|
|
||||||
content = f"```{language}\n{content}\n```"
|
|
||||||
|
|
||||||
# Convert text to DoclingDocument by wrapping as markdown
|
|
||||||
return text_to_docling_document(content, name=f"{path.stem}.md")
|
|
||||||
else:
|
|
||||||
# Fallback: try to read as text and convert to DoclingDocument
|
|
||||||
content = path.read_text(encoding="utf-8")
|
|
||||||
return text_to_docling_document(content, name=f"{path.stem}.md")
|
|
||||||
except ImportError:
|
|
||||||
raise
|
|
||||||
except Exception as e:
|
|
||||||
raise ValueError(
|
|
||||||
f"Failed to parse file: {path} - {type(e).__name__}: {e}"
|
|
||||||
) from e
|
|
||||||
|
|
@ -151,7 +151,7 @@ class ChunkRepository:
|
||||||
"""Create chunks and embeddings for a document from DoclingDocument."""
|
"""Create chunks and embeddings for a document from DoclingDocument."""
|
||||||
# Lazy imports to avoid loading docling during module import
|
# Lazy imports to avoid loading docling during module import
|
||||||
from haiku.rag.chunker import chunker
|
from haiku.rag.chunker import chunker
|
||||||
from haiku.rag.utils import text_to_docling_document
|
from haiku.rag.converters import get_converter
|
||||||
|
|
||||||
# Optionally preprocess markdown before chunking
|
# Optionally preprocess markdown before chunking
|
||||||
processed_document = document
|
processed_document = document
|
||||||
|
|
@ -166,7 +166,8 @@ class ChunkRepository:
|
||||||
processed_markdown = result
|
processed_markdown = result
|
||||||
if not isinstance(processed_markdown, str):
|
if not isinstance(processed_markdown, str):
|
||||||
raise ValueError("Preprocessor must return a markdown string")
|
raise ValueError("Preprocessor must return a markdown string")
|
||||||
processed_document = text_to_docling_document(
|
converter = get_converter(self.store._config)
|
||||||
|
processed_document = converter.convert_text(
|
||||||
processed_markdown, name="content.md"
|
processed_markdown, name="content.md"
|
||||||
)
|
)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
|
|
||||||
|
|
@ -5,7 +5,6 @@ import sys
|
||||||
from collections.abc import Callable
|
from collections.abc import Callable
|
||||||
from functools import wraps
|
from functools import wraps
|
||||||
from importlib import metadata
|
from importlib import metadata
|
||||||
from io import BytesIO
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from types import ModuleType
|
from types import ModuleType
|
||||||
|
|
||||||
|
|
@ -93,34 +92,6 @@ async def is_up_to_date() -> tuple[bool, Version, Version]:
|
||||||
return running_version >= pypi_version, running_version, pypi_version
|
return running_version >= pypi_version, running_version, pypi_version
|
||||||
|
|
||||||
|
|
||||||
def text_to_docling_document(text: str, name: str = "content.md"):
|
|
||||||
"""Convert text content to a DoclingDocument.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
text: The text content to convert.
|
|
||||||
name: The name to use for the document stream (defaults to "content.md").
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
A DoclingDocument created from the text content.
|
|
||||||
"""
|
|
||||||
try:
|
|
||||||
import docling # noqa: F401
|
|
||||||
except ImportError as e:
|
|
||||||
raise ImportError(
|
|
||||||
"Docling is required for document conversion. "
|
|
||||||
"Install with: pip install haiku.rag-slim[docling]"
|
|
||||||
) from e
|
|
||||||
|
|
||||||
from docling.document_converter import DocumentConverter
|
|
||||||
from docling_core.types.io import DocumentStream
|
|
||||||
|
|
||||||
bytes_io = BytesIO(text.encode("utf-8"))
|
|
||||||
doc_stream = DocumentStream(name=name, stream=bytes_io)
|
|
||||||
converter = DocumentConverter()
|
|
||||||
result = converter.convert(doc_stream)
|
|
||||||
return result.document
|
|
||||||
|
|
||||||
|
|
||||||
def load_callable(path: str):
|
def load_callable(path: str):
|
||||||
"""Load a callable from a dotted path or file path.
|
"""Load a callable from a dotted path or file path.
|
||||||
|
|
||||||
|
|
@ -208,4 +179,4 @@ def prefetch_models():
|
||||||
"POST", f"{base_url}/api/pull", json={"model": model}
|
"POST", f"{base_url}/api/pull", json={"model": model}
|
||||||
) as r:
|
) as r:
|
||||||
for _ in r.iter_lines():
|
for _ in r.iter_lines():
|
||||||
pass
|
pass
|
||||||
31
src/haiku/rag/converters/__init__.py
Normal file
31
src/haiku/rag/converters/__init__.py
Normal file
|
|
@ -0,0 +1,31 @@
|
||||||
|
"""Document converter abstraction for haiku.rag."""
|
||||||
|
|
||||||
|
from haiku.rag.config import AppConfig, Config
|
||||||
|
from haiku.rag.converters.base import DocumentConverter
|
||||||
|
|
||||||
|
__all__ = ["DocumentConverter", "get_converter"]
|
||||||
|
|
||||||
|
|
||||||
|
def get_converter(config: AppConfig = Config) -> DocumentConverter:
|
||||||
|
"""Get a document converter instance based on configuration.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
config: Configuration to use. Defaults to global Config.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DocumentConverter instance configured according to the config.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If the converter provider is not recognized.
|
||||||
|
"""
|
||||||
|
if config.processing.converter == "docling-local":
|
||||||
|
from haiku.rag.converters.docling_local import DoclingLocalConverter
|
||||||
|
|
||||||
|
return DoclingLocalConverter()
|
||||||
|
|
||||||
|
# if config.processing.converter == "docling-serve":
|
||||||
|
# from haiku.rag.converters.docling_serve import DoclingServeConverter
|
||||||
|
#
|
||||||
|
# return DoclingServeConverter(config)
|
||||||
|
|
||||||
|
raise ValueError(f"Unsupported converter provider: {config.processing.converter}")
|
||||||
55
src/haiku/rag/converters/base.py
Normal file
55
src/haiku/rag/converters/base.py
Normal file
|
|
@ -0,0 +1,55 @@
|
||||||
|
"""Base class for document converters."""
|
||||||
|
|
||||||
|
from abc import ABC, abstractmethod
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
|
||||||
|
|
||||||
|
class DocumentConverter(ABC):
|
||||||
|
"""Abstract base class for document converters.
|
||||||
|
|
||||||
|
Document converters are responsible for converting various document formats
|
||||||
|
(PDF, DOCX, HTML, etc.) into DoclingDocument format for further processing.
|
||||||
|
"""
|
||||||
|
|
||||||
|
@property
|
||||||
|
@abstractmethod
|
||||||
|
def supported_extensions(self) -> list[str]:
|
||||||
|
"""Return list of file extensions supported by this converter.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of file extensions (including the dot, e.g., [".pdf", ".docx"]).
|
||||||
|
"""
|
||||||
|
pass
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
def convert_file(self, path: Path) -> DoclingDocument:
|
||||||
|
"""Convert a file to DoclingDocument format.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
path: Path to the file to convert.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DoclingDocument representation of the file.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If the file cannot be converted.
|
||||||
|
"""
|
||||||
|
pass
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
def convert_text(self, text: str, name: str = "content.md") -> DoclingDocument:
|
||||||
|
"""Convert text content to DoclingDocument format.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text: The text content to convert.
|
||||||
|
name: The name to use for the document (defaults to "content.md").
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DoclingDocument representation of the text.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If the text cannot be converted.
|
||||||
|
"""
|
||||||
|
pass
|
||||||
192
src/haiku/rag/converters/docling_local.py
Normal file
192
src/haiku/rag/converters/docling_local.py
Normal file
|
|
@ -0,0 +1,192 @@
|
||||||
|
"""Local docling converter implementation."""
|
||||||
|
|
||||||
|
from io import BytesIO
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import ClassVar
|
||||||
|
|
||||||
|
<<<<<<<< HEAD:haiku_rag_slim/haiku/rag/reader.py
|
||||||
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
from haiku.rag.utils import text_to_docling_document
|
||||||
|
========
|
||||||
|
from docling.document_converter import DocumentConverter as DoclingDocConverter
|
||||||
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
from docling_core.types.io import DocumentStream
|
||||||
|
|
||||||
|
from haiku.rag.converters.base import DocumentConverter
|
||||||
|
>>>>>>>> f7975a3d5946 (Introduce converters for supporting more than local docling document conversion. Transform existing FileReader and utils to "docling-local" converter):src/haiku/rag/converters/docling_local.py
|
||||||
|
|
||||||
|
# Check if docling is available
|
||||||
|
try:
|
||||||
|
import docling # noqa: F401
|
||||||
|
|
||||||
|
DOCLING_AVAILABLE = True
|
||||||
|
except ImportError:
|
||||||
|
DOCLING_AVAILABLE = False
|
||||||
|
|
||||||
|
|
||||||
|
class DoclingLocalConverter(DocumentConverter):
|
||||||
|
"""Converter that uses local docling for document conversion.
|
||||||
|
|
||||||
|
This converter runs docling locally in-process to convert documents.
|
||||||
|
It handles various document formats including PDF, DOCX, HTML, and plain text.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Extensions supported by docling
|
||||||
|
docling_extensions: ClassVar[list[str]] = [
|
||||||
|
".adoc",
|
||||||
|
".asc",
|
||||||
|
".asciidoc",
|
||||||
|
".bmp",
|
||||||
|
".csv",
|
||||||
|
".docx",
|
||||||
|
".html",
|
||||||
|
".xhtml",
|
||||||
|
".jpeg",
|
||||||
|
".jpg",
|
||||||
|
".md",
|
||||||
|
".pdf",
|
||||||
|
".png",
|
||||||
|
".pptx",
|
||||||
|
".tiff",
|
||||||
|
".xlsx",
|
||||||
|
".xml",
|
||||||
|
".webp",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Plain text extensions that we'll read directly
|
||||||
|
text_extensions: ClassVar[list[str]] = [
|
||||||
|
".astro",
|
||||||
|
".c",
|
||||||
|
".cpp",
|
||||||
|
".css",
|
||||||
|
".go",
|
||||||
|
".h",
|
||||||
|
".hpp",
|
||||||
|
".java",
|
||||||
|
".js",
|
||||||
|
".json",
|
||||||
|
".kt",
|
||||||
|
".mdx",
|
||||||
|
".mjs",
|
||||||
|
".php",
|
||||||
|
".py",
|
||||||
|
".rb",
|
||||||
|
".rs",
|
||||||
|
".svelte",
|
||||||
|
".swift",
|
||||||
|
".ts",
|
||||||
|
".tsx",
|
||||||
|
".txt",
|
||||||
|
".vue",
|
||||||
|
".yaml",
|
||||||
|
".yml",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Code file extensions with their markdown language identifiers for syntax highlighting
|
||||||
|
code_markdown_identifier: ClassVar[dict[str, str]] = {
|
||||||
|
".astro": "astro",
|
||||||
|
".c": "c",
|
||||||
|
".cpp": "cpp",
|
||||||
|
".css": "css",
|
||||||
|
".go": "go",
|
||||||
|
".h": "c",
|
||||||
|
".hpp": "cpp",
|
||||||
|
".java": "java",
|
||||||
|
".js": "javascript",
|
||||||
|
".json": "json",
|
||||||
|
".kt": "kotlin",
|
||||||
|
".mjs": "javascript",
|
||||||
|
".php": "php",
|
||||||
|
".py": "python",
|
||||||
|
".rb": "ruby",
|
||||||
|
".rs": "rust",
|
||||||
|
".svelte": "svelte",
|
||||||
|
".swift": "swift",
|
||||||
|
".ts": "typescript",
|
||||||
|
".tsx": "tsx",
|
||||||
|
".vue": "vue",
|
||||||
|
".yaml": "yaml",
|
||||||
|
".yml": "yaml",
|
||||||
|
}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def supported_extensions(self) -> list[str]:
|
||||||
|
"""Return list of file extensions supported by this converter."""
|
||||||
|
return self.docling_extensions + self.text_extensions
|
||||||
|
|
||||||
|
def convert_file(self, path: Path) -> DoclingDocument:
|
||||||
|
"""Convert a file to DoclingDocument using local docling.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
path: Path to the file to convert.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DoclingDocument representation of the file.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If the file cannot be converted.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
file_extension = path.suffix.lower()
|
||||||
|
|
||||||
|
if file_extension in self.docling_extensions:
|
||||||
|
# Use docling for complex document formats
|
||||||
|
<<<<<<<< HEAD:haiku_rag_slim/haiku/rag/reader.py
|
||||||
|
if not DOCLING_AVAILABLE:
|
||||||
|
raise ImportError(
|
||||||
|
"Docling is required for processing this file type. "
|
||||||
|
"Install with: pip install haiku.rag-slim[docling]"
|
||||||
|
)
|
||||||
|
from docling.document_converter import DocumentConverter
|
||||||
|
|
||||||
|
converter = DocumentConverter()
|
||||||
|
========
|
||||||
|
converter = DoclingDocConverter()
|
||||||
|
>>>>>>>> f7975a3d5946 (Introduce converters for supporting more than local docling document conversion. Transform existing FileReader and utils to "docling-local" converter):src/haiku/rag/converters/docling_local.py
|
||||||
|
result = converter.convert(path)
|
||||||
|
return result.document
|
||||||
|
elif file_extension in self.text_extensions:
|
||||||
|
# Read plain text files directly
|
||||||
|
content = path.read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
# Wrap code files (but not plain txt) in markdown code blocks for better presentation
|
||||||
|
if file_extension in self.code_markdown_identifier:
|
||||||
|
language = self.code_markdown_identifier[file_extension]
|
||||||
|
content = f"```{language}\n{content}\n```"
|
||||||
|
|
||||||
|
# Convert text to DoclingDocument by wrapping as markdown
|
||||||
|
return self.convert_text(content, name=f"{path.stem}.md")
|
||||||
|
else:
|
||||||
|
# Fallback: try to read as text and convert to DoclingDocument
|
||||||
|
content = path.read_text(encoding="utf-8")
|
||||||
|
<<<<<<<< HEAD:haiku_rag_slim/haiku/rag/reader.py
|
||||||
|
return text_to_docling_document(content, name=f"{path.stem}.md")
|
||||||
|
except ImportError:
|
||||||
|
raise
|
||||||
|
========
|
||||||
|
return self.convert_text(content, name=f"{path.stem}.md")
|
||||||
|
>>>>>>>> f7975a3d5946 (Introduce converters for supporting more than local docling document conversion. Transform existing FileReader and utils to "docling-local" converter):src/haiku/rag/converters/docling_local.py
|
||||||
|
except Exception:
|
||||||
|
raise ValueError(f"Failed to parse file: {path}")
|
||||||
|
|
||||||
|
def convert_text(self, text: str, name: str = "content.md") -> DoclingDocument:
|
||||||
|
"""Convert text content to DoclingDocument using local docling.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text: The text content to convert.
|
||||||
|
name: The name to use for the document (defaults to "content.md").
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DoclingDocument representation of the text.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If the text cannot be converted.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
bytes_io = BytesIO(text.encode("utf-8"))
|
||||||
|
doc_stream = DocumentStream(name=name, stream=bytes_io)
|
||||||
|
converter = DoclingDocConverter()
|
||||||
|
result = converter.convert(doc_stream)
|
||||||
|
return result.document
|
||||||
|
except Exception as e:
|
||||||
|
raise ValueError(f"Failed to convert text to DoclingDocument: {e}")
|
||||||
|
|
@ -1,12 +1,13 @@
|
||||||
import pytest
|
import pytest
|
||||||
from datasets import Dataset
|
from datasets import Dataset
|
||||||
|
|
||||||
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
from haiku.rag.store.engine import Store
|
from haiku.rag.store.engine import Store
|
||||||
from haiku.rag.store.models.chunk import Chunk
|
from haiku.rag.store.models.chunk import Chunk
|
||||||
from haiku.rag.store.models.document import Document
|
from haiku.rag.store.models.document import Document
|
||||||
from haiku.rag.store.repositories.chunk import ChunkRepository
|
from haiku.rag.store.repositories.chunk import ChunkRepository
|
||||||
from haiku.rag.store.repositories.document import DocumentRepository
|
from haiku.rag.store.repositories.document import DocumentRepository
|
||||||
from haiku.rag.utils import text_to_docling_document
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
|
|
@ -23,9 +24,8 @@ async def test_chunk_repository_operations(qa_corpus: Dataset, temp_db_path):
|
||||||
|
|
||||||
# Create a document first with chunks
|
# Create a document first with chunks
|
||||||
document = Document(content=document_text, metadata={"source": "test"})
|
document = Document(content=document_text, metadata={"source": "test"})
|
||||||
from haiku.rag.utils import text_to_docling_document
|
converter = get_converter(Config)
|
||||||
|
docling_document = converter.convert_text(document_text, name="test.md")
|
||||||
docling_document = text_to_docling_document(document_text, name="test.md")
|
|
||||||
created_document = await doc_repo._create_and_chunk(document, docling_document)
|
created_document = await doc_repo._create_and_chunk(document, docling_document)
|
||||||
assert created_document.id is not None
|
assert created_document.id is not None
|
||||||
|
|
||||||
|
|
@ -70,7 +70,8 @@ async def test_create_chunks_for_document(qa_corpus: Dataset, temp_db_path):
|
||||||
assert document_id is not None, "Document ID should not be None"
|
assert document_id is not None, "Document ID should not be None"
|
||||||
|
|
||||||
# Convert text to DoclingDocument
|
# Convert text to DoclingDocument
|
||||||
docling_document = text_to_docling_document(document_text, name="test.md")
|
converter = get_converter(Config)
|
||||||
|
docling_document = converter.convert_text(document_text, name="test.md")
|
||||||
|
|
||||||
# Test creating chunks for the document
|
# Test creating chunks for the document
|
||||||
chunks = await chunk_repo.create_chunks_for_document(document_id, docling_document)
|
chunks = await chunk_repo.create_chunks_for_document(document_id, docling_document)
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,8 @@ import pytest
|
||||||
from datasets import Dataset
|
from datasets import Dataset
|
||||||
|
|
||||||
from haiku.rag.chunker import Chunker
|
from haiku.rag.chunker import Chunker
|
||||||
from haiku.rag.utils import text_to_docling_document
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
|
|
@ -11,7 +12,8 @@ async def test_chunker(qa_corpus: Dataset):
|
||||||
doc_text = qa_corpus[0]["document_extracted"]
|
doc_text = qa_corpus[0]["document_extracted"]
|
||||||
|
|
||||||
# Convert text to DoclingDocument
|
# Convert text to DoclingDocument
|
||||||
doc = text_to_docling_document(doc_text, name="test.md")
|
converter = get_converter(Config)
|
||||||
|
doc = converter.convert_text(doc_text, name="test.md")
|
||||||
|
|
||||||
chunks = await chunker.chunk(doc)
|
chunks = await chunker.chunk(doc)
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,8 @@
|
||||||
import pytest
|
import pytest
|
||||||
from datasets import Dataset
|
from datasets import Dataset
|
||||||
|
|
||||||
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
from haiku.rag.store.engine import Store
|
from haiku.rag.store.engine import Store
|
||||||
from haiku.rag.store.models.document import Document
|
from haiku.rag.store.models.document import Document
|
||||||
from haiku.rag.store.repositories.document import DocumentRepository
|
from haiku.rag.store.repositories.document import DocumentRepository
|
||||||
|
|
@ -24,9 +26,8 @@ async def test_create_document_with_chunks(qa_corpus: Dataset, temp_db_path):
|
||||||
)
|
)
|
||||||
|
|
||||||
# Convert text to DoclingDocument for chunk creation
|
# Convert text to DoclingDocument for chunk creation
|
||||||
from haiku.rag.utils import text_to_docling_document
|
converter = get_converter(Config)
|
||||||
|
docling_document = converter.convert_text(document_text, name="test.md")
|
||||||
docling_document = text_to_docling_document(document_text, name="test.md")
|
|
||||||
|
|
||||||
# Create the document with chunks in the database
|
# Create the document with chunks in the database
|
||||||
created_document = await doc_repo._create_and_chunk(document, docling_document)
|
created_document = await doc_repo._create_and_chunk(document, docling_document)
|
||||||
|
|
|
||||||
|
|
@ -3,11 +3,11 @@ from pathlib import Path
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from haiku.rag.config import Config
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
from haiku.rag.store.engine import Store
|
from haiku.rag.store.engine import Store
|
||||||
from haiku.rag.store.models.document import Document
|
from haiku.rag.store.models.document import Document
|
||||||
from haiku.rag.store.repositories.chunk import ChunkRepository
|
from haiku.rag.store.repositories.chunk import ChunkRepository
|
||||||
from haiku.rag.store.repositories.document import DocumentRepository
|
from haiku.rag.store.repositories.document import DocumentRepository
|
||||||
from haiku.rag.utils import text_to_docling_document
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
@pytest.mark.parametrize(
|
||||||
|
|
@ -63,7 +63,8 @@ def add_marker(text: str) -> str:
|
||||||
|
|
||||||
chunk_repo.embedder.embed = fake_embed # type: ignore[assignment]
|
chunk_repo.embedder.embed = fake_embed # type: ignore[assignment]
|
||||||
|
|
||||||
docling = text_to_docling_document(document.content, name="test.md")
|
converter = get_converter(Config)
|
||||||
|
docling = converter.convert_text(document.content, name="test.md")
|
||||||
chunks = await chunk_repo.create_chunks_for_document(created_doc.id, docling)
|
chunks = await chunk_repo.create_chunks_for_document(created_doc.id, docling)
|
||||||
|
|
||||||
assert any(marker in c.content for c in chunks)
|
assert any(marker in c.content for c in chunks)
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,8 @@
|
||||||
import tempfile
|
import tempfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from haiku.rag.reader import FileReader
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
|
|
||||||
|
|
||||||
def test_code_file_wrapped_in_code_block():
|
def test_code_file_wrapped_in_code_block():
|
||||||
|
|
@ -15,7 +16,8 @@ def test_code_file_wrapped_in_code_block():
|
||||||
f.flush()
|
f.flush()
|
||||||
temp_path = Path(f.name)
|
temp_path = Path(f.name)
|
||||||
|
|
||||||
document = FileReader.parse_file(temp_path)
|
converter = get_converter(Config)
|
||||||
|
document = converter.convert_file(temp_path)
|
||||||
result = document.export_to_markdown()
|
result = document.export_to_markdown()
|
||||||
|
|
||||||
assert result.startswith("```\n")
|
assert result.startswith("```\n")
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,8 @@
|
||||||
import pytest
|
import pytest
|
||||||
from datasets import Dataset
|
from datasets import Dataset
|
||||||
|
|
||||||
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
from haiku.rag.store.engine import Store
|
from haiku.rag.store.engine import Store
|
||||||
from haiku.rag.store.models.document import Document
|
from haiku.rag.store.models.document import Document
|
||||||
from haiku.rag.store.repositories.chunk import ChunkRepository
|
from haiku.rag.store.repositories.chunk import ChunkRepository
|
||||||
|
|
@ -33,9 +35,8 @@ async def test_search_qa_corpus(qa_corpus: Dataset, temp_db_path):
|
||||||
document = Document(content=document_text)
|
document = Document(content=document_text)
|
||||||
|
|
||||||
# Create the document with chunks and embeddings
|
# Create the document with chunks and embeddings
|
||||||
from haiku.rag.utils import text_to_docling_document
|
converter = get_converter(Config)
|
||||||
|
docling_document = converter.convert_text(document_text, name="test.md")
|
||||||
docling_document = text_to_docling_document(document_text, name="test.md")
|
|
||||||
created_document = await doc_repo._create_and_chunk(document, docling_document)
|
created_document = await doc_repo._create_and_chunk(document, docling_document)
|
||||||
documents.append((created_document, doc_data))
|
documents.append((created_document, doc_data))
|
||||||
|
|
||||||
|
|
@ -81,9 +82,8 @@ async def test_chunks_include_document_info(temp_db_path):
|
||||||
)
|
)
|
||||||
|
|
||||||
# Create the document with chunks
|
# Create the document with chunks
|
||||||
from haiku.rag.utils import text_to_docling_document
|
converter = get_converter(Config)
|
||||||
|
docling_document = converter.convert_text(document.content, name="test.md")
|
||||||
docling_document = text_to_docling_document(document.content, name="test.md")
|
|
||||||
created_document = await doc_repo._create_and_chunk(document, docling_document)
|
created_document = await doc_repo._create_and_chunk(document, docling_document)
|
||||||
|
|
||||||
# Search for chunks
|
# Search for chunks
|
||||||
|
|
@ -119,9 +119,8 @@ async def test_chunks_include_document_title(temp_db_path):
|
||||||
)
|
)
|
||||||
|
|
||||||
# Create the document with chunks
|
# Create the document with chunks
|
||||||
from haiku.rag.utils import text_to_docling_document
|
converter = get_converter(Config)
|
||||||
|
dl = converter.convert_text(document.content, name="title-test.md")
|
||||||
dl = text_to_docling_document(document.content, name="title-test.md")
|
|
||||||
await doc_repo._create_and_chunk(document, dl)
|
await doc_repo._create_and_chunk(document, dl)
|
||||||
|
|
||||||
# Perform a search that should find this document
|
# Perform a search that should find this document
|
||||||
|
|
@ -151,11 +150,10 @@ async def test_search_score_types(temp_db_path):
|
||||||
"Computer vision systems can interpret and analyze visual information from images.",
|
"Computer vision systems can interpret and analyze visual information from images.",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
converter = get_converter(Config)
|
||||||
for content in documents_content:
|
for content in documents_content:
|
||||||
document = Document(content=content)
|
document = Document(content=content)
|
||||||
from haiku.rag.utils import text_to_docling_document
|
docling_document = converter.convert_text(content, name="test.md")
|
||||||
|
|
||||||
docling_document = text_to_docling_document(content, name="test.md")
|
|
||||||
await doc_repo._create_and_chunk(document, docling_document)
|
await doc_repo._create_and_chunk(document, docling_document)
|
||||||
|
|
||||||
query = "machine learning"
|
query = "machine learning"
|
||||||
|
|
|
||||||
|
|
@ -1,11 +1,13 @@
|
||||||
from haiku.rag.utils import text_to_docling_document
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
|
|
||||||
|
|
||||||
def test_text_to_docling_document():
|
def test_text_to_docling_document():
|
||||||
"""Test the text_to_docling_document utility function."""
|
"""Test text to DoclingDocument conversion."""
|
||||||
# Test basic text conversion
|
# Test basic text conversion
|
||||||
simple_text = "This is a simple text document."
|
simple_text = "This is a simple text document."
|
||||||
doc = text_to_docling_document(simple_text)
|
converter = get_converter(Config)
|
||||||
|
doc = converter.convert_text(simple_text)
|
||||||
|
|
||||||
# Verify it returns a DoclingDocument
|
# Verify it returns a DoclingDocument
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
|
@ -18,7 +20,7 @@ def test_text_to_docling_document():
|
||||||
|
|
||||||
|
|
||||||
def test_text_to_docling_document_with_custom_name():
|
def test_text_to_docling_document_with_custom_name():
|
||||||
"""Test text_to_docling_document with custom name parameter."""
|
"""Test text to DoclingDocument conversion with custom name parameter."""
|
||||||
code_text = """# Python Code
|
code_text = """# Python Code
|
||||||
|
|
||||||
```python
|
```python
|
||||||
|
|
@ -27,7 +29,8 @@ def hello():
|
||||||
return True
|
return True
|
||||||
```"""
|
```"""
|
||||||
|
|
||||||
doc = text_to_docling_document(code_text, name="hello.md")
|
converter = get_converter(Config)
|
||||||
|
doc = converter.convert_text(code_text, name="hello.md")
|
||||||
|
|
||||||
# Verify it's a valid DoclingDocument
|
# Verify it's a valid DoclingDocument
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
|
@ -41,7 +44,7 @@ def hello():
|
||||||
|
|
||||||
|
|
||||||
def test_text_to_docling_document_markdown_content():
|
def test_text_to_docling_document_markdown_content():
|
||||||
"""Test text_to_docling_document with markdown content."""
|
"""Test text to DoclingDocument conversion with markdown content."""
|
||||||
markdown_text = """# Test Document
|
markdown_text = """# Test Document
|
||||||
|
|
||||||
This is a test document with:
|
This is a test document with:
|
||||||
|
|
@ -58,7 +61,8 @@ def test():
|
||||||
|
|
||||||
**Bold text** and *italic text*."""
|
**Bold text** and *italic text*."""
|
||||||
|
|
||||||
doc = text_to_docling_document(markdown_text, name="test.md")
|
converter = get_converter(Config)
|
||||||
|
doc = converter.convert_text(markdown_text, name="test.md")
|
||||||
|
|
||||||
# Verify it's a DoclingDocument
|
# Verify it's a DoclingDocument
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
|
@ -73,8 +77,9 @@ def test():
|
||||||
|
|
||||||
|
|
||||||
def test_text_to_docling_document_empty_content():
|
def test_text_to_docling_document_empty_content():
|
||||||
"""Test text_to_docling_document with empty content."""
|
"""Test text to DoclingDocument conversion with empty content."""
|
||||||
doc = text_to_docling_document("")
|
converter = get_converter(Config)
|
||||||
|
doc = converter.convert_text("")
|
||||||
|
|
||||||
# Should still create a valid DoclingDocument
|
# Should still create a valid DoclingDocument
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
|
@ -87,7 +92,7 @@ def test_text_to_docling_document_empty_content():
|
||||||
|
|
||||||
|
|
||||||
def test_text_to_docling_document_unicode_content():
|
def test_text_to_docling_document_unicode_content():
|
||||||
"""Test text_to_docling_document with unicode content."""
|
"""Test text to DoclingDocument conversion with unicode content."""
|
||||||
unicode_text = """# 测试文档
|
unicode_text = """# 测试文档
|
||||||
|
|
||||||
这是一个包含中文的测试文档。
|
这是一个包含中文的测试文档。
|
||||||
|
|
@ -101,7 +106,8 @@ function saludar() {
|
||||||
|
|
||||||
Emoji test: 🚀 ✅ 📝"""
|
Emoji test: 🚀 ✅ 📝"""
|
||||||
|
|
||||||
doc = text_to_docling_document(unicode_text, name="unicode.md")
|
converter = get_converter(Config)
|
||||||
|
doc = converter.convert_text(unicode_text, name="unicode.md")
|
||||||
|
|
||||||
# Verify it's a DoclingDocument
|
# Verify it's a DoclingDocument
|
||||||
from docling_core.types.doc.document import DoclingDocument
|
from docling_core.types.doc.document import DoclingDocument
|
||||||
|
|
|
||||||
|
|
@ -1,10 +1,11 @@
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
|
from haiku.rag.config import Config
|
||||||
|
from haiku.rag.converters import get_converter
|
||||||
from haiku.rag.store.engine import Store
|
from haiku.rag.store.engine import Store
|
||||||
from haiku.rag.store.models.document import Document
|
from haiku.rag.store.models.document import Document
|
||||||
from haiku.rag.store.repositories.chunk import ChunkRepository
|
from haiku.rag.store.repositories.chunk import ChunkRepository
|
||||||
from haiku.rag.store.repositories.document import DocumentRepository
|
from haiku.rag.store.repositories.document import DocumentRepository
|
||||||
from haiku.rag.utils import text_to_docling_document
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
|
|
@ -34,7 +35,8 @@ async def test_version_rollback_on_create_failure(temp_db_path):
|
||||||
# Attempt to create document with chunks; expect failure and rollback
|
# Attempt to create document with chunks; expect failure and rollback
|
||||||
content = "Hello, rollback!"
|
content = "Hello, rollback!"
|
||||||
doc = Document(content=content)
|
doc = Document(content=content)
|
||||||
dl_doc = text_to_docling_document(content, name="test.md")
|
converter = get_converter(Config)
|
||||||
|
dl_doc = converter.convert_text(content, name="test.md")
|
||||||
|
|
||||||
with pytest.raises(RuntimeError):
|
with pytest.raises(RuntimeError):
|
||||||
await repo._create_and_chunk(doc, dl_doc)
|
await repo._create_and_chunk(doc, dl_doc)
|
||||||
|
|
@ -65,7 +67,8 @@ async def test_version_rollback_on_update_failure(temp_db_path):
|
||||||
# Create a valid document first (with real chunking and stubbed embeddings)
|
# Create a valid document first (with real chunking and stubbed embeddings)
|
||||||
base_content = "Base content"
|
base_content = "Base content"
|
||||||
base_doc = Document(content=base_content)
|
base_doc = Document(content=base_content)
|
||||||
base_dl = text_to_docling_document(base_content, name="base.md")
|
converter = get_converter(Config)
|
||||||
|
base_dl = converter.convert_text(base_content, name="base.md")
|
||||||
created = await repo._create_and_chunk(base_doc, base_dl)
|
created = await repo._create_and_chunk(base_doc, base_dl)
|
||||||
|
|
||||||
# Force new chunk creation to fail during update after writing
|
# Force new chunk creation to fail during update after writing
|
||||||
|
|
@ -80,7 +83,7 @@ async def test_version_rollback_on_update_failure(temp_db_path):
|
||||||
# Attempt update
|
# Attempt update
|
||||||
updated_content = "Updated content"
|
updated_content = "Updated content"
|
||||||
created.content = updated_content
|
created.content = updated_content
|
||||||
updated_dl = text_to_docling_document(updated_content, name="updated.md")
|
updated_dl = converter.convert_text(updated_content, name="updated.md")
|
||||||
|
|
||||||
with pytest.raises(RuntimeError):
|
with pytest.raises(RuntimeError):
|
||||||
await repo._update_and_rechunk(created, updated_dl)
|
await repo._update_and_rechunk(created, updated_dl)
|
||||||
|
|
@ -140,13 +143,14 @@ async def test_vacuum_with_retention_threshold(temp_db_path):
|
||||||
repo.chunk_repository.embedder.embed = fake_embed # type: ignore[assignment]
|
repo.chunk_repository.embedder.embed = fake_embed # type: ignore[assignment]
|
||||||
|
|
||||||
# Create first document
|
# Create first document
|
||||||
|
converter = get_converter(Config)
|
||||||
doc1 = Document(content="First document")
|
doc1 = Document(content="First document")
|
||||||
dl_doc1 = text_to_docling_document("First document", name="doc1.md")
|
dl_doc1 = converter.convert_text("First document", name="doc1.md")
|
||||||
await repo._create_and_chunk(doc1, dl_doc1)
|
await repo._create_and_chunk(doc1, dl_doc1)
|
||||||
|
|
||||||
# Create second document
|
# Create second document
|
||||||
doc2 = Document(content="Second document")
|
doc2 = Document(content="Second document")
|
||||||
dl_doc2 = text_to_docling_document("Second document", name="doc2.md")
|
dl_doc2 = converter.convert_text("Second document", name="doc2.md")
|
||||||
await repo._create_and_chunk(doc2, dl_doc2)
|
await repo._create_and_chunk(doc2, dl_doc2)
|
||||||
|
|
||||||
# Get initial version counts (should have multiple versions from creates)
|
# Get initial version counts (should have multiple versions from creates)
|
||||||
|
|
@ -201,7 +205,6 @@ async def test_vacuum_completes_before_context_exit(temp_db_path, monkeypatch):
|
||||||
"""Test that background vacuum completes when context manager exits."""
|
"""Test that background vacuum completes when context manager exits."""
|
||||||
from haiku.rag.client import HaikuRAG
|
from haiku.rag.client import HaikuRAG
|
||||||
from haiku.rag.config import Config
|
from haiku.rag.config import Config
|
||||||
from haiku.rag.utils import text_to_docling_document
|
|
||||||
|
|
||||||
# Set aggressive vacuum retention for this test
|
# Set aggressive vacuum retention for this test
|
||||||
monkeypatch.setattr(Config.storage, "vacuum_retention_seconds", 0)
|
monkeypatch.setattr(Config.storage, "vacuum_retention_seconds", 0)
|
||||||
|
|
@ -209,9 +212,10 @@ async def test_vacuum_completes_before_context_exit(temp_db_path, monkeypatch):
|
||||||
async with HaikuRAG(db_path=temp_db_path) as client:
|
async with HaikuRAG(db_path=temp_db_path) as client:
|
||||||
# Create multiple documents - each creation triggers automatic vacuum with retention=0
|
# Create multiple documents - each creation triggers automatic vacuum with retention=0
|
||||||
# This aggressively cleans up old versions between operations
|
# This aggressively cleans up old versions between operations
|
||||||
|
converter = get_converter(Config)
|
||||||
for i in range(3):
|
for i in range(3):
|
||||||
doc = Document(content=f"Test document {i}")
|
doc = Document(content=f"Test document {i}")
|
||||||
dl_doc = text_to_docling_document(f"Test document {i}", name=f"test{i}.md")
|
dl_doc = converter.convert_text(f"Test document {i}", name=f"test{i}.md")
|
||||||
await client.document_repository._create_and_chunk(doc, dl_doc)
|
await client.document_repository._create_and_chunk(doc, dl_doc)
|
||||||
|
|
||||||
# After context exit, automatic vacuum should have kept versions minimal
|
# After context exit, automatic vacuum should have kept versions minimal
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue