From 8fe8b6d291b0c03e1f5302e0488aa860c225048e Mon Sep 17 00:00:00 2001 From: Chris McDonough Date: Mon, 22 Jun 2026 13:37:22 -0400 Subject: [PATCH] Thread chunker.chunk() off the asyncio event loop Extracts the CPU-bound HybridChunker/HierarchicalChunker work into a sync helper and wraps it with asyncio.to_thread so large documents don't block the event loop during chunking. Fixes #455. --- .../haiku/rag/chunkers/docling_local.py | 41 +++++++++++-------- 1 file changed, 25 insertions(+), 16 deletions(-) diff --git a/haiku_rag_slim/haiku/rag/chunkers/docling_local.py b/haiku_rag_slim/haiku/rag/chunkers/docling_local.py index fc07bb22..42bfaffe 100644 --- a/haiku_rag_slim/haiku/rag/chunkers/docling_local.py +++ b/haiku_rag_slim/haiku/rag/chunkers/docling_local.py @@ -1,3 +1,4 @@ +import asyncio from functools import cache from typing import TYPE_CHECKING, cast @@ -105,24 +106,12 @@ class DoclingLocalChunker(DocumentChunker): "Must be 'hybrid' or 'hierarchical'." ) - async def chunk(self, document: "DoclingDocument") -> list[Chunk]: - """Split the document into chunks with metadata. + def _chunk_sync(self, document: "DoclingDocument") -> list[Chunk]: + """Synchronous chunking helper (CPU-bound, no I/O). - Extracts structured metadata from each DocChunk including: - - doc_item_refs: JSON pointer references to DocItems (e.g., "#/texts/5") - - headings: Section heading hierarchy - - labels: Semantic labels for each doc_item (e.g., "paragraph", "table") - - page_numbers: Page numbers where content appears - - Args: - document: The DoclingDocument to be split into chunks. - - Returns: - List of Chunk containing content and structured metadata. + Runs the underlying HybridChunker/HierarchicalChunker and extracts + structured metadata from each DocChunk. """ - if document is None: - return [] - raw_chunks = list(self.chunker.chunk(document)) result: list[Chunk] = [] @@ -172,3 +161,23 @@ class DoclingLocalChunker(DocumentChunker): ) return result + + async def chunk(self, document: "DoclingDocument") -> list[Chunk]: + """Split the document into chunks with metadata. + + Extracts structured metadata from each DocChunk including: + - doc_item_refs: JSON pointer references to DocItems (e.g., "#/texts/5") + - headings: Section heading hierarchy + - labels: Semantic labels for each doc_item (e.g., "paragraph", "table") + - page_numbers: Page numbers where content appears + + Args: + document: The DoclingDocument to be split into chunks. + + Returns: + List of Chunk containing content and structured metadata. + """ + if document is None: + return [] + + return await asyncio.to_thread(self._chunk_sync, document)