Add doctor --duplicates-out YAML export; summarize terminal report

This commit is contained in:
Yiorgis Gozadinos 2026-06-26 16:44:22 +03:00
parent 044ac62e49
commit f1e1a16f9b
No known key found for this signature in database
6 changed files with 129 additions and 23 deletions

View file

@ -3,7 +3,7 @@
### Added ### Added
- `doctor` reports groups of near-duplicate documents (revisions sharing most of their chunks), flagging the largest member as the likely one to keep. Tuned via the `doctor.duplicates` config block (`containment_threshold`, `candidate_threshold`, `twin_similarity`, `min_chunks`). - `doctor` reports groups of near-duplicate documents (revisions sharing most of their chunks), flagging the largest member as the likely one to keep; `--duplicates-out PATH` writes the groups to a YAML file. Tuned via the `doctor.duplicates` config block (`containment_threshold`, `candidate_threshold`, `twin_similarity`, `min_chunks`).
### Security ### Security

View file

@ -289,9 +289,11 @@ At the end, a separate "Versions" section lists runtime package versions:
Check the database for consistency problems and print a pass/warn/fail report: Check the database for consistency problems and print a pass/warn/fail report:
```bash ```bash
haiku-rag doctor [--db /path/to/your.lancedb] haiku-rag doctor [--db /path/to/your.lancedb] [--duplicates-out groups.yaml]
``` ```
`--duplicates-out PATH` additionally writes the near-duplicate document groups to a YAML file (one block per group with `keep` and a list of `documents`, each carrying `document_id`, `document`, `chunks`, `contained_fraction`, and `keep_suggested`) for offline review.
Checks include: Checks include:
- required tables are present - required tables are present

View file

@ -198,7 +198,7 @@ class HaikuRAGApp: # pragma: no cover
f" [repr.attrib_name]docling-document schema[/repr.attrib_name]: {info.packages['docling_document_schema']}" f" [repr.attrib_name]docling-document schema[/repr.attrib_name]: {info.packages['docling_document_schema']}"
) )
async def doctor(self) -> bool: async def doctor(self, duplicates_out: Path | None = None) -> bool:
"""Run health checks and print a report. Returns True if any check failed.""" """Run health checks and print a report. Returns True if any check failed."""
import os import os
@ -213,7 +213,9 @@ class HaikuRAGApp: # pragma: no cover
self.console.print("[red]Database path does not exist.[/red]") self.console.print("[red]Database path does not exist.[/red]")
return True return True
report = await run_doctor(self.config, self.db_path, dict(os.environ)) report = await run_doctor(
self.config, self.db_path, dict(os.environ), duplicates_out=duplicates_out
)
glyphs = { glyphs = {
Severity.OK: "[green]✓[/green]", Severity.OK: "[green]✓[/green]",
@ -245,6 +247,10 @@ class HaikuRAGApp: # pragma: no cover
f"[yellow]{report.count(Severity.WARN)} warning(s)[/yellow], " f"[yellow]{report.count(Severity.WARN)} warning(s)[/yellow], "
f"[red]{report.count(Severity.FAIL)} failure(s)[/red]" f"[red]{report.count(Severity.FAIL)} failure(s)[/red]"
) )
if duplicates_out is not None:
self.console.print(
f"[dim]Duplicate-document groups written to {duplicates_out}[/dim]"
)
return report.failed return report.failed
async def history(self, table: str | None = None, limit: int | None = None): async def history(self, table: str | None = None, limit: int | None = None):

View file

@ -593,9 +593,14 @@ def doctor( # pragma: no cover
"--db", "--db",
help="Path to the LanceDB database file", help="Path to the LanceDB database file",
), ),
duplicates_out: Path | None = typer.Option(
None,
"--duplicates-out",
help="Write near-duplicate document groups to this YAML file",
),
): ):
app = create_app(db) app = create_app(db)
if asyncio.run(app.doctor()): if asyncio.run(app.doctor(duplicates_out=duplicates_out)):
raise typer.Exit(code=1) raise typer.Exit(code=1)

View file

@ -1,11 +1,12 @@
import asyncio import asyncio
import json import json
from collections.abc import Mapping from collections.abc import Callable, Mapping
from enum import StrEnum from enum import StrEnum
from pathlib import Path from pathlib import Path
import httpx import httpx
import numpy as np import numpy as np
import yaml
from pydantic import BaseModel, Field from pydantic import BaseModel, Field
from haiku.rag.config import AppConfig from haiku.rag.config import AppConfig
@ -243,6 +244,7 @@ class _DuplicateFamily(BaseModel):
members: list[str] members: list[str]
superset: str superset: str
pairs: list[tuple[str, str, float, float]] pairs: list[tuple[str, str, float, float]]
sizes: dict[str, int]
def _unit(vector: np.ndarray) -> np.ndarray: def _unit(vector: np.ndarray) -> np.ndarray:
@ -335,7 +337,12 @@ def _duplicate_families(
if i in component and j in component if i in component and j in component
) )
families.append( families.append(
_DuplicateFamily(members=members, superset=superset, pairs=pairs) _DuplicateFamily(
members=members,
superset=superset,
pairs=pairs,
sizes={d: int(normalized[d].shape[0]) for d in members},
)
) )
return sorted(families, key=lambda f: f.members) return sorted(families, key=lambda f: f.members)
@ -356,13 +363,52 @@ def _common_path_prefix(labels: list[str]) -> str:
return lo[: cut + 1] if cut > 16 else "" return lo[: cut + 1] if cut > 16 else ""
def _write_duplicates_out(
path: Path, families: list[_DuplicateFamily], label: Callable[[str], str]
) -> None:
"""One block per group; ``keep_suggested`` marks the superset and
``contained_fraction`` is how much of the document is covered by the rest."""
groups = []
for n, family in enumerate(families, start=1):
contained = dict.fromkeys(family.members, 0.0)
for a, b, a_to_b, b_to_a in family.pairs:
contained[a] = max(contained[a], a_to_b)
contained[b] = max(contained[b], b_to_a)
groups.append(
{
"group": n,
"keep": family.superset,
"documents": [
{
"document_id": member,
"document": label(member),
"chunks": family.sizes[member],
"contained_fraction": round(contained[member], 3),
"keep_suggested": member == family.superset,
}
for member in family.members
],
}
)
with open(path, "w", encoding="utf-8") as handle:
yaml.safe_dump({"groups": groups}, handle, sort_keys=False, allow_unicode=True)
def _check_duplicate_documents( def _check_duplicate_documents(
doc_vectors: dict[str, np.ndarray], doc_vectors: dict[str, np.ndarray],
uri_by_doc: Mapping[str, str | None], uri_by_doc: Mapping[str, str | None],
title_by_doc: Mapping[str, str | None], title_by_doc: Mapping[str, str | None],
cfg: DuplicateDetectionConfig, cfg: DuplicateDetectionConfig,
yaml_path: Path | None = None,
) -> CheckResult: ) -> CheckResult:
families = _duplicate_families(doc_vectors, cfg) families = _duplicate_families(doc_vectors, cfg)
def label(doc_id: str) -> str:
return uri_by_doc.get(doc_id) or title_by_doc.get(doc_id) or doc_id
if yaml_path is not None:
_write_duplicates_out(yaml_path, families, label)
if not families: if not families:
return CheckResult( return CheckResult(
name="duplicate_documents", name="duplicate_documents",
@ -370,21 +416,20 @@ def _check_duplicate_documents(
message="No near-duplicate documents detected.", message="No near-duplicate documents detected.",
) )
def label(doc_id: str) -> str: # The terminal report is a summary: show the first few groups whole and
return uri_by_doc.get(doc_id) or title_by_doc.get(doc_id) or doc_id # point at the YAML export for the rest. One block per shown group — a
# header, each member on its own numbered line, then a compact overlap line.
prefix = _common_path_prefix([label(m) for f in families for m in f.members]) shown = families[:_SAMPLE_LIMIT]
prefix = _common_path_prefix([label(m) for f in shown for m in f.members])
def short(doc_id: str) -> str: def short(doc_id: str) -> str:
text = label(doc_id) text = label(doc_id)
return text[len(prefix) :] if prefix and text.startswith(prefix) else text return text[len(prefix) :] if prefix and text.startswith(prefix) else text
# One block per group: a header, each member on its own numbered line, then a
# compact overlap summary referencing those numbers. Not truncated.
details: list[str] = [] details: list[str] = []
if prefix: if prefix:
details.append(f"common path: {prefix}") details.append(f"common path: {prefix}")
for n, family in enumerate(families, start=1): for n, family in enumerate(shown, start=1):
number = {member: i for i, member in enumerate(family.members, start=1)} number = {member: i for i, member in enumerate(family.members, start=1)}
details.append( details.append(
f"group {n}{len(family.members)} docs, keep #{number[family.superset]}:" f"group {n}{len(family.members)} docs, keep #{number[family.superset]}:"
@ -396,6 +441,11 @@ def _check_duplicate_documents(
for a, b, ab, ba in family.pairs for a, b, ab, ba in family.pairs
) )
details.append(f" overlap: {overlaps}") details.append(f" overlap: {overlaps}")
if len(families) > len(shown):
details.append(
f"... (+{len(families) - len(shown)} more groups; "
"use --duplicates-out to export all)"
)
total_docs = sum(len(f.members) for f in families) total_docs = sum(len(f.members) for f in families)
return CheckResult( return CheckResult(
@ -413,7 +463,10 @@ def _check_duplicate_documents(
async def run_db_checks( async def run_db_checks(
store: Store, config: AppConfig, stats: dict store: Store,
config: AppConfig,
stats: dict,
duplicates_out: Path | None = None,
) -> list[CheckResult]: ) -> list[CheckResult]:
"""Referential and content-integrity checks against an open read-only Store. """Referential and content-integrity checks against an open read-only Store.
@ -614,7 +667,11 @@ async def run_db_checks(
doc_vectors = {doc_id: vectors[idx] for doc_id, idx in indices_by_doc.items()} doc_vectors = {doc_id: vectors[idx] for doc_id, idx in indices_by_doc.items()}
results.append( results.append(
_check_duplicate_documents( _check_duplicate_documents(
doc_vectors, uri_by_doc, title_by_doc, config.doctor.duplicates doc_vectors,
uri_by_doc,
title_by_doc,
config.doctor.duplicates,
yaml_path=duplicates_out,
) )
) )
@ -925,7 +982,10 @@ async def run_provider_checks(config: AppConfig) -> list[CheckResult]:
async def run_doctor( async def run_doctor(
config: AppConfig, db_path: Path, environ: dict[str, str] config: AppConfig,
db_path: Path,
environ: dict[str, str],
duplicates_out: Path | None = None,
) -> DoctorReport: ) -> DoctorReport:
"""Open the database read-only and run every diagnostic check. """Open the database read-only and run every diagnostic check.
@ -956,7 +1016,9 @@ async def run_doctor(
read_only=True, read_only=True,
skip_migration_check=True, skip_migration_check=True,
) as store: ) as store:
results += await run_db_checks(store, config, stats) results += await run_db_checks(
store, config, stats, duplicates_out=duplicates_out
)
results.append(_check_api_keys(config, environ)) results.append(_check_api_keys(config, environ))
results += await run_provider_checks(config) results += await run_provider_checks(config)

View file

@ -5,6 +5,7 @@ from unittest.mock import AsyncMock, MagicMock
import lancedb import lancedb
import numpy as np import numpy as np
import pytest import pytest
import yaml
from typer.testing import CliRunner from typer.testing import CliRunner
from haiku.rag.cli import _cli as cli from haiku.rag.cli import _cli as cli
@ -1022,8 +1023,8 @@ def test_duplicate_families_threshold_is_configurable():
assert set(flagged[0].members) == {"a", "b"} assert set(flagged[0].members) == {"a", "b"}
def test_duplicate_documents_report_lists_all_groups_untruncated(): def test_duplicate_documents_report_truncates_summary():
pairs = 7 # more than the old detail cap of 5 pairs = 7 # more than the terminal detail cap of 5
spec: dict[str, list[int]] = {} spec: dict[str, list[int]] = {}
for k in range(pairs): for k in range(pairs):
idx = [3 * k, 3 * k + 1, 3 * k + 2] idx = [3 * k, 3 * k + 1, 3 * k + 2]
@ -1033,11 +1034,12 @@ def test_duplicate_documents_report_lists_all_groups_untruncated():
uris = {d: f"file:///srv/shared/library/docs/{d}.pdf" for d in spec} uris = {d: f"file:///srv/shared/library/docs/{d}.pdf" for d in spec}
result = _check_duplicate_documents(docs, uris, {}, _stage2_cfg()) result = _check_duplicate_documents(docs, uris, {}, _stage2_cfg())
assert result.severity is Severity.WARN assert result.severity is Severity.WARN
# The summary message still reports the full total.
assert f"{pairs} group(s)" in result.message assert f"{pairs} group(s)" in result.message
assert sum(1 for d in result.details if d.startswith("group ")) == pairs # The terminal detail shows only the first few groups and points at export.
assert not any("more)" in d for d in result.details) assert sum(1 for d in result.details if d.startswith("group ")) == 5
assert any("more groups" in d and "--duplicates-out" in d for d in result.details)
assert any("keep #" in d for d in result.details) assert any("keep #" in d for d in result.details)
assert any("100%" in d for d in result.details)
def test_duplicate_documents_report_factors_common_path(): def test_duplicate_documents_report_factors_common_path():
@ -1051,6 +1053,35 @@ def test_duplicate_documents_report_factors_common_path():
assert not any(base in d for d in member_lines) assert not any(base in d for d in member_lines)
def test_duplicate_documents_writes_yaml(tmp_path):
# a,b identical (a 4-chunk duplicate); c distinct and excluded.
docs = _docs({"a": [0, 1, 2, 3], "b": [0, 1, 2, 3], "c": [4, 5, 6]}, dim=8)
uris = {"a": "file:///x/a.pdf", "b": "file:///x/b.pdf", "c": "file:///x/c.pdf"}
out = tmp_path / "dups.yaml"
_check_duplicate_documents(docs, uris, {}, _stage2_cfg(), yaml_path=out)
data = yaml.safe_load(out.read_text())
assert len(data["groups"]) == 1
group = data["groups"][0]
assert group["group"] == 1 and group["keep"] == "a"
docs_out = group["documents"]
assert [d["document_id"] for d in docs_out] == ["a", "b"]
assert [d["document"] for d in docs_out] == ["file:///x/a.pdf", "file:///x/b.pdf"]
assert all(d["chunks"] == 4 for d in docs_out)
assert {d["document_id"]: d["keep_suggested"] for d in docs_out} == {
"a": True,
"b": False,
}
def test_duplicate_documents_writes_empty_yaml_when_none(tmp_path):
docs = _docs({"a": [0, 1, 2], "b": [3, 4, 5]}, dim=6) # distinct
out = tmp_path / "dups.yaml"
_check_duplicate_documents(
docs, {"a": "u", "b": "v"}, {}, _stage2_cfg(), yaml_path=out
)
assert yaml.safe_load(out.read_text()) == {"groups": []}
async def _build_dup_db(path, docs: dict[str, list[int]], *, vector_dim: int = 8): async def _build_dup_db(path, docs: dict[str, list[int]], *, vector_dim: int = 8):
"""Build a multi-document database with one-hot chunk vectors.""" """Build a multi-document database with one-hot chunk vectors."""
eye = np.eye(vector_dim) eye = np.eye(vector_dim)