Merge pull request #158 from ggozad/feat/evaluations-meta

Record useful meta in evaluations experiments
This commit is contained in:
Yiorgis Gozadinos 2025-11-24 10:46:36 +02:00 committed by GitHub
commit bb3526c957
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 30 additions and 0 deletions

View file

@ -4,6 +4,17 @@
### Added
- **Database Inspector**: New `inspect` CLI command launches interactive TUI for browsing documents and chunks & searching
- **Evaluations**: Added comprehensive experiment metadata tracking for better reproducibility
- Records dataset name, test case count, and all model configurations
- Tracks embedder settings: provider, model, and vector dimensions
- Tracks QA model: provider and model name
- Tracks judge model: provider and model name for LLM evaluation
- Tracks processing parameters: `chunk_size` and `context_chunk_radius`
- Tracks retrieval configuration: `retrieval_limit` for number of chunks retrieved
- Tracks reranking configuration: `rerank_provider` and `rerank_model`
- Enables comparison of evaluation runs with different configurations in Logfire
### Changed
## [0.18.0] - 2025-11-21

View file

@ -206,11 +206,30 @@ async def run_qa_benchmark(
return await qa.answer(question)
eval_name = name if name is not None else f"{spec.key}_qa_evaluation"
experiment_metadata = {
"dataset": spec.key,
"test_cases": len(cases),
"embedder_provider": config.embeddings.provider,
"embedder_model": config.embeddings.model,
"embedder_dim": config.embeddings.vector_dim,
"qa_provider": config.qa.provider,
"qa_model": config.qa.model,
"judge_provider": "ollama",
"judge_model": QA_JUDGE_MODEL,
"chunk_size": config.processing.chunk_size,
"context_chunk_radius": config.processing.context_chunk_radius,
"retrieval_limit": 3,
"rerank_provider": config.reranking.provider,
"rerank_model": config.reranking.model,
}
report = await evaluation_dataset.evaluate(
answer_question,
name=eval_name,
max_concurrency=1,
progress=True,
metadata=experiment_metadata,
)
passing_cases = sum(