diff --git a/CHANGELOG.md b/CHANGELOG.md index 355eabd9..de7fe3b8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,7 @@ ### Changed +- Eval judge pinned to `qwen3.8`: `DEFAULT_JUDGE_MODEL` is `ollama:qwen3.8`, and the reference configs use `Inferact/Qwen3.8-27B-NVFP4` with `extra_body.chat_template_kwargs.reasoning_effort: low`. Results in `docs/benchmarks.md` were judged by `Qwen3.6-35B-A3B-NVFP4` and are not re-judged. - `import_documents` embeds chunks across the whole batch in one pass instead of per document. ### Fixed diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 68340bb0..3a2ce679 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -70,13 +70,13 @@ evaluations run hotpotqa --config /path/to/haiku.rag.yaml --db /path/to/custom.l If no config file is specified, the script searches standard locations: `./haiku.rag.yaml`, user config directory, then falls back to defaults. -To pin the LLM judge in YAML (rather than the default `ollama:qwen3.6`). These are the recommended settings: +To pin the LLM judge in YAML (rather than the default `ollama:qwen3.8`). These are the recommended settings: ```yaml evaluations: judge: provider: openai - name: RedHatAI/Qwen3.6-35B-A3B-NVFP4 + name: Inferact/Qwen3.8-27B-NVFP4 base_url: http://localhost:8000/v1 # optional, for OpenAI-compatible servers (vLLM, LM Studio, etc.) temperature: 0.6 max_tokens: 16384 @@ -85,7 +85,7 @@ evaluations: top_k: 20 min_p: 0 chat_template_kwargs: - enable_thinking: true + reasoning_effort: low # qwen3.8: low | medium | xhigh (default) ``` ### Restricting the corpus @@ -121,11 +121,13 @@ Filtering affects searches only — a run without `--skip-db` still populates th ### QA Accuracy -`pydantic-evals` coordinates an LLM judge to determine whether the capability's answer is correct. The default judge is `ollama:qwen3.6`, pinned so changes to the capability model don't change the judge underneath. Set `evaluations.judge` in `haiku.rag.yaml` to override (including a custom `base_url` for any OpenAI-compatible endpoint). Accuracy is the fraction of correctly answered questions. +`pydantic-evals` coordinates an LLM judge to determine whether the capability's answer is correct. The default judge is `ollama:qwen3.8`, pinned so changes to the capability model don't change the judge underneath. Set `evaluations.judge` in `haiku.rag.yaml` to override (including a custom `base_url` for any OpenAI-compatible endpoint). Accuracy is the fraction of correctly answered questions. A dataset that brings its own deterministic evaluator is scored by that evaluator instead, and no judge runs. T²-RAGBench is the only such dataset today, scored by `NumberMatchEvaluator`. -We picked `qwen3.6` over the previously-pinned `gpt-oss` after a 4-cell calibration (gpt-oss / qwen3.6 as both answerer and judge, with Claude Opus 4.7 as a reference). `qwen3.6` had κ ≥ 0.66 vs the reference on both same-family and cross-family answerers (vs ~0.39–0.55 for `gpt-oss`) and showed no measurable self-preference bias, while `gpt-oss` was ~10 pp more lenient on its own outputs. +`qwen3.8` replaced `qwen3.6` after a 120-case calibration on ORB, stratified 60 pass / 60 fail: agreement 0.950, Cohen's κ 0.900, and in all 6 disagreements it matched or beat `qwen3.6` (4 were `qwen3.6` failing answers that were equivalent in different notation). It emits no reasoning content, so it avoids the thinking spirals that made `qwen3.6` exceed its output budget and drop verdicts. `reasoning_effort` changes its verdicts in 1 case per 120, so the cheaper `low` is pinned. + +Before that, we picked `qwen3.6` over the previously-pinned `gpt-oss` after a 4-cell calibration (gpt-oss / qwen3.6 as both answerer and judge, with Claude Opus 4.7 as a reference). `qwen3.6` had κ ≥ 0.66 vs the reference on both same-family and cross-family answerers (vs ~0.39–0.55 for `gpt-oss`) and showed no measurable self-preference bias, while `gpt-oss` was ~10 pp more lenient on its own outputs. ### Citation Retrieval @@ -135,7 +137,7 @@ This is computed alongside QA accuracy from the same capability run, no extra in ## Current results -Numbers measured under the current pinned judge (`ollama:qwen3.6`) on a recent `haiku.rag` version. +Numbers below were measured under `Qwen3.6-35B-A3B-NVFP4` as judge, on a recent `haiku.rag` version. The pinned judge is now `qwen3.8`; rows are not re-judged, so compare rows to each other rather than to runs judged by `qwen3.8`. ### OpenRAG Bench (ORB) diff --git a/evaluations/configs/hotpotqa.yaml b/evaluations/configs/hotpotqa.yaml index 606ae86d..84aca56d 100644 --- a/evaluations/configs/hotpotqa.yaml +++ b/evaluations/configs/hotpotqa.yaml @@ -25,8 +25,8 @@ qa: evaluations: judge: provider: openai - name: RedHatAI/Qwen3.6-35B-A3B-NVFP4 - base_url: http://vllm:11430/v1 + name: Inferact/Qwen3.8-27B-NVFP4 + base_url: http://vllm:11439/v1 temperature: 0.6 max_tokens: 16384 extra_body: @@ -34,4 +34,4 @@ evaluations: top_k: 20 min_p: 0 chat_template_kwargs: - enable_thinking: true + reasoning_effort: low diff --git a/evaluations/configs/mtrag_clapnq.yaml b/evaluations/configs/mtrag_clapnq.yaml index 7fa34d84..5dbac876 100644 --- a/evaluations/configs/mtrag_clapnq.yaml +++ b/evaluations/configs/mtrag_clapnq.yaml @@ -44,8 +44,8 @@ qa: evaluations: judge: provider: openai - name: RedHatAI/Qwen3.6-35B-A3B-NVFP4 - base_url: http://vllm:11430/v1 + name: Inferact/Qwen3.8-27B-NVFP4 + base_url: http://vllm:11439/v1 temperature: 0.6 max_tokens: 16384 extra_body: @@ -53,4 +53,4 @@ evaluations: top_k: 20 min_p: 0 chat_template_kwargs: - enable_thinking: true + reasoning_effort: low diff --git a/evaluations/configs/orb_multimodal.yaml b/evaluations/configs/orb_multimodal.yaml index 2dad063e..b868fb3c 100644 --- a/evaluations/configs/orb_multimodal.yaml +++ b/evaluations/configs/orb_multimodal.yaml @@ -29,8 +29,8 @@ qa: evaluations: judge: provider: openai - name: RedHatAI/Qwen3.6-35B-A3B-NVFP4 - base_url: http://vllm:11430/v1 + name: Inferact/Qwen3.8-27B-NVFP4 + base_url: http://vllm:11439/v1 temperature: 0.6 max_tokens: 16384 extra_body: @@ -38,4 +38,4 @@ evaluations: top_k: 20 min_p: 0 chat_template_kwargs: - enable_thinking: true + reasoning_effort: low diff --git a/evaluations/configs/orb_multimodal_nemotron.yaml b/evaluations/configs/orb_multimodal_nemotron.yaml index eca56502..99e3899b 100644 --- a/evaluations/configs/orb_multimodal_nemotron.yaml +++ b/evaluations/configs/orb_multimodal_nemotron.yaml @@ -30,8 +30,8 @@ qa: evaluations: judge: provider: openai - name: RedHatAI/Qwen3.6-35B-A3B-NVFP4 - base_url: http://vllm:11430/v1 + name: Inferact/Qwen3.8-27B-NVFP4 + base_url: http://vllm:11439/v1 temperature: 0.6 max_tokens: 16384 extra_body: @@ -39,4 +39,4 @@ evaluations: top_k: 20 min_p: 0 chat_template_kwargs: - enable_thinking: true + reasoning_effort: low diff --git a/evaluations/configs/orb_text.yaml b/evaluations/configs/orb_text.yaml index 2475bac8..18f87cb9 100644 --- a/evaluations/configs/orb_text.yaml +++ b/evaluations/configs/orb_text.yaml @@ -31,8 +31,8 @@ qa: evaluations: judge: provider: openai - name: RedHatAI/Qwen3.6-35B-A3B-NVFP4 - base_url: http://vllm:11430/v1 + name: Inferact/Qwen3.8-27B-NVFP4 + base_url: http://vllm:11439/v1 temperature: 0.6 max_tokens: 16384 extra_body: @@ -40,4 +40,4 @@ evaluations: top_k: 20 min_p: 0 chat_template_kwargs: - enable_thinking: true + reasoning_effort: low diff --git a/evaluations/evaluations/benchmark.py b/evaluations/evaluations/benchmark.py index cb9f534c..0eb34507 100644 --- a/evaluations/evaluations/benchmark.py +++ b/evaluations/evaluations/benchmark.py @@ -47,10 +47,11 @@ TARGETS: tuple[Target, ...] = ("rag-capability", "analysis-capability") # Sampling follows Qwen's recommendation for thinking mode; its model cards # forbid greedy decoding. Only the keys ollama honours are set: it silently # ignores `top_k`, `min_p` and `chat_template_kwargs`. The vLLM reference -# configs under `evaluations/configs/` carry those too. +# configs under `evaluations/configs/` carry those too, plus +# `reasoning_effort`, which qwen3.8 reads from `chat_template_kwargs`. DEFAULT_JUDGE_MODEL = ModelConfig( provider="ollama", - name="qwen3.6", + name="qwen3.8", temperature=0.6, max_tokens=16384, extra_body={"top_p": 0.95}, diff --git a/evaluations/tests/test_benchmark.py b/evaluations/tests/test_benchmark.py index 16b43013..13a4e20a 100644 --- a/evaluations/tests/test_benchmark.py +++ b/evaluations/tests/test_benchmark.py @@ -677,6 +677,7 @@ class TestRunQaBenchmarkJudgeModel: from evaluations.benchmark import DEFAULT_JUDGE_MODEL assert DEFAULT_JUDGE_MODEL.temperature == 0.6 + assert DEFAULT_JUDGE_MODEL.name == "qwen3.8" assert DEFAULT_JUDGE_MODEL.max_tokens == 16384 assert DEFAULT_JUDGE_MODEL.extra_body == {"top_p": 0.95} diff --git a/evaluations/tests/test_reference_configs.py b/evaluations/tests/test_reference_configs.py index b25f762e..142ccb34 100644 --- a/evaluations/tests/test_reference_configs.py +++ b/evaluations/tests/test_reference_configs.py @@ -15,7 +15,7 @@ PINNED_JUDGE_SAMPLING = { "top_p": 0.95, "top_k": 20, "min_p": 0, - "chat_template_kwargs": {"enable_thinking": True}, + "chat_template_kwargs": {"reasoning_effort": "low"}, }, }