Add T²-RAGBench TAT-DQA subset; generalize subset layout

This commit is contained in:
Yiorgis Gozadinos 2026-06-04 17:18:06 +03:00
parent de9731d5f3
commit d5b0aafa2b
No known key found for this signature in database
4 changed files with 60 additions and 14 deletions

View file

@ -3,7 +3,7 @@
### Added
- `t2_finqa` evaluation dataset (T²-RAGBench FinQA subset, `G4KMU/t2-ragbench`): 2,789 single-page PDFs / 8,281 numeric QA, ingested via docling with `uri = context_id` and gold retrieval keyed on `context_id`. QA is scored with a deterministic `NumberMatchEvaluator` (relative tolerance 0.01) via the new `DatasetSpec.qa_evaluator`, bypassing the LLM judge.
- `t2_finqa` and `t2_tatdqa` evaluation datasets (T²-RAGBench subsets, `G4KMU/t2-ragbench`): financial-report PDFs ingested via docling with `uri = context_id` and gold retrieval keyed on `context_id`. QA is scored with a deterministic `NumberMatchEvaluator` (relative tolerance 0.01) via the new `DatasetSpec.qa_evaluator`, bypassing the LLM judge.
### Fixed

View file

@ -1,7 +1,7 @@
from evaluations.config import DatasetSpec
from .open_rag_bench import ORB_MULTIMODAL_SPEC, ORB_TEXT_SPEC
from .t2_ragbench import T2_FINQA_SPEC
from .t2_ragbench import T2_FINQA_SPEC, T2_TATDQA_SPEC
from .wix import WIX_SPEC
DATASETS: dict[str, DatasetSpec] = {
@ -11,6 +11,7 @@ DATASETS: dict[str, DatasetSpec] = {
ORB_TEXT_SPEC,
ORB_MULTIMODAL_SPEC,
T2_FINQA_SPEC,
T2_TATDQA_SPEC,
)
}

View file

@ -13,7 +13,23 @@ from evaluations.config import DatasetSpec, DocumentPayload, RetrievalSample
from evaluations.evaluators import MAPEvaluator, NumberMatchEvaluator
REPO_ID = "G4KMU/t2-ragbench"
SPLITS = ("dev", "test", "train")
# Per-subset layout: which metadata files hold the rows, and the repo prefix the
# PDF lives under ({split} is filled from the row).
_SUBSETS: dict[str, dict[str, Any]] = {
"FinQA": {
"metadata": tuple(
f"data/FinQA/{s}/metadata.jsonl" for s in ("dev", "test", "train")
),
"pdf_prefix": "data/FinQA/{split}/",
},
"TAT-DQA": {
"metadata": tuple(
f"data/TAT-DQA/{s}/metadata.jsonl" for s in ("dev", "test", "train")
),
"pdf_prefix": "data/TAT-DQA/{split}/",
},
}
def get_cache_dir() -> Path:
@ -31,12 +47,8 @@ def _load_rows(subset: str) -> list[dict[str, Any]]:
return cached
rows: list[dict[str, Any]] = []
for split in SPLITS:
path = hf_hub_download(
REPO_ID,
f"data/{subset}/{split}/metadata.jsonl",
repo_type="dataset",
)
for metadata_file in _SUBSETS[subset]["metadata"]:
path = hf_hub_download(REPO_ID, metadata_file, repo_type="dataset")
with open(path) as f:
for line in f:
if not line.strip():
@ -54,11 +66,8 @@ def download_t2_pdf(subset: str, split: str, file_name: str) -> Path:
if dest.exists():
return dest
src = hf_hub_download(
REPO_ID,
f"data/{subset}/{split}/{file_name}",
repo_type="dataset",
)
repo_path = _SUBSETS[subset]["pdf_prefix"].format(split=split) + file_name
src = hf_hub_download(REPO_ID, repo_path, repo_type="dataset")
shutil.copyfile(src, dest)
return dest
@ -143,3 +152,9 @@ T2_FINQA_SPEC = _t2_spec(
key="t2_finqa",
db_filename="t2_ragbench_finqa.lancedb",
)
T2_TATDQA_SPEC = _t2_spec(
subset="TAT-DQA",
key="t2_tatdqa",
db_filename="t2_ragbench_tatdqa.lancedb",
)

View file

@ -271,6 +271,36 @@ class TestT2RAGBench:
assert out.read_bytes() == b"%PDF-fake"
assert out.name == "FinQA_dev_pdf_V_2008_page_17.pdf"
def test_pdf_repo_path_per_subset(self, tmp_path: Path) -> None:
from unittest.mock import patch
blob = tmp_path / "blob"
blob.write_bytes(b"%PDF-fake")
cache = tmp_path / "cache"
cache.mkdir()
cases = [
(
"FinQA",
"dev",
"pdf/V/2008/page_17.pdf",
"data/FinQA/dev/pdf/V/2008/page_17.pdf",
),
("TAT-DQA", "dev", "raw/abc123.pdf", "data/TAT-DQA/dev/raw/abc123.pdf"),
]
for subset, split, file_name, expected_repo_path in cases:
with (
patch(
"evaluations.datasets.t2_ragbench.get_cache_dir",
return_value=cache,
),
patch(
"evaluations.datasets.t2_ragbench.hf_hub_download",
return_value=str(blob),
) as dl,
):
download_t2_pdf(subset, split, file_name)
assert dl.call_args.args[1] == expected_repo_path
def test_load_corpus_dedupes_by_context_id(self) -> None:
from unittest.mock import patch