Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions reproducibility/data/manifest.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"content_hash": "bf8a69ea2a8d4191400848b0275b78b01698b0d9286c90a5e58ce4f5d13d4142",
"generated_at": "2026-06-18T02:48:37Z",
"content_hash": "3beb9828d1ab49f0cdf29807467b3706e6574068fcaf02fc2187e6b1323f1b92",
"generated_at": "2026-06-19T16:15:27Z",
"querygym_version": "0.1.0",
"row_count": 2270,
"run_count": 1150,
Expand Down
4 changes: 2 additions & 2 deletions reproducibility/data/results.csv
Original file line number Diff line number Diff line change
Expand Up @@ -904,8 +904,8 @@ schema_version,run_id,dataset_id,method_id,model,retriever_id,retriever,params_h
1,4433c120e4ab9ff0,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,03ac266c,"{""collection_path"":""msmarco/collection.tsv"",""dataset_type"":""msmarco"",""mode"":""fs"",""num_examples"":4,""train_qrels_path"":""msmarco/qrels.train.tsv"",""train_queries_path"":""msmarco/queries.train.tsv"",""train_split"":""train""}",1.0,128,recall_100,0.94,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/03ac266c.json
1,adae675b937bb6dd,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,5e04bb35,"{""mode"":""cot"",""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,0.7065,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/5e04bb35.json
1,adae675b937bb6dd,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,5e04bb35,"{""mode"":""cot"",""num_examples"":4,""train_split"":""train""}",1.0,128,recall_100,0.9433,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/5e04bb35.json
1,5e651b293bffa080,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,a5413fcf,"{""mode"":""zs"",""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,23.0,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json
1,5e651b293bffa080,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,a5413fcf,"{""mode"":""zs"",""num_examples"":4,""train_split"":""train""}",1.0,128,recall_100,0.9493,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json
1,c46db351c9adb090,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,a5413fcf,"{""mode"":""zs"",""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,0.7116,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json
1,c46db351c9adb090,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,a5413fcf,"{""mode"":""zs"",""num_examples"":4,""train_split"":""train""}",1.0,128,recall_100,0.9493,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json
1,54b3cb89c8c63f5c,beir-v1.0.0-scifact,query2e,Qwen/Qwen2.5-72B-Instruct,bge-base-en-v1.5,BGE-base-en-v1.5,eaea15f7,"{""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,0.7382,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2e/Qwen/Qwen2.5-72B-Instruct/bge-base-en-v1.5/eaea15f7.json
1,54b3cb89c8c63f5c,beir-v1.0.0-scifact,query2e,Qwen/Qwen2.5-72B-Instruct,bge-base-en-v1.5,BGE-base-en-v1.5,eaea15f7,"{""num_examples"":4,""train_split"":""train""}",1.0,128,recall_100,0.9567,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2e/Qwen/Qwen2.5-72B-Instruct/bge-base-en-v1.5/eaea15f7.json
1,c9364b60bcba9cc6,beir-v1.0.0-scifact,query2e,Qwen/Qwen2.5-72B-Instruct,bm25,BM25,bedaf076,"{""judge_rel_mode"":""positive"",""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,0.6969,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2e/Qwen/Qwen2.5-72B-Instruct/bm25/bedaf076.json
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"schema_version": 1,
"run_id": "5e651b293bffa080",
"run_id": "c46db351c9adb090",
"params_hash": "a5413fcf",
"submitted_at": "2026-05-19T00:00:00Z",
"querygym_version": "legacy-backfill",
Expand Down Expand Up @@ -44,7 +44,7 @@
}
},
"metrics": {
"ndcg_cut_10": 23.0,
"ndcg_cut_10": 0.7116,
"recall_100": 0.9493
},
"timing": {},
Expand Down
23 changes: 23 additions & 0 deletions reproducibility/lib/validate.py
Original file line number Diff line number Diff line change
Expand Up @@ -99,6 +99,7 @@ def validate(
"""
_jsonschema_validate(payload)
_validate_no_absolute_paths(payload)
_validate_metric_ranges(payload)

if not skip_registry_checks:
if dataset_registry is None:
Expand Down Expand Up @@ -172,6 +173,28 @@ def _validate_registries(
)


def _validate_metric_ranges(payload: Mapping[str, Any]) -> None:
"""Reject metric values outside [0, 1].

Every metric on the leaderboard (nDCG, recall, MAP, …) is a normalized
ranking score bounded in [0, 1]. The JSON Schema only types these as
`number`, so a corrupt value (e.g. a missing decimal point producing 23.0)
would otherwise pass schema validation and silently inflate aggregates. This
guard catches it at validate time so the aggregator and CI block any run that
carries an impossible score."""
offenders = [
f"{name}={value}"
for name, value in payload["metrics"].items()
if not (0.0 <= float(value) <= 1.0)
]
if offenders:
raise ValidationError(
f"metric value(s) outside [0, 1]: {sorted(offenders)}. "
"All ranking metrics are normalized scores; a value out of range "
"indicates a corrupt or mis-scaled result."
)


def _validate_no_absolute_paths(payload: Mapping[str, Any]) -> None:
"""Reject machine-specific absolute paths in the config — run identities
must be host-independent. build_run_summary normalizes these automatically;
Expand Down
27 changes: 27 additions & 0 deletions reproducibility/tests/test_repro_schema.py
Original file line number Diff line number Diff line change
Expand Up @@ -182,6 +182,33 @@ def test_validator_rejects_metric_outside_eval_metrics():
validate(payload)


def test_validator_rejects_metric_above_one():
"""A corrupt score (e.g. 23.0 from a dropped decimal point) must be caught —
every ranking metric is normalized to [0, 1]."""
payload = _load_fixture()
payload["metrics"]["ndcg_cut_10"] = 23.0
payload["run_id"] = compute_run_id(payload)
with pytest.raises(ValidationError, match=r"outside \[0, 1\]"):
validate(payload)


def test_validator_rejects_metric_below_zero():
payload = _load_fixture()
payload["metrics"]["ndcg_cut_10"] = -0.1
payload["run_id"] = compute_run_id(payload)
with pytest.raises(ValidationError, match=r"outside \[0, 1\]"):
validate(payload)


def test_validator_accepts_metric_at_bounds():
"""The bounds themselves are valid (a perfect or zero score)."""
payload = _load_fixture()
payload["metrics"]["ndcg_cut_10"] = 1.0
payload["metrics"]["recall_1000"] = 0.0
payload["run_id"] = compute_run_id(payload)
validate(payload) # must not raise


# ---------- Validator: hash-level rejections ---------------------------------


Expand Down
Loading