diff --git a/reproducibility/data/manifest.json b/reproducibility/data/manifest.json index 7eca052..5646348 100644 --- a/reproducibility/data/manifest.json +++ b/reproducibility/data/manifest.json @@ -1,6 +1,6 @@ { - "content_hash": "bf8a69ea2a8d4191400848b0275b78b01698b0d9286c90a5e58ce4f5d13d4142", - "generated_at": "2026-06-18T02:48:37Z", + "content_hash": "3beb9828d1ab49f0cdf29807467b3706e6574068fcaf02fc2187e6b1323f1b92", + "generated_at": "2026-06-19T16:15:27Z", "querygym_version": "0.1.0", "row_count": 2270, "run_count": 1150, diff --git a/reproducibility/data/results.csv b/reproducibility/data/results.csv index c422498..8583bcd 100644 --- a/reproducibility/data/results.csv +++ b/reproducibility/data/results.csv @@ -904,8 +904,8 @@ schema_version,run_id,dataset_id,method_id,model,retriever_id,retriever,params_h 1,4433c120e4ab9ff0,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,03ac266c,"{""collection_path"":""msmarco/collection.tsv"",""dataset_type"":""msmarco"",""mode"":""fs"",""num_examples"":4,""train_qrels_path"":""msmarco/qrels.train.tsv"",""train_queries_path"":""msmarco/queries.train.tsv"",""train_split"":""train""}",1.0,128,recall_100,0.94,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/03ac266c.json 1,adae675b937bb6dd,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,5e04bb35,"{""mode"":""cot"",""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,0.7065,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/5e04bb35.json 1,adae675b937bb6dd,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,5e04bb35,"{""mode"":""cot"",""num_examples"":4,""train_split"":""train""}",1.0,128,recall_100,0.9433,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/5e04bb35.json -1,5e651b293bffa080,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,a5413fcf,"{""mode"":""zs"",""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,23.0,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json -1,5e651b293bffa080,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,a5413fcf,"{""mode"":""zs"",""num_examples"":4,""train_split"":""train""}",1.0,128,recall_100,0.9493,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json +1,c46db351c9adb090,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,a5413fcf,"{""mode"":""zs"",""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,0.7116,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json +1,c46db351c9adb090,beir-v1.0.0-scifact,query2doc,openai/gpt-4.1-nano,splade-pp,SPLADE++,a5413fcf,"{""mode"":""zs"",""num_examples"":4,""train_split"":""train""}",1.0,128,recall_100,0.9493,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json 1,54b3cb89c8c63f5c,beir-v1.0.0-scifact,query2e,Qwen/Qwen2.5-72B-Instruct,bge-base-en-v1.5,BGE-base-en-v1.5,eaea15f7,"{""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,0.7382,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2e/Qwen/Qwen2.5-72B-Instruct/bge-base-en-v1.5/eaea15f7.json 1,54b3cb89c8c63f5c,beir-v1.0.0-scifact,query2e,Qwen/Qwen2.5-72B-Instruct,bge-base-en-v1.5,BGE-base-en-v1.5,eaea15f7,"{""num_examples"":4,""train_split"":""train""}",1.0,128,recall_100,0.9567,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2e/Qwen/Qwen2.5-72B-Instruct/bge-base-en-v1.5/eaea15f7.json 1,c9364b60bcba9cc6,beir-v1.0.0-scifact,query2e,Qwen/Qwen2.5-72B-Instruct,bm25,BM25,bedaf076,"{""judge_rel_mode"":""positive"",""num_examples"":4,""train_split"":""train""}",1.0,128,ndcg_cut_10,0.6969,300,0.0,legacy-backfill,reproducibility/data/runs/beir-v1.0.0-scifact/query2e/Qwen/Qwen2.5-72B-Instruct/bm25/bedaf076.json diff --git a/reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json b/reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json index c910b99..a80ff4c 100644 --- a/reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json +++ b/reproducibility/data/runs/beir-v1.0.0-scifact/query2doc/openai/gpt-4.1-nano/splade-pp/a5413fcf.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "run_id": "5e651b293bffa080", + "run_id": "c46db351c9adb090", "params_hash": "a5413fcf", "submitted_at": "2026-05-19T00:00:00Z", "querygym_version": "legacy-backfill", @@ -44,7 +44,7 @@ } }, "metrics": { - "ndcg_cut_10": 23.0, + "ndcg_cut_10": 0.7116, "recall_100": 0.9493 }, "timing": {}, diff --git a/reproducibility/lib/validate.py b/reproducibility/lib/validate.py index e7e7482..a3a7505 100644 --- a/reproducibility/lib/validate.py +++ b/reproducibility/lib/validate.py @@ -99,6 +99,7 @@ def validate( """ _jsonschema_validate(payload) _validate_no_absolute_paths(payload) + _validate_metric_ranges(payload) if not skip_registry_checks: if dataset_registry is None: @@ -172,6 +173,28 @@ def _validate_registries( ) +def _validate_metric_ranges(payload: Mapping[str, Any]) -> None: + """Reject metric values outside [0, 1]. + + Every metric on the leaderboard (nDCG, recall, MAP, …) is a normalized + ranking score bounded in [0, 1]. The JSON Schema only types these as + `number`, so a corrupt value (e.g. a missing decimal point producing 23.0) + would otherwise pass schema validation and silently inflate aggregates. This + guard catches it at validate time so the aggregator and CI block any run that + carries an impossible score.""" + offenders = [ + f"{name}={value}" + for name, value in payload["metrics"].items() + if not (0.0 <= float(value) <= 1.0) + ] + if offenders: + raise ValidationError( + f"metric value(s) outside [0, 1]: {sorted(offenders)}. " + "All ranking metrics are normalized scores; a value out of range " + "indicates a corrupt or mis-scaled result." + ) + + def _validate_no_absolute_paths(payload: Mapping[str, Any]) -> None: """Reject machine-specific absolute paths in the config — run identities must be host-independent. build_run_summary normalizes these automatically; diff --git a/reproducibility/tests/test_repro_schema.py b/reproducibility/tests/test_repro_schema.py index 65a64ee..0fc42cc 100644 --- a/reproducibility/tests/test_repro_schema.py +++ b/reproducibility/tests/test_repro_schema.py @@ -182,6 +182,33 @@ def test_validator_rejects_metric_outside_eval_metrics(): validate(payload) +def test_validator_rejects_metric_above_one(): + """A corrupt score (e.g. 23.0 from a dropped decimal point) must be caught — + every ranking metric is normalized to [0, 1].""" + payload = _load_fixture() + payload["metrics"]["ndcg_cut_10"] = 23.0 + payload["run_id"] = compute_run_id(payload) + with pytest.raises(ValidationError, match=r"outside \[0, 1\]"): + validate(payload) + + +def test_validator_rejects_metric_below_zero(): + payload = _load_fixture() + payload["metrics"]["ndcg_cut_10"] = -0.1 + payload["run_id"] = compute_run_id(payload) + with pytest.raises(ValidationError, match=r"outside \[0, 1\]"): + validate(payload) + + +def test_validator_accepts_metric_at_bounds(): + """The bounds themselves are valid (a perfect or zero score).""" + payload = _load_fixture() + payload["metrics"]["ndcg_cut_10"] = 1.0 + payload["metrics"]["recall_1000"] = 0.0 + payload["run_id"] = compute_run_id(payload) + validate(payload) # must not raise + + # ---------- Validator: hash-level rejections ---------------------------------