Skip to content

Commit 2e692f4

Browse files
Ronald Tseronaldtse
authored andcommitted
feat(imf): WO10 — RESULTS.md -> metadata metrics generator + CI provenance gate
Every metadata metrics block is now generated, never hand-written: models/metrics-sources.yaml pins each model to a RESULTS.md table (repo, ref, path, anchor) and extraction is by table position (row label + optional column header). 'imf metrics' diffs every metadata source against its pinned table and a new CI job fails the release on drift — every number traceable to a documented protocol, including negative results where relevant. All four models' existing metadata verified against sources (khm: secryst RESULTS@23261bb; urd: rababa-urdu RESULTS@5225b17; heb: rababa RESULTS@82d508b). 6 parser specs (inline markdown, no network).
1 parent 9235d2d commit 2e692f4

5 files changed

Lines changed: 373 additions & 0 deletions

File tree

‎.github/workflows/test.yml‎

Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -68,3 +68,15 @@ jobs:
6868
pip install -e "./runtime[dev]"
6969
- name: interscript-ml runtime tests (tiny-graph zips, golden e2e)
7070
run: python -m pytest runtime/tests -v
71+
72+
metrics-provenance:
73+
runs-on: ubuntu-latest
74+
steps:
75+
- uses: actions/checkout@v4
76+
- uses: actions/setup-python@v5
77+
with: { python-version: "3.11" }
78+
- run: |
79+
python -m pip install --upgrade pip
80+
pip install -e ".[dev]"
81+
- name: WO10 — metadata metrics must match RESULTS.md sources
82+
run: PYTHONPATH=src python -m imf metrics

‎models/metrics-sources.yaml‎

Lines changed: 42 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
1+
# WO10: where each model's metadata metrics come from. Every metrics
2+
# block is GENERATED from these RESULTS.md tables and CI fails on drift.
3+
# Refs are pinned; bump them deliberately when RESULTS.md is re-published.
4+
khm-latn-1.0:
5+
repo: secryst/secryst
6+
ref: 23261bbf0d03e562d4d0fad107250349c4777368
7+
path: docs/RESULTS.md
8+
anchor: khmer-transliteration-2026-08-14
9+
protocol: "greedy decode; 895 held-out pairs; split 16,120/895/895 seed 42; ByT5-small early stop @ep15"
10+
tables:
11+
- {row: "ByT5-small, early stop @ep15", column: CER, as: cer}
12+
- {row: "ByT5-small, early stop @ep15", column: EM, as: em}
13+
urd-g2p-1.0:
14+
repo: interscript/rababa-urdu
15+
ref: 5225b17df356afb4ad23b4721a3b5af03e9f71ab
16+
path: docs/RESULTS.md
17+
anchor: g2p-urdu-text-ipa
18+
display_anchor: g2p-urdu-text--ipa
19+
protocol: "greedy decode; 12,699 held-out words from the 635K humair025 urdu-g2p dictionary; ByT5-small"
20+
tables:
21+
- {row: "CER (char-level)", as: cer}
22+
- {row: "Exact match", as: em}
23+
urd-diac-1.0:
24+
repo: interscript/rababa-urdu
25+
ref: 5225b17df356afb4ad23b4721a3b5af03e9f71ab
26+
path: docs/RESULTS.md
27+
anchor: diacritization-urdu-text-text-haraqat
28+
display_anchor: diacritization-urdu-text--text--haraqat
29+
protocol: "greedy decode; 11,940 held-out; labels derived IPA->haraqat (deterministic conversion, 597K pairs); ByT5-small, 2 epochs"
30+
tables:
31+
- {row: "CER", as: cer}
32+
heb-diac-1.0:
33+
repo: interscript/rababa
34+
ref: 82d508bd496572ddbc94a8fd7bd1aacdf7875a03
35+
path: docs/RESULTS.md
36+
anchor: hebrew-diacritization
37+
protocol: "beam=1 greedy decode (the v1 runtime path); Nakdimon test split, 5,095 examples; ByT5-base s43"
38+
tables:
39+
- {row: "beam=1 (s43/v4)", as: der_greedy}
40+
- row: "s43 (production)"
41+
as: der_beam4
42+
protocol: "beam=4 standard decode (reference quality; beam search is not in v1 runtimes); Nakdimon test split, 5,095 examples; ByT5-base s43"

‎src/imf/cli.py‎

Lines changed: 32 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -69,6 +69,32 @@ def _cmd_pack(args: argparse.Namespace) -> int:
6969
return 0
7070

7171

72+
def _cmd_metrics(args: argparse.Namespace) -> int:
73+
import sys
74+
75+
import yaml
76+
77+
from imf.metrics import check_against_metadata
78+
79+
mapping = Path(args.mapping)
80+
entries = yaml.safe_load(mapping.read_text(encoding="utf-8"))
81+
entries = entries.get("models", entries)
82+
problems: list[str] = []
83+
for model_id, spec in entries.items():
84+
metadata = Path(spec.get("metadata")) if spec.get("metadata") else (
85+
Path("models") / model_id.rsplit("-", 1)[0] / f"{model_id}.metadata.yaml"
86+
)
87+
if not metadata.is_file():
88+
problems.append(f"{model_id}: metadata source not found at {metadata}")
89+
continue
90+
problems += check_against_metadata(model_id, metadata, mapping)
91+
for problem in problems:
92+
print(f"error: {problem}", file=sys.stderr)
93+
label = "all models trace to RESULTS.md" if not problems else "MISMATCH"
94+
print(f"metrics provenance: {label} ({len(entries)} models)")
95+
return 0 if not problems else 1
96+
97+
7298
def _default_readme(metadata: ModelMetadata) -> str:
7399
return (
74100
f"# {metadata.id}\n\n"
@@ -198,6 +224,12 @@ def build_parser() -> argparse.ArgumentParser:
198224
p_golden.add_argument("--max-len", type=int, default=256)
199225
p_golden.set_defaults(func=_cmd_golden)
200226

227+
p_metrics = sub.add_parser(
228+
"metrics", help="WO10: check every metadata metrics block against its RESULTS.md source"
229+
)
230+
p_metrics.add_argument("--mapping", type=Path, default=Path("models/metrics-sources.yaml"))
231+
p_metrics.set_defaults(func=_cmd_metrics)
232+
201233
return parser
202234

203235

‎src/imf/metrics.py‎

Lines changed: 213 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,213 @@
1+
"""WO10: RESULTS.md -> metadata metrics generator.
2+
3+
Every IMF zip's metrics block is generated from a RESULTS.md table —
4+
never hand-written — and CI refuses to release a model whose metadata
5+
disagrees with the documented protocol numbers.
6+
7+
Sources are pinned in models/metrics-sources.yaml (repo, ref, path,
8+
anchor). Extraction is by table position (row label + optional column
9+
header), not regexes over prose: the mapping says where a number lives,
10+
the parser reads exactly that cell.
11+
"""
12+
13+
from __future__ import annotations
14+
15+
import re
16+
import urllib.request
17+
from dataclasses import dataclass
18+
from pathlib import Path
19+
from typing import Any
20+
21+
import yaml
22+
23+
24+
class MetricsError(ValueError):
25+
"""The RESULTS.md source cannot yield the mapped metrics."""
26+
27+
28+
@dataclass(frozen=True)
29+
class TableSpec:
30+
row: str
31+
column: str | None = None # None: first value cell in the row
32+
as_name: str = ""
33+
protocol: str | None = None # override the source-level protocol
34+
35+
36+
@dataclass(frozen=True)
37+
class SourceSpec:
38+
repo: str
39+
ref: str
40+
path: str
41+
anchor: str
42+
protocol: str
43+
tables: tuple[TableSpec, ...]
44+
display_anchor: str = ""
45+
46+
47+
def _slugify(heading: str) -> str:
48+
text = heading.strip().lower()
49+
text = re.sub(r"[^\w\s-]", "", text, flags=re.UNICODE)
50+
return re.sub(r"\s+", "-", text).strip("-")
51+
52+
53+
def _cell_to_value(cell: str) -> float:
54+
text = cell.replace("**", "").replace("%", "").replace(",", "").strip()
55+
match = re.search(r"-?\d+(?:\.\d+)?", text)
56+
if not match:
57+
raise MetricsError(f"no numeric value in cell {cell!r}")
58+
return float(match.group())
59+
60+
61+
def parse_tables(markdown: str, anchor: str) -> list[dict[str, Any]]:
62+
"""All tables in the section whose heading slugifies to `anchor`."""
63+
lines = markdown.splitlines()
64+
start = None
65+
for index, line in enumerate(lines):
66+
if line.startswith("## "):
67+
if start is not None:
68+
end = index
69+
break
70+
if _slugify(line[3:]) == anchor:
71+
start = index
72+
else:
73+
end = len(lines) if start is not None else None
74+
if start is None:
75+
raise MetricsError(f"section anchor {anchor!r} not found")
76+
77+
tables: list[dict[str, Any]] = []
78+
index = start
79+
while index < end:
80+
line = lines[index]
81+
if line.startswith("|") and index + 1 < end and set(lines[index + 1]) <= set("|-: "):
82+
header = [cell.strip() for cell in line.strip("|").split("|")]
83+
index += 2
84+
rows: list[dict[str, Any]] = []
85+
while index < end and lines[index].startswith("|"):
86+
cells = [cell.strip() for cell in lines[index].strip("|").split("|")]
87+
rows.append({"label": cells[0], "cells": cells, "header": header})
88+
index += 1
89+
tables.append({"header": header, "rows": rows})
90+
else:
91+
index += 1
92+
if not tables:
93+
raise MetricsError(f"no tables under anchor {anchor!r}")
94+
return tables
95+
96+
97+
def extract(
98+
markdown: str, anchor: str, specs: tuple[TableSpec, ...]
99+
) -> list[dict[str, Any]]:
100+
tables = parse_tables(markdown, anchor)
101+
out: list[dict[str, Any]] = []
102+
for spec in specs:
103+
found = None
104+
for table in tables:
105+
for row in table["rows"]:
106+
if spec.row.lower() in row["label"].lower():
107+
found = row
108+
break
109+
if found:
110+
break
111+
if found is None:
112+
raise MetricsError(f"row {spec.row!r} not found under {anchor!r}")
113+
if spec.column is None:
114+
values = [
115+
_cell_to_value(cell)
116+
for cell in found["cells"][1:]
117+
if "%" in cell or re.search(r"\d", cell.replace("**", ""))
118+
]
119+
if not values:
120+
raise MetricsError(f"no value cells in row {spec.row!r}")
121+
value = values[0]
122+
else:
123+
try:
124+
column_index = found["header"].index(spec.column)
125+
except ValueError as e:
126+
raise MetricsError(
127+
f"column {spec.column!r} not in table header {found['header']}"
128+
) from e
129+
value = _cell_to_value(found["cells"][column_index])
130+
out.append({"name": spec.as_name or spec.row, "value": value})
131+
return out
132+
133+
134+
def load_source(source: SourceSpec, cache_dir: Path | None = None) -> str:
135+
url = (
136+
f"https://raw.githubusercontent.com/{source.repo}/{source.ref}/{source.path}"
137+
)
138+
if cache_dir is not None:
139+
cached = cache_dir / _cache_name(source)
140+
if cached.is_file():
141+
return cached.read_text(encoding="utf-8")
142+
with urllib.request.urlopen(url) as response:
143+
text = response.read().decode("utf-8")
144+
if cache_dir is not None:
145+
cache_dir.mkdir(parents=True, exist_ok=True)
146+
(cache_dir / _cache_name(source)).write_text(text, encoding="utf-8")
147+
return text
148+
149+
150+
def _cache_name(source: SourceSpec) -> str:
151+
return f"{source.repo.replace('/', '_')}@{source.ref}_{source.path.replace('/', '_')}"
152+
153+
154+
155+
156+
def generate_metrics(
157+
model_id: str,
158+
mapping_path: Path | str,
159+
cache_dir: Path | None = None,
160+
) -> list[dict[str, Any]]:
161+
raw = yaml.safe_load(Path(mapping_path).read_text(encoding="utf-8"))
162+
entry = raw.get("models", raw).get(model_id)
163+
if entry is None:
164+
raise MetricsError(f"no metrics source mapped for {model_id!r}")
165+
source = SourceSpec(
166+
repo=entry["repo"],
167+
ref=entry["ref"],
168+
path=entry["path"],
169+
anchor=entry["anchor"],
170+
protocol=entry["protocol"],
171+
display_anchor=entry.get("display_anchor", ""),
172+
tables=tuple(
173+
TableSpec(
174+
row=t["row"],
175+
column=t.get("column"),
176+
as_name=t.get("as", ""),
177+
protocol=t.get("protocol"),
178+
)
179+
for t in entry["tables"]
180+
),
181+
)
182+
markdown = load_source(source, cache_dir)
183+
extracted = extract(markdown, source.anchor, source.tables)
184+
source_ref = f"{source.path}#{source.display_anchor or source.anchor}"
185+
return [
186+
{
187+
"name": m["name"],
188+
"value": m["value"],
189+
"protocol": spec.protocol or source.protocol,
190+
"source": source_ref,
191+
}
192+
for m, spec in zip(extracted, source.tables, strict=True)
193+
]
194+
195+
196+
def check_against_metadata(
197+
model_id: str,
198+
metadata_path: Path | str,
199+
mapping_path: Path | str,
200+
cache_dir: Path | None = None,
201+
) -> list[str]:
202+
"""Diff generated metrics vs a metadata source file. Returns problems."""
203+
generated = generate_metrics(model_id, mapping_path, cache_dir)
204+
meta = yaml.safe_load(Path(metadata_path).read_text(encoding="utf-8"))
205+
recorded = meta.get("metrics", [])
206+
problems: list[str] = []
207+
if [(m["name"], m["value"]) for m in generated] != [(m["name"], m["value"]) for m in recorded]:
208+
problems.append(
209+
f"{model_id}: metrics mismatch — generated "
210+
f"{[(m['name'], m['value']) for m in generated]} vs metadata "
211+
f"{[(m['name'], m['value']) for m in recorded]}"
212+
)
213+
return problems

‎tests/test_imf_metrics.py‎

Lines changed: 74 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,74 @@
1+
"""Tests for the WO10 metrics generator (inline markdown; no network —
2+
the network provenance check runs in CI as `imf metrics`)."""
3+
4+
from __future__ import annotations
5+
6+
import pytest
7+
8+
from imf.metrics import MetricsError, TableSpec, _slugify, extract, parse_tables
9+
10+
SAMPLE = """# Results
11+
12+
## G2P (Urdu text → IPA)
13+
14+
### Best result
15+
16+
| Metric | Value | Test set |
17+
|---|---|---|
18+
| PER (word-level) | 72.0%* | 12,699 held-out |
19+
| **CER (char-level)** | **14.77%** | 12,699 held-out |
20+
| Exact match | 33.6% | 12,699 held-out |
21+
22+
## Khmer transliteration (2026-08-14)
23+
24+
| System | EM | CER | n |
25+
|---|---|---|---|
26+
| ByT5-small, early stop @ep15 | **59.66%** | **27.42%** | 895 |
27+
28+
## Key findings
29+
30+
prose without tables
31+
"""
32+
33+
34+
def test_slugify_matches_section_headings() -> None:
35+
assert _slugify("G2P (Urdu text → IPA)") == "g2p-urdu-text-ipa"
36+
assert _slugify("Khmer transliteration (2026-08-14)") == "khmer-transliteration-2026-08-14"
37+
38+
39+
def test_parse_tables_scopes_to_anchor_section() -> None:
40+
tables = parse_tables(SAMPLE, "g2p-urdu-text-ipa")
41+
assert len(tables) == 1
42+
labels = [row["label"] for row in tables[0]["rows"]]
43+
assert "PER (word-level)" in labels
44+
45+
46+
def test_extract_row_mode_takes_first_value_cell() -> None:
47+
metrics = extract(
48+
SAMPLE,
49+
"g2p-urdu-text-ipa",
50+
(
51+
TableSpec(row="CER (char-level)", as_name="cer"),
52+
TableSpec(row="Exact match", as_name="em"),
53+
),
54+
)
55+
assert metrics == [{"name": "cer", "value": 14.77}, {"name": "em", "value": 33.6}]
56+
57+
58+
def test_extract_column_mode() -> None:
59+
metrics = extract(
60+
SAMPLE,
61+
"khmer-transliteration-2026-08-14",
62+
(TableSpec(row="ByT5-small, early stop", column="CER", as_name="cer"),),
63+
)
64+
assert metrics == [{"name": "cer", "value": 27.42}]
65+
66+
67+
def test_missing_anchor_raises() -> None:
68+
with pytest.raises(MetricsError, match="anchor"):
69+
parse_tables(SAMPLE, "no-such-section")
70+
71+
72+
def test_missing_row_raises() -> None:
73+
with pytest.raises(MetricsError, match="row"):
74+
extract(SAMPLE, "g2p-urdu-text-ipa", (TableSpec(row="Nonexistent row"),))

0 commit comments

Comments
 (0)