Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
48 changes: 48 additions & 0 deletions artifacts/tech_report_v1/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,48 @@
# Regenerating the tech report

Run from the repository root:

```bash
uv sync --locked --extra dev --extra analysis
uv run --locked python -m artifacts.tech_report_v1.regenerate \
--submission-directory logs/self_tuning \
--output-root scoring_results_tech_report
```

Supply the complete submission pool and a new, empty output directory. The
runner executes the leaderboard, Muon, and framework notebooks in order,
including the 0–20% target-relaxation sweep. Outputs include figures, tables,
CSVs, executed notebooks, and a `manifest.json` with input/source hashes and
completion status. Each section records its actual settings in
`out/run_metadata.json`. Failed runs must not be used as results.

The default is non-strict self-tuning scoring over nine workloads. Pool and
workload changes affect scores. Configure filters and scoring settings in the
paired scripts; add submission labels in `report_utils.py`. The Muon, selected
sweep, and framework views require their configured submissions. Recheck the
framework view's batch-size assumptions when submission recipes change.

The Muon figure loads `muon_torch_replicated_torch_hps` from the supplied log
directory as a matched vanilla comparison. It keeps the standard leaderboard's
non-Muon reference pool and records available study counts in its audit CSV.

Sections share loading/scoring in `report_data.py`, table rendering in
`report_tables.py`, and labels, styles, paths, and PDF/PNG export in
`report_utils.py`. Notebook and command-line plots use the same renderers.

For interactive use, select the repository's `.venv` kernel. Run Section 4
before Muon; the framework notebook runs independently. After editing either
member of a notebook/script pair, synchronize it before running:

```bash
uv run jupytext --sync artifacts/tech_report_v1/section_4_leaderboards/score_submissions.py
uv run jupytext --sync artifacts/tech_report_v1/section_6_algorithms/muon_workloads.py
uv run jupytext --sync artifacts/tech_report_v1/section_7_frameworks/framework_comparison.py
```

Source notebooks have no saved outputs. Review executed notebooks and artifacts
in the fresh output directory; synchronization alone does not refresh results.
Section-local `out/` artifacts can be checked in; refresh them from a completed
run before committing generated results.
Paper export is separate, and paper tables remain hand-maintained. Appendix
curves use their [own workflow](appendix_training_curves/curve_plotting/README.md).
194 changes: 194 additions & 0 deletions artifacts/tech_report_v1/regenerate.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,194 @@
"""Execute the current report notebooks in dependency order from fresh logs."""

import argparse
import hashlib
import importlib.metadata
import json
import os
from pathlib import Path
import platform
import subprocess
import sys

import jupytext
import nbformat
from jupyter_client import KernelManager
from nbclient import NotebookClient

from artifacts.tech_report_v1.report_utils import REPO_ROOT, REPORT_ROOT

NOTEBOOKS = (
('section_4_leaderboards', 'score_submissions'),
('section_6_algorithms', 'muon_workloads'),
('section_7_frameworks', 'framework_comparison'),
)


def fingerprints(paths, root):
"""Hash contents, including uncommitted changes and newly supplied logs."""
return {
str(path.relative_to(root)): hashlib.sha256(path.read_bytes()).hexdigest()
for path in sorted(paths)
}


def read_paired_notebook(script):
"""Fail on unsynchronized sources instead of trusting file timestamps."""
notebook = jupytext.read(script)
paired = jupytext.read(script.with_suffix('.ipynb'))

def cells(nb):
return [(c.cell_type, c.source) for c in nb.cells]

if cells(notebook) != cells(paired):
raise ValueError(
f'Notebook pair differs: {script}. Reconcile edits, then synchronize '
'with jupytext before regenerating.'
)
return notebook


def execute_notebook(notebook, destination, env, timeout):
# Capture actual settings from the executed namespace, not duplicated defaults.
notebook.cells.append(
nbformat.v4.new_code_cell(r"""
import dataclasses, json
from pathlib import Path
_keys = (
'SUBMISSION_DIRECTORY', 'INCLUDE_SUBMISSIONS', 'EXCLUDE_SUBMISSIONS',
'STRICT', 'SELF_TUNING_RULESET', 'MIN_TAU', 'MAX_TAU', 'NUM_POINTS',
'SCALE', 'TARGET_RELAXATION_FRACTION', 'SELECTED_SUBMISSIONS',
'LOAD_RESULTS_FROM', 'SAVE_RESULTS_TO',
)
_metadata = {key: globals()[key] for key in _keys if key in globals()}
if _metadata.get('LOAD_RESULTS_FROM'):
raise ValueError('Regeneration requires fresh logs; disable LOAD_RESULTS_FROM.')
if 'WORKLOAD_CONFIG' in globals():
_metadata['workload_config'] = dataclasses.asdict(WORKLOAD_CONFIG)
if 'results' in globals():
_metadata['submission_pool'] = sorted(results)
if 'SWEEP_PERCENTAGES' in globals():
_metadata['sweep_percentages'] = list(SWEEP_PERCENTAGES)
Path(OUTPUT_DIR, 'run_metadata.json').write_text(
json.dumps(_metadata, indent=2) + '\n'
)
""")
)
manager = KernelManager(kernel_name='python3')
# Use the invoking uv environment even if another python3 kernel is installed.
manager.kernel_spec.argv[0] = sys.executable
client = NotebookClient(notebook, km=manager, timeout=timeout)
try:
client.execute(cwd=str(REPO_ROOT), env=env)
finally:
notebook.cells.pop()
destination.parent.mkdir(parents=True, exist_ok=True)
nbformat.write(notebook, destination)


def regenerate(submission_directory, output_root, *, timeout=1800):
submission_directory = submission_directory.resolve()
output_root = output_root.resolve()
if not submission_directory.is_dir():
raise ValueError(f'Input directory does not exist: {submission_directory}')
if output_root.exists() and any(output_root.iterdir()):
raise ValueError(
f'Output directory must be empty: {output_root}. '
'Choose a new directory to avoid mixing results from different inputs.'
)
notebooks = [
(section, name, read_paired_notebook(REPORT_ROOT / section / f'{name}.py'))
for section, name in NOTEBOOKS
]

def inputs():
return fingerprints(
submission_directory.rglob('eval_measurements.csv'), submission_directory
)

def sources():
return fingerprints(
[
*REPORT_ROOT.rglob('*.py'),
*REPO_ROOT.joinpath('scoring').glob('*.py'),
*REPO_ROOT.joinpath('scoring').glob('workload_targets*.json'),
REPO_ROOT / 'pyproject.toml',
REPO_ROOT / 'uv.lock',
],
REPO_ROOT,
)

manifest = {
'status': 'running',
'git_revision': subprocess.check_output(
['git', 'rev-parse', 'HEAD'], cwd=REPO_ROOT, text=True
).strip(),
'python': platform.python_version(),
'packages': {
name: importlib.metadata.version(name)
for name in ('numpy', 'pandas', 'matplotlib', 'jupytext', 'nbclient')
},
'submission_directory': str(submission_directory),
'input_sha256': inputs(),
'source_sha256': sources(),
}
if not manifest['input_sha256']:
raise ValueError(
'No eval_measurements.csv files found in the input directory.'
)
output_root.mkdir(parents=True, exist_ok=True)
manifest_path = output_root / 'manifest.json'

def save_manifest():
manifest_path.write_text(json.dumps(manifest, indent=2) + '\n')

save_manifest()
env = {
**os.environ,
'ALGOPERF_REPORT_INPUT': str(submission_directory),
'ALGOPERF_REPORT_OUTPUT': str(output_root),
'MPLBACKEND': 'module://matplotlib_inline.backend_inline',
}
try:
for section, name, notebook in notebooks:
print(f'Executing {section}/{name} ...', flush=True)
execute_notebook(
notebook, output_root / section / f'{name}.ipynb', env, timeout
)
if (
inputs() != manifest['input_sha256']
or sources() != manifest['source_sha256']
):
raise RuntimeError('Inputs or source changed during regeneration; rerun.')
manifest['outputs_sha256'] = fingerprints(
(p for p in output_root.rglob('*') if p.is_file() and p != manifest_path),
output_root,
)
manifest['status'] = 'complete'
except Exception as error:
manifest.update(status='failed', error=str(error))
raise
finally:
save_manifest()
print(f'Completed. Results and executed notebooks: {output_root}', flush=True)


def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
'--submission-directory', type=Path, default=REPO_ROOT / 'logs/self_tuning'
)
parser.add_argument(
'--output-root',
type=Path,
default=REPO_ROOT / 'scoring_results_tech_report',
)
parser.add_argument(
'--timeout', type=int, default=1800, help='Seconds per cell.'
)
args = parser.parse_args()
regenerate(args.submission_directory, args.output_root, timeout=args.timeout)


if __name__ == '__main__':
main()
121 changes: 121 additions & 0 deletions artifacts/tech_report_v1/report_data.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,121 @@
"""Shared log loading, summaries, and canonical scoring for report sections."""

from pathlib import Path
import pickle

import pandas as pd

from artifacts.tech_report_v1.report_utils import DISPLAY_TO_RAW, pretty


DEFAULT_MIN_TAU = 1.0
DEFAULT_MAX_TAU = 4.0
DEFAULT_PROFILE_POINTS = 100


def load_runs(
directory, *, include=(), exclude=(), cache=None, save_cache=None
):
"""Load a named comparison pool from logs or an explicitly supplied cache."""
from scoring import scoring_utils

def names(value):
return (
{s.strip() for s in value.split(',') if s.strip()}
if isinstance(value, str)
else set(value)
)

include, exclude = names(include), names(exclude)
directory = Path(directory)
if cache:
with Path(cache).open('rb') as handle:
available = {
DISPLAY_TO_RAW.get(name, name): frame
for name, frame in pickle.load(handle).items()
}
else:
available = {p.name: None for p in directory.iterdir() if p.is_dir()}
missing = include - available.keys()
if missing:
raise ValueError(f'Requested submissions are missing: {sorted(missing)}')
selected = sorted((include or available.keys()) - exclude)
if not selected:
raise ValueError('No submissions selected for scoring.')
runs = {}
for raw in selected:
name = pretty(raw)
if name in runs:
raise ValueError(f'Duplicate submission display name: {name}')
print(f'Loading {name} ({raw})', flush=True)
frame = (
available[raw]
if cache
else scoring_utils.get_experiment_df(str(directory / raw))
)
if frame.empty:
raise ValueError(f'No evaluation measurements found for {raw}.')
runs[name] = frame
if save_cache:
path = Path(save_cache)
path.parent.mkdir(parents=True, exist_ok=True)
with path.open('wb') as handle:
pickle.dump(runs, handle)
return runs


def write_summaries(runs, config, directory):
"""Refresh study summaries even when the parsed runs came from a cache."""
from scoring.score_submissions import get_submission_summary

directory = Path(directory)
directory.mkdir(parents=True, exist_ok=True)
summaries = {}
for name, frame in runs.items():
summary = get_submission_summary(frame, config)
raw = DISPLAY_TO_RAW.get(name, name)
summary.to_csv(directory / f'{raw}_summary.csv')
summaries[name] = summary
return summaries


def score_runs(
runs,
config,
output_dir,
*,
time_col='score',
artifact_suffix='',
min_tau=DEFAULT_MIN_TAU,
max_tau=DEFAULT_MAX_TAU,
num_points=DEFAULT_PROFILE_POINTS,
scale='linear',
strict=False,
self_tuning_ruleset=True,
):
"""Return profiles, normalized scores, and target times for the same pool."""
from scoring import performance_profile

output_dir = Path(output_dir)
output_dir.mkdir(parents=True, exist_ok=True)
profiles = performance_profile.compute_performance_profiles(
runs,
config,
time_col=time_col,
min_tau=min_tau,
max_tau=max_tau,
num_points=num_points,
scale=scale,
strict=strict,
self_tuning_ruleset=self_tuning_ruleset,
verbosity=0,
output_dir=str(output_dir),
artifact_suffix=artifact_suffix,
)
scores = performance_profile.compute_leaderboard_score(
profiles, normalize=True
)
times = pd.read_csv(
output_dir / f'time_to_targets{artifact_suffix}.csv', index_col=0
)
return profiles, scores, times.loc[scores.index]
Loading
Loading