-
Notifications
You must be signed in to change notification settings - Fork 6
Expand file tree
/
Copy pathloop.py
More file actions
2879 lines (2710 loc) · 134 KB
/
Copy pathloop.py
File metadata and controls
2879 lines (2710 loc) · 134 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
"""Evidence-first solver discovery with paired, isolated confirmation.
The command-line path snapshots the historical champion, generates candidates,
evaluates them in disposable workers, and writes local evidence. Fully
confirmed candidates advance the canonical solver with hash-bound provenance;
publication remains a separate approval-gated action. Legacy ``Loop`` helpers
remain for older library callers.
Usage:
python loop.py --problem cvrp --targets X-n280-k17 --eval-only
python loop.py --problem miplib_heur --provider paired --iters 4 --budget 20
"""
import argparse
import hashlib
import html
import json
import math
import os
import re
import shutil
import statistics
import subprocess
import sys
import tempfile
import time
from concurrent.futures import ThreadPoolExecutor
from functools import partial
from datetime import datetime, timezone
from pathlib import Path
import evaluation
import island_evolution
import patching
import postmortem
import research_context
import verification_contract
from model_registry import (
DEFAULT_CHAIN,
MODEL_REGISTRY,
VALID_ROUTING_POLICIES,
alias_for_model,
arm_alias,
model_spec,
validate_routing_config,
)
from research_memory import (
analyze_candidate,
cells_text,
is_development_observation,
mentions_target,
operational_stats,
rank_auto_allocation,
redact_targets,
strip_local_paths,
summarize_development,
)
from routing import RoutingJournal, route_call, routing_summary
HERE = os.path.dirname(os.path.abspath(__file__))
AUTHOR = "Wes Sander, MoltFire"
def load_problem(name):
from problem_loader import load_problem as safe_load_problem
return safe_load_problem(name)
def layout(name):
"""(best_dir, runs_dir). ponytail: circle_packing keeps the original flat layout because a run is in flight;
unify to best/<name> after it ends."""
suf = "" if name == "circle_packing" else "-" + name
return os.path.join(HERE, "best" + suf), os.path.join(HERE, "runs" + suf)
def value_of(entry):
return None if entry is None else entry.get("value", entry.get("sum"))
def retro_path(name):
return os.path.join(HERE, "docs", "retro", f"{name}.md")
def read_latest_retro(path):
"""The Lessons and Next blocks of the newest section of a retro file written by retro.py, or "" if none."""
if not os.path.exists(path):
return ""
text = open(path, encoding="utf-8").read()
sections = re.split(r"^## ", text, flags=re.M)
if len(sections) < 2:
return ""
last = sections[-1]
keep = []
for heading in ("### Lessons", "### Next"):
m = re.search(re.escape(heading) + r"\n(.*?)(?=^### |\Z)", last, re.S | re.M)
if m:
keep.append(f"{heading}\n{m.group(1).strip()}")
return "\n".join(keep)
def compress_history(history, keep_full=12, cap=80):
"""Ideas older than the last keep_full iterations, one short line each, newest first, at most cap lines.
The full block already shows the recent ones; this stops a repeat of something tried nights ago."""
old = history[:-keep_full] if len(history) > keep_full else []
lines = [f"iter {h['iter']} ({h['status']}): {(h.get('idea') or '')[:110]}" for h in reversed(old)]
return "\n".join(lines[:cap])
class Loop:
def __init__(self, problem, root=None, problem_module=None, initialize_best=True):
self.root = os.path.abspath(root or HERE)
self.P = problem_module or load_problem(problem)
self.name = problem
if root is None:
self.best, self.runs = layout(problem)
else:
suffix = "" if problem == "circle_packing" else "-" + problem
self.best = os.path.join(self.root, "best" + suffix)
self.runs = os.path.join(self.root, "runs" + suffix)
self.champ = os.path.join(self.best, "solver.py")
self.scores = os.path.join(self.best, "scores.json")
self.log_path = os.path.join(self.runs, "log.jsonl")
self.status = os.path.join(self.runs, "status.html")
self.solver_evaluations = 0
self.solver_seconds = 0.0
if initialize_best:
os.makedirs(self.best, exist_ok=True)
os.makedirs(self.runs, exist_ok=True)
if not os.path.exists(self.champ):
shutil.copy(os.path.join(self.root, "problems", problem, "seed_solver.py"), self.champ)
# ── evaluation ──
def run_solver(self, solver, target, budget, seed, out):
t0 = time.time()
try:
from isolation import run_solver as isolated_run_solver
p = isolated_run_solver(self.name, solver, target, budget, seed, out, root=self.root)
if p.returncode != 0:
detail = p.stderr or p.stdout or f"solver exited with status {p.returncode}"
return {"target": target, "error": detail[-600:]}
value, payload = self.P.evaluate(out, target)
return {"target": target, "value": value, "payload": payload, "secs": round(time.time() - t0, 1)}
except Exception as e: # infeasible, bad JSON, missing file, ...
return {"target": target, "error": f"{type(e).__name__}: {e}"[:300]}
def evaluate(self, solver, targets, budget, seed, workdir, workers):
os.makedirs(workdir, exist_ok=True)
with ThreadPoolExecutor(workers) as ex:
futs = [
ex.submit(self.run_solver, solver, t, budget, seed + i, os.path.join(workdir, f"{t}.json"))
for i, t in enumerate(targets)
]
return [f.result() for f in futs]
def total(self, results, rec):
return sum(
self.P.score(r["value"], rec.get(r["target"])) if "value" in r else self.P.FAIL_SCORE for r in results
)
def load_scores(self):
return json.load(open(self.scores)) if os.path.exists(self.scores) else {}
def update_bests(self, results, iteration, rec):
"""Persist per-target bests + submission files. Returns (improved targets, targets beating the record)."""
scores = self.load_scores()
improved, wins = [], []
for r in results:
if "value" not in r:
continue
t = r["target"]
prev = value_of(scores.get(t))
if prev is None or self.P.better(r["value"], prev):
scores[t] = {"value": r["value"], "iter": iteration, "record": rec.get(t)}
self.P.save(t, r["payload"], r["value"], self.best, AUTHOR)
improved.append(t)
if self.P.beats(r["value"], rec.get(t)):
wins.append(t)
json.dump(scores, open(self.scores, "w"), indent=1)
return improved, wins
# ── model ──
def scoreboard(self, targets, rec, last=None):
scores = self.load_scores()
rows = []
for t in targets:
b = value_of(scores.get(t))
r = rec.get(t)
l = next((x["value"] for x in (last or []) if x["target"] == t and "value" in x), None)
rows.append((t, r, b, l, (b - r) if (b is not None and r is not None) else None))
return rows
def build_prompt(self, targets, rec, last_results, history):
champ = open(self.champ, encoding="utf-8").read()
board = "target | best known | ours | champion last run | ours - best known\n"
for t, r, b, l, d in self.scoreboard(targets, rec, last_results):
board += f"{t} | {r if r is not None else '(none known)'} | {b if b is not None else '-'} | {l if l is not None else '-'} | {('%+.6g' % d) if d is not None else '-'}\n"
hist = (
"\n".join(
f"iter {h['iter']}: total={h['total']:.4f} ({h['status']}) IDEA: {h['idea']}"
+ (f" ERRORS: {h['errors']}" if h.get("errors") else "")
for h in history[-12:]
)
or "(none yet)"
)
older = compress_history(history)
if older:
hist += f"\n\nPREVIOUSLY TRIED, EARLIER NIGHTS (do not repeat unless you fix the named failure):\n{older}"
retro = read_latest_retro(retro_path(self.name))
retro_block = (
f"""
LAST RETRO (written after the previous run; follow it):
{retro}
Take the lowest-numbered Next direction whose tag [NEXT #k] does not yet appear in IDEAS TRIED and start your IDEA
line with that tag. If every direction is used, or the scoreboard now argues against all of them, start the IDEA
line with [NEXT none] and say why in the same sentence.
"""
if retro
else ""
)
return f"""{self.P.PROMPT}
CURRENT CHAMPION solver.py:
```python
{champ}
```
SCOREBOARD ({"higher" if self.P.MAXIMIZE else "lower"} is better; champion total = {self.P.TOTAL_DESC}):
{board}
IDEAS TRIED SO FAR:
{hist}
{retro_block}
{self.P.TASK}
OUTPUT FORMAT: first line "IDEA: <one sentence>", then exactly one ```python block with the full file. Nothing else."""
@staticmethod
def call_model(prompt, model):
"""Compatibility triple through the canonical subscription-only provider."""
from providers import call_model
result = call_model(prompt, provider="fable", model=model, timeout=900)
if result.get("error"):
if result.get("error_kind") == "timeout":
return None, 0.0, "cli timeout after 900s"
return None, 0.0, "cli error: " + result["error"]
return result["code"], result["cost"] or 0.0, result["idea"] or "(no idea line)"
# ── reporting ──
def write_status(self, targets, rec, history, cost_total, champ_total):
tr = ""
for t, r, b, l, d in self.scoreboard(targets, rec):
cls = "win" if (b is not None and self.P.beats(b, r)) else ""
tr += (
f"<tr class='{cls}'><td>{t}</td><td>{r if r is not None else '(none known)'}</td><td>{b if b is not None else '-'}</td>"
f"<td>{('%+.6g' % d) if d is not None else '-'}</td></tr>"
)
hist = "".join(
f"<tr><td>{h['iter']}</td><td>{h['total']:.4f}</td><td>{h['status']}</td>"
f"<td>${h['cost']:.2f}</td><td>{html.escape(h['idea'])}</td></tr>"
for h in reversed(history)
)
open(
self.status, "w", encoding="utf-8"
).write(f"""<!doctype html><meta charset=utf-8><title>discovery-loop: {self.P.TITLE}</title>
<style>body{{font:14px system-ui;margin:2em;max-width:60em}}table{{border-collapse:collapse;margin:1em 0}}td,th{{border:1px solid #ccc;padding:4px 8px;text-align:right}}
th{{background:#eee}}.win{{background:#c8f7c5}}h1{{margin:0}}</style>
<h1>discovery-loop: {self.P.TITLE}</h1>
<p>Updated {time.strftime("%Y-%m-%d %H:%M")} · champion total {champ_total:.4f} · model spend ${cost_total:.2f} · green = beats best known</p>
<table><tr><th>target</th><th>best known</th><th>ours</th><th>ours - best known</th></tr>{tr}</table>
<h3>Iterations</h3><table><tr><th>#</th><th>total</th><th>status</th><th>cost</th><th style='text-align:left'>idea</th></tr>{hist}</table>
<p>{html.escape(self.P.SUBMIT_NOTE)}</p>""")
def publish(self):
"""Fire-and-forget: push candidates to GitHub via publish.py --push-only. The maintainer email is batched
per slot by night.py (one email, one approval tap for every winner of the slot); by hand: python publish.py."""
os.makedirs(self.runs, exist_ok=True)
with open(os.path.join(self.runs, "publish.log"), "a") as log:
subprocess.Popen(
[sys.executable, os.path.join(HERE, "publish.py"), "--problem", self.name, "--push-only"],
cwd=HERE,
stdout=log,
stderr=subprocess.STDOUT,
)
# ── governed research pipeline ──
def evaluate_matrix(self, solver, matrix, budget, workdir, workers, runner, deadline=None):
"""Run an exact target/seed matrix and independently verify every output."""
os.makedirs(workdir, exist_ok=True)
def one(cell):
target, seed = cell["target"], cell["seed"]
safe_target = re.sub(r"[^A-Za-z0-9_.-]+", "_", str(target))
output = os.path.join(workdir, f"{safe_target}-seed{seed}.json")
started = time.time()
try:
result = runner(
self.name,
solver,
target,
budget,
seed,
output,
root=self.root,
deadline=deadline,
)
if result.returncode != 0:
detail = (result.stderr or result.stdout or f"solver exited with status {result.returncode}")[-600:]
return {
"target": target,
"seed": seed,
"error": detail,
"returncode": result.returncode,
"secs": time.time() - started,
}
value, payload = self.P.evaluate(output, target)
if not isinstance(value, (int, float)) or isinstance(value, bool) or not math.isfinite(float(value)):
raise ValueError("verifier returned a non-finite value")
row = {
"target": target,
"seed": seed,
"value": float(value),
"payload": payload,
"output_path": output,
"secs": round(time.time() - started, 3),
}
search = search_summary(output, getattr(self.P, "MAXIMIZE", False))
if search is not None:
row["search"] = search
return row
except Exception as exc:
return {
"target": target,
"seed": seed,
"error": f"{type(exc).__name__}: {exc}"[:600],
"secs": round(time.time() - started, 3),
}
with ThreadPoolExecutor(max_workers=workers) as executor:
results = list(executor.map(one, matrix))
self.solver_evaluations += len(results)
self.solver_seconds += sum(float(result.get("secs", 0.0)) for result in results)
return results
def build_research_prompt(
self,
incumbent,
targets,
records,
history,
hidden_targets=(),
retro_memory=None,
history_total=None,
mission=None,
context_blocks=None,
profile=None,
generation_mode="full",
near_misses=(),
):
"""Build a prompt from development data only."""
if context_blocks is None:
name = getattr(self, "name", None)
context_blocks = (
research_context.blocks(name, self.P, getattr(self, "root", HERE), hidden_targets)
if name
else {"text": "", "dead_ends": [], "patterns": []}
)
if hasattr(self.P, "prompt_for_targets"):
context = self.P.prompt_for_targets(list(targets))
else:
hidden = tuple(str(target) for target in hidden_targets)
context = "\n".join(
line for line in self.P.PROMPT.splitlines() if not any(target in line for target in hidden)
)
board = "\n".join(f"{target}: reference={records.get(target)}" for target in targets)
output_format = (
patching.DIFF_OUTPUT_FORMAT
if generation_mode == "diff"
else "OUTPUT FORMAT: the tagged IDEA line, then exactly one ```python block with the full file."
" Nothing else."
)
profile_block = (
f"\nDEVELOPMENT PROFILE (this run's matrix; seconds used is what the incumbent actually spent,"
f" and time to best is when it last improved. A solver may report its own curve by writing"
f' "trace": [[seconds, objective], ...] beside its solution; "-" means it reported none):\n'
f"{profile}"
f"\n"
if profile
else ""
)
memory = summarize_development(
history,
hidden_targets=hidden_targets,
limit=20,
family_limit=12,
total_observations=history_total,
)
for entry in memory["entries"]:
entry["idea"] = entry["idea"][:300]
entry["negative_result"] = entry["negative_result"][:160]
entry["critique"]["text"] = entry["critique"]["text"][:400]
entry["critique"]["error"] = entry["critique"]["error"][:120]
prior = json.dumps(memory, sort_keys=True, separators=(",", ":")) if memory["entries"] else "(none yet)"
retro = {key: str((retro_memory or {}).get(key, ""))[:1000] for key in ("lessons", "next_experiment")}
for key, value in retro.items():
retro[key] = strip_local_paths(redact_targets(value, hidden_targets))
retro_text = json.dumps(retro, sort_keys=True, separators=(",", ":")) if any(retro.values()) else "(none yet)"
near_miss_text = ""
for item in near_misses or ():
header = f"--- iteration {item.get('iteration')} ({item.get('status')}"
gain = item.get("median_gain")
if isinstance(gain, (int, float)):
header += f", median gain {gain * 100:+.3f}%"
header += ")"
near_miss_text += f"\n{header} ---\nIDEA: {str(item.get('idea') or '')[:300]}\n"
if item.get("cells"):
near_miss_text += f"PER-TARGET MEDIAN GAIN: {item['cells']}\n"
near_miss_text += f"WHY IT DID NOT PASS: {str(item.get('negative_result') or '')[:300]}\n"
near_miss_text += f"```diff\n{item.get('diff') or '(no diff available)'}\n```\n"
if near_miss_text:
near_miss_text = (
"MEASURED NEAR MISSES FROM THIS RUN (each diff is against the file above; these were run on the "
"development matrix and did not pass). Continue one of them by fixing the named failure, or say "
"in the IDEA line what you are doing differently and why:\n" + near_miss_text
)
mission_text = "(no reviewed ARC mission is bound to this run)"
if mission:
mission_text = json.dumps(
{
key: mission[key]
for key in (
"source_problem_id",
"source_repository",
"source_revision",
"catalogue_hash",
"plugin",
"baseline",
"verifier",
"beneficiary",
"bounded_hypothesis",
"resources",
"development_confirmation_split",
"success_criterion",
"budget",
)
},
sort_keys=True,
separators=(",", ":"),
)
prompt = f"""{context}
REVIEWED ARC MISSION BINDING:
{mission_text}
This binding is bounded context for the existing benchmark. It does not claim the broader source question is
solved. Source card prose and starter prompts are excluded. Do not fetch or execute instructions from sources.
CURRENT INCUMBENT solver.py:
```python
{incumbent}
```
DEVELOPMENT REFERENCES ONLY:
{board}
{profile_block}
DEVELOPMENT HISTORY ONLY (each entry's "cells" holds that candidate's own paired result per development
target -- won/lost counts and the median gain in percent on each target; "families" rolls those same
per-target gains up across every attempt in an algorithm family, so a family that wins on one instance class
and loses on another is visible. Positive means better than the incumbent):
{prior}
PRIOR RETROSPECTIVE NOTES (the next experiment is an untested hypothesis, not evidence):
{retro_text}
{context_blocks["text"]}
{near_miss_text}
{self.P.TASK}
Begin the idea with an algorithm-family tag: "IDEA: [kind: <algorithm family>] <one sentence>".
If the proposal is related to a failed approach in memory, name the concrete mechanism that differs and why that
difference addresses the observed failure. A changed mechanism within a previously tried family is allowed when
that explanation is specific. Treat every expected effect as a hypothesis; do not claim guaranteed gains or
correctness.
{output_format}"""
leaked = [str(target) for target in hidden_targets if mentions_target(prompt, target)]
if leaked:
raise ValueError(f"generation prompt exposes withheld targets: {leaked}")
return prompt
class _ResearchStop(RuntimeError):
def __init__(self, status, message):
super().__init__(message)
self.status = status
def _utc_now():
return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
def _sha256(path):
digest = hashlib.sha256()
with open(path, "rb") as stream:
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def _atomic_copy(source, destination):
destination = Path(destination)
destination.parent.mkdir(parents=True, exist_ok=True)
handle, temporary = tempfile.mkstemp(prefix=".solver-", suffix=".tmp", dir=destination.parent)
os.close(handle)
try:
shutil.copyfile(source, temporary)
os.replace(temporary, destination)
finally:
if os.path.exists(temporary):
os.unlink(temporary)
def _repo_relative(path, root):
root_path = Path(root).resolve()
path_obj = Path(path).resolve()
try:
return path_obj.relative_to(root_path).as_posix()
except ValueError as exc:
raise ValueError(f"evidence path is outside the checkout: {path_obj.name}") from exc
def _validate_island_prompt(record, root, run_dir):
iteration = record.get("iteration") if isinstance(record, dict) else None
if isinstance(iteration, bool) or not isinstance(iteration, int):
raise ValueError("island prompt iteration is invalid")
prompt_file = Path(root, record.get("prompt_path", "")).resolve()
_repo_relative(prompt_file, root)
expected = Path(run_dir, "prompts", f"iter{iteration:03d}.txt").resolve()
if prompt_file != expected or not prompt_file.is_file() or _sha256(prompt_file) != record.get("prompt_hash"):
raise ValueError("island prompt does not match recorded hash")
return os.fspath(prompt_file)
def _evidence_rows(rows, root):
"""Drop bulky solution payloads and normalize local paths before serialization."""
allowed = ("target", "seed", "value", "score", "failed", "error", "returncode", "secs")
clean = []
for row in rows:
item = {key: row[key] for key in allowed if key in row}
if row.get("output_path"):
item["output_path"] = _repo_relative(row["output_path"], root)
clean.append(item)
return clean
_DEVELOPMENT_AGGREGATE_LIMIT = 80
def _read_development_memory(path, aggregate_limit=_DEVELOPMENT_AGGREGATE_LIMIT):
"""Read a bounded candidate window and all prior AST fingerprints.
Retrospective rows share the JSONL for append-only provenance but are not
candidate attempts. The complete fingerprint set prevents an old exact
candidate from becoming novel again when it leaves the prompt window.
"""
if aggregate_limit < 1:
raise ValueError("development aggregate limit must be positive")
if not os.path.exists(path):
return {"history": [], "fingerprints": set(), "total_observations": 0}
history = []
fingerprints = set()
total_observations = 0
with open(path, encoding="utf-8") as stream:
for number, line in enumerate(stream, 1):
try:
entry = json.loads(line)
except json.JSONDecodeError as exc:
raise ValueError(f"invalid development history line {number}") from exc
if not isinstance(entry, dict):
raise ValueError(f"invalid development history line {number}")
if not is_development_observation(entry):
continue
total_observations += 1
fingerprint = entry.get("fingerprint")
if isinstance(fingerprint, str) and re.fullmatch(r"[0-9a-f]{64}", fingerprint):
fingerprints.add(fingerprint)
history.append(entry)
if len(history) > aggregate_limit:
del history[0]
return {
"history": history,
"fingerprints": fingerprints,
"total_observations": total_observations,
}
def _history_entry(record, run_id, hidden_targets):
projected = summarize_development(
[{**record, "run_id": run_id}], hidden_targets=hidden_targets, limit=1, family_limit=1
)["entries"]
return projected[0]
def _incumbent_provenance(loop, root, read_json):
provenance_path = os.path.join(loop.best, "confirmation.json")
provenance = read_json(provenance_path)
if provenance is None:
return {"classification": "historical_best_unvalidated"}
if not isinstance(provenance, dict) or provenance.get("problem") != loop.name:
raise ValueError("invalid confirmed incumbent provenance")
if provenance.get("candidate_hash") != _sha256(loop.champ):
raise ValueError("confirmed incumbent solver hash does not match provenance")
evidence_rel = provenance.get("evidence_path")
if not isinstance(evidence_rel, str):
raise ValueError("confirmed incumbent provenance lacks evidence path")
evidence_path = Path(root, evidence_rel).resolve()
_repo_relative(evidence_path, root)
if provenance.get("evidence_hash") != _sha256(evidence_path):
raise ValueError("confirmed incumbent evidence hash does not match provenance")
evidence = read_json(evidence_path)
if (
not isinstance(evidence, dict)
or evidence.get("problem") != loop.name
or evidence.get("status") != "completed"
or evidence.get("confirmed") is not True
or evidence.get("candidate_hash") != provenance.get("candidate_hash")
):
raise ValueError("confirmed incumbent provenance points to incompatible evidence")
return {
"classification": "confirmed_prior_candidate",
"provenance_path": _repo_relative(provenance_path, root),
"evidence_path": evidence_rel,
"evidence_hash": provenance["evidence_hash"],
}
def _validate_completed_evidence(
evidence, root, problem, provider, model, routing_policy, routing_chain, disabled_families, run_dir=None
):
if evidence.get("problem") != problem or evidence.get("provider") != provider or evidence.get("model") != model:
raise ValueError("completed run id belongs to different problem or provider settings")
routing = evidence.get("routing")
if routing:
if (
routing.get("mode") != routing_policy
or routing.get("configured_chain") != list(routing_chain)
or routing.get("disabled_families") != list(disabled_families)
):
raise ValueError("completed run id belongs to different routing settings")
elif routing_policy != "scheduled" or tuple(routing_chain) != tuple(DEFAULT_CHAIN) or disabled_families:
raise ValueError("historical evidence has no routing metadata for the requested settings")
candidate_path = evidence.get("candidate_path")
if candidate_path:
candidate = Path(root, candidate_path).resolve()
_repo_relative(candidate, root)
if not candidate.is_file() or _sha256(candidate) != evidence.get("candidate_hash"):
raise ValueError("completed candidate does not match recorded hash")
artifacts = evidence.get("artifacts") or {}
if not isinstance(artifacts, dict):
raise ValueError("completed evidence artifacts are invalid")
for relative, expected_hash in artifacts.items():
artifact = Path(root, relative).resolve()
_repo_relative(artifact, root)
if not artifact.is_file() or _sha256(artifact) != expected_hash:
raise ValueError("completed artifact does not match recorded hash")
evolution = (evidence.get("development") or {}).get("evolution") or {}
if evolution:
population = island_evolution.validate_population(evolution.get("population"), root, run_dir)
incumbent = evidence.get("legacy_incumbent") or {}
island_evolution.validate_membership(
population,
(evidence.get("development") or {}).get("candidates", []),
incumbent.get("path"),
incumbent.get("sha256"),
)
for pending in (evidence.get("development") or {}).get("pending_generations", []):
island_evolution.validate_plan(pending, population)
island_evolution.validate_plan_files(pending, root, run_dir)
for candidate in (evidence.get("development") or {}).get("candidates", []):
island_evolution.validate_plan_files(candidate, root, run_dir)
_validate_island_prompt(candidate, root, run_dir)
def _provider_names(mode):
return ("fable", "astra") if mode == "paired" else (mode,)
def _opposite_provider(provider):
return "astra" if provider == "fable" else "fable"
def _paired_iterations_complete(records):
by_iteration = {}
for record in records:
families = by_iteration.setdefault(record.get("iteration"), set())
if not record.get("generation_error") and record.get("family"):
families.add(record["family"])
return bool(by_iteration) and all(families == {"anthropic", "openai"} for families in by_iteration.values())
def _auto_route_chain(problem, history, chain, disabled_families=()):
enabled = [alias for alias in chain if model_spec(alias)["family"] not in set(disabled_families)]
choices = [
{"alias": alias, "problem": problem, "actual_model": model_spec(alias)["model"], "role": "generation"}
for alias in enabled
]
allocation = rank_auto_allocation(operational_stats(history), choices)
selected = (allocation["exploration"][0] if allocation["exploration"] else None) or allocation["primary"]
if selected is None:
return tuple(chain), allocation
primary = selected["alias"]
return (primary, *(alias for alias in chain if alias != primary)), allocation
_ROUTING_WAIT_ROUNDS = 4
_ROUTING_WAIT_MARGIN_SECONDS = 600.0
def _routing_wait_seconds(response, deadline, wait_round, purpose="generation", now=None):
"""Seconds to wait before re-routing, or None when the call is done or waiting cannot help.
Waiting is only ever worth it for a generation that still has a deadline to spend: a critique that cannot
route degrades to promising_unreviewed, and with no deadline there is nothing bounding the wait.
"""
if purpose != "generation" or deadline is None or wait_round >= _ROUTING_WAIT_ROUNDS:
return None
if not response.get("error"):
return None
retry_after = response.get("retry_after_seconds")
if not isinstance(retry_after, (int, float)) or isinstance(retry_after, bool) or not math.isfinite(retry_after):
return None
wait = max(30.0, float(retry_after) + 5.0)
now = time.time() if now is None else float(now)
if now + wait + _ROUTING_WAIT_MARGIN_SECONDS > deadline:
return None
return wait
def _routing_sleep(seconds, deadline, check=None, sleep_fn=time.sleep, clock=time.time, slice_seconds=30.0):
"""Sleep in slices so a pause or the deadline still interrupts a long breaker wait."""
end = clock() + seconds
while True:
remaining = end - clock()
if remaining <= 0 or (deadline is not None and clock() >= deadline):
return
if check:
check()
sleep_fn(min(slice_seconds, remaining))
def _call_with_budget(
call_model_fn,
prompt,
provider,
ledger,
call_budget,
purpose,
usage,
deadline=None,
model=None,
*,
routing_policy="scheduled",
routing_chain=DEFAULT_CHAIN,
disabled_families=(),
routing_journal=None,
routing_checkpoint=None,
routing_scope=None,
legacy_provider_callback=False,
allowance_remaining=None,
wait_check=None,
):
before = ledger.snapshot()
timeout = 900.0 if deadline is None else min(900.0, deadline - time.time())
if timeout <= 0:
raise _ResearchStop("timeout", "research deadline reached before model call")
if routing_journal is None:
response = call_model_fn(
prompt,
provider=provider,
model=model,
timeout=timeout,
max_cost=call_budget,
ledger=ledger,
purpose=purpose,
)
attempts = [{"family": model_spec(provider)["family"], "model": response.get("model"), "status": "completed"}]
else:
# The arm label names a family; the chain decides which model that family runs on.
requested_alias = arm_alias(provider, routing_chain)
if model:
requested_alias = alias_for_model(model)
attempts = []
for wait_round in range(_ROUTING_WAIT_ROUNDS + 1):
response = route_call(
prompt,
requested_alias=requested_alias,
policy=routing_policy,
chain=routing_chain,
disabled_families=disabled_families,
ledger=ledger,
max_cost=call_budget,
purpose=purpose,
call_fn=call_model_fn,
journal=routing_journal,
deadline=deadline,
checkpoint=routing_checkpoint,
scope=routing_scope,
legacy_provider_callback=legacy_provider_callback,
allowance_remaining=allowance_remaining,
)
attempts.extend(response.get("_routing_attempts", []))
wait = _routing_wait_seconds(response, deadline, wait_round, purpose)
if wait is None:
break
# Every route sits behind a breaker that clears on its own (usage window, hung CLI): wait it out
# instead of ending the slot with hours of deadline left (2026-09-15, 2026-09-18).
print(f"[routing] all routes breaker-blocked; waiting {wait:.0f}s before retry {wait_round + 1}")
sys.stdout.flush()
_routing_sleep(wait, deadline, check=wait_check)
after = ledger.snapshot()
physical = [item for item in attempts if item.get("physical", item.get("status") != "skipped")]
usage["calls"] += len(physical)
usage["by_purpose"][purpose] = usage["by_purpose"].get(purpose, 0) + len(physical)
for attempt in physical:
family = attempt.get("family")
actual_model = attempt.get("model")
if family:
usage["by_family"][family] = usage["by_family"].get(family, 0) + 1
if actual_model:
usage["by_model"][actual_model] = usage["by_model"].get(actual_model, 0) + 1
usage["by_provider"][response.get("provider", provider)] = usage["by_provider"].get(
response.get("provider", provider), 0
) + len(physical)
routed_charge = sum(float(item.get("charged_allowance") or 0.0) for item in physical)
usage["charged"] = round(
usage["charged"]
+ (routed_charge if routing_journal is not None else max(0.0, after["spent"] - before["spent"])),
8,
)
if response.get("error_kind") == "budget_exhausted":
raise _ResearchStop("budget_exhausted", response.get("error") or "model allowance exhausted")
if response.get("error_kind") in (
"usage_limit",
"quota_exhausted",
"authentication",
"unavailable",
"model_unavailable",
"rate_limited_unclassified",
"routing_unavailable",
):
raise _ResearchStop("provider_unavailable", response.get("error") or "subscription provider unavailable")
return response
DEFAULT_SCREEN_FRACTION = 0.25
DEFAULT_DEVELOPMENT_SEEDS = 3
DEFAULT_POSTMORTEM_LIMIT = 3
DEFAULT_POSTMORTEM_BUDGET = 0.5
MAX_TRACE_POINTS = 64
def search_summary(path, maximize):
"""The optional best-so-far curve a solver may write beside its solution, validated and bounded.
The loop sees a final objective and a wall time, which cannot tell a solver that converged after four
seconds of its hundred from one that was still improving when the budget ran out -- the difference
between "search better" and "search harder", and the question the incumbent profile could not answer.
A solver may record its own curve as ``"trace": [[seconds, objective], ...]``. Nothing here scores it:
the independent verifier remains the only source of a value, and a missing or malformed trace is simply
not reported.
"""
try:
with open(path, encoding="utf-8") as stream:
data = json.load(stream)
except (OSError, UnicodeDecodeError, json.JSONDecodeError):
return None
trace = data.get("trace") if isinstance(data, dict) else None
if not isinstance(trace, list) or not trace:
return None
points = []
for item in trace[:MAX_TRACE_POINTS]:
if not isinstance(item, (list, tuple)) or len(item) != 2:
return None
seconds, objective = item
for number in (seconds, objective):
if isinstance(number, bool) or not isinstance(number, (int, float)) or not math.isfinite(number):
return None
if seconds < 0:
return None
points.append((float(seconds), float(objective)))
points.sort(key=lambda point: point[0])
ahead = (lambda new, best: new > best) if maximize else (lambda new, best: new < best)
best = points[0][1]
time_to_best = points[0][0]
improvements = 0
for seconds, objective in points[1:]:
if ahead(objective, best):
best, time_to_best, improvements = objective, seconds, improvements + 1
return {"points": len(points), "improvements": improvements, "time_to_best": round(time_to_best, 3)}
def _profile_number(value, digits=4):
if not isinstance(value, (int, float)) or isinstance(value, bool) or not math.isfinite(float(value)):
return "-"
return f"{float(value):.{digits}g}"
def development_profile(incumbent_rows, records, time_budget, last_comparison=None):
"""Per-target incumbent numbers for the prompt: value, gap to reference, and time actually used.
The scoreboard alone says which targets are behind; it does not say whether the solver is spending its
time budget or returning early, which is the difference between "search harder" and "search better".
"""
deltas = {}
if isinstance(last_comparison, dict):
for pair in last_comparison.get("pairs") or []:
deltas.setdefault(pair["target"], []).append(pair["gain"])
ordered = []
by_target = {}
for row in incumbent_rows:
target = row["target"]
if target not in by_target:
by_target[target] = {"values": [], "secs": [], "failed": 0}
ordered.append(target)
entry = by_target[target]
if isinstance(row.get("value"), (int, float)) and not isinstance(row.get("value"), bool):
entry["values"].append(float(row["value"]))
if isinstance(row.get("secs"), (int, float)) and not isinstance(row.get("secs"), bool):
entry["secs"].append(float(row["secs"]))
entry["failed"] += bool(row.get("failed"))
search = row.get("search")
if isinstance(search, dict):
entry.setdefault("time_to_best", []).append(search.get("time_to_best"))
entry.setdefault("improvements", []).append(search.get("improvements"))
lines = [
"target | reference | incumbent | seconds used | seconds allowed | time to best | improvements | "
"last candidate delta"
]
for target in ordered:
entry = by_target[target]
value = statistics.median(entry["values"]) if entry["values"] else None
seconds = max(entry["secs"]) if entry["secs"] else None
delta = statistics.median(deltas[target]) if deltas.get(target) else None
incumbent = "failed" if entry["failed"] and not entry["values"] else _profile_number(value, 8)
curve = [value for value in entry.get("time_to_best", []) if isinstance(value, (int, float))]
counts = [value for value in entry.get("improvements", []) if isinstance(value, (int, float))]
lines.append(
f"{target} | {_profile_number(records.get(target), 8)} | {incumbent} | "
f"{_profile_number(seconds, 3)} | {_profile_number(time_budget, 3)} | "
f"{_profile_number(statistics.median(curve), 3) if curve else '-'} | "
f"{_profile_number(statistics.median(counts), 3) if counts else '-'} | "
+ (f"{delta * 100:+.3f}%" if delta is not None else "-")
)
return "\n".join(lines)
def screen_cells(matrix, fraction):
"""The leading cells a candidate must survive before the rest of the matrix is spent on it.
Measured on 2026-09-14/16/17 (77 candidates with full pairs): a quarter-matrix prefix screened 6 of 44
rejected candidates and none of the 33 promising ones, recovering 5.8% of development solver seconds.
A half-matrix screen saved 1.3%; a two-target screen saved more but discarded a candidate that went on
to pass, so the fraction is deliberately conservative.
"""
if not fraction or float(fraction) <= 0:
return []
targets = []
for cell in matrix:
if cell["target"] not in targets:
targets.append(cell["target"])
count = max(2, math.ceil(len(targets) * float(fraction)))
if count >= len(targets):
return []
leading = set(targets[:count])
return [cell for cell in matrix if cell["target"] in leading]
def race_stages(matrix, screen_fraction):
"""The cell groups a candidate must survive in order, cheapest first.
A single-seed verdict is not a measurement: the same solver on the same target moved 0.06%-0.57% between
seeds on cvrp confirmation runs (2026-09-08 to 09-17), while the acceptance threshold is 0.01%. Every
replication seed therefore has to be spent before a candidate can be called promising -- and, because a
night only fits so many solver seconds, spent only on candidates still ahead after the cheaper stages.
"""
seeds = sorted({cell["seed"] for cell in matrix})
first = [cell for cell in matrix if cell["seed"] == seeds[0]]
screen = screen_cells(first, screen_fraction)
screen_keys = {(cell["target"], cell["seed"]) for cell in screen}
stages = []
if screen:
stages.append({"kind": "screen", "seed": seeds[0], "cells": screen})
rest = [cell for cell in first if (cell["target"], cell["seed"]) not in screen_keys]
if rest:
stages.append({"kind": "first_seed", "seed": seeds[0], "cells": rest})