From fc8ddfd2f45d4ad3647b7adc6f5839ec71b3c3a5 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 15 Sep 2026 00:22:40 +0000 Subject: [PATCH 1/2] Wire dead-ends, pattern library and schedule heuristic into the governed loop - New research_context.py injects two sanitized blocks into every generation prompt: problem-scoped entries from problems/_dead_ends.json and a ranked cross-problem pattern digest from problems/*/patterns/ + nightly/patterns/. Injected ids/names are recorded in evidence.json under prompt_context. - Plugins declare PATTERN_TAGS used to rank transferable patterns. - Hidden-target sanitization and the prompt leak check are now token-boundary aware, so single-character targets (e.g. "4") no longer mangle text like de-004 or 400s or false-flag a prompt. - schedule_night scores governed runs/research history as well as legacy loop reports; night planning orders non-trial research slots by its information-gain heuristic and records the advisory allocation as schedule_plan in night status and dry-run output. Trial order, providers and allowances are unchanged. Co-Authored-By: Wes Sander --- CHANGELOG.md | 6 + README.md | 4 +- docs/RESEARCH-IMPLEMENTATION.md | 4 + loop.py | 25 ++- night.py | 38 ++++ problems/circle_packing/problem.py | 1 + problems/cvrp/problem.py | 1 + problems/matrix_multiplication/problem.py | 1 + problems/miplib/problem.py | 1 + problems/miplib_heur/problem.py | 1 + problems/miplib_open/problem.py | 1 + problems/pglib_opf/problem.py | 1 + research_context.py | 164 ++++++++++++++ research_memory.py | 32 ++- scripts/schedule_night.py | 16 ++ tests/test_research_context.py | 255 ++++++++++++++++++++++ 16 files changed, 540 insertions(+), 11 deletions(-) create mode 100644 research_context.py create mode 100644 tests/test_research_context.py diff --git a/CHANGELOG.md b/CHANGELOG.md index fe12f82..7b541dd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,11 @@ # Changelog +## Cross-problem prompt context and schedule advice, 2026-09-14 + +- Generation prompts now carry the repo-wide dead-ends ledger (problem-scoped, sanitized) and a ranked cross-problem pattern digest aggregated from `problems/*/patterns/` and `nightly/patterns/`; injected ids are recorded in `evidence.json` under `prompt_context`. +- Plugins declare structural `PATTERN_TAGS` used to rank transferable patterns. +- `scripts/schedule_night.py` scores governed `runs/research/*/run.json` history alongside legacy loop reports; nightly planning orders non-trial research slots by its heuristic and records the advisory allocation as `schedule_plan` in the night status, without touching the counterbalanced trial assignments. + ## ARC-AGI-N local companion, 2026-09-08 - Import a bounded, content-hashed local snapshot of reviewed ARC-AGI-N catalogue records without pulling, executing upstream code, or passing upstream prose into model prompts. diff --git a/README.md b/README.md index cd34eaa..8984d89 100644 --- a/README.md +++ b/README.md @@ -212,8 +212,8 @@ Supporting scripts: - `scripts/loop_report.py` — build the dashboard; `--dir` regenerates one for an old run - `scripts/publish_draft.py` — draft generation for verified record breaks -- `scripts/dead_ends.py` — repo-wide ledger of failed approaches (`list`/`check`/`record`), consulted before each night's candidate design -- `scripts/schedule_night.py` — split the night's compute budget across problems by expected information gain +- `scripts/dead_ends.py` — repo-wide ledger of failed approaches (`list`/`check`/`record`); matching entries are injected into every night's candidate prompt automatically +- `scripts/schedule_night.py` — scores problems by expected information gain; orders non-trial night slots and records the advisory allocation - `scripts/detached.py` — `status` prints the dashboard path when one exists ## Nightly integration diff --git a/docs/RESEARCH-IMPLEMENTATION.md b/docs/RESEARCH-IMPLEMENTATION.md index 03b37ea..426ae6c 100644 --- a/docs/RESEARCH-IMPLEMENTATION.md +++ b/docs/RESEARCH-IMPLEMENTATION.md @@ -68,6 +68,10 @@ The prompt projection is bounded and sanitized: development provider, actual mod Auto allocation prioritizes enabled model-role choices with fewer than three attributable development attempts; only after each enabled choice reaches that threshold can it rank a mature primary. It does not change the fixed routing fallback chain and receives no holdout, confirmation, promotion, or reward feedback. This is capacity and attribution work, not evidence that a model is better. +`research_context.py` injects two bounded, sanitized blocks into every generation prompt: the repo-wide dead-ends ledger (`problems/_dead_ends.json`, entries for the active problem plus `general`) and the cross-problem pattern digest (`problems/*/patterns/` and `nightly/patterns/`, ranked by `PATTERN_TAGS` overlap then recorded outcomes). Withheld target names and local paths are stripped before injection, and `evidence.json` records exactly which dead-end ids and pattern names the run saw under `prompt_context`. Plugins declare their structural features via `PATTERN_TAGS`; a problem with no tags still receives the pattern digest ranked by outcomes. + +Nightly slot planning consults `scripts/schedule_night.py`: non-trial research slots are ordered by its information-gain heuristic after the counterbalanced trial pair, and the advisory allocation is recorded in the night's status as `schedule_plan`. The trial pair's order and allowances are unchanged; the heuristic scores governed `runs/research/*//run.json` history alongside legacy `loop_report.json` runs and can never alter the fixed trial assignments. + The local evidence scan covered 31 solver candidates across `runs-cvrp`, `runs-miplib_heur`, and `runs`: 0 syntax failures and 0 exact AST duplicates. It found seven structured CVRP development-history records, but none had actual-model, model, or role fields. The immediate rationale is therefore prospective: cheap duplicate prevention and reliable attribution before allocation. It has not demonstrated saved compute or a discovery gain. Consciously deferred: near-similarity suppression, bandit allocation, additional model calls, and any holdout-fed reward. ## Dashboard contract diff --git a/loop.py b/loop.py index 01d9a03..1d131cd 100644 --- a/loop.py +++ b/loop.py @@ -29,6 +29,7 @@ from pathlib import Path import evaluation +import research_context from model_registry import ( DEFAULT_CHAIN, MODEL_REGISTRY, @@ -40,8 +41,11 @@ from research_memory import ( analyze_candidate, is_development_observation, + mentions_target, operational_stats, rank_auto_allocation, + redact_targets, + strip_local_paths, summarize_development, ) from routing import RoutingJournal, route_call, routing_summary @@ -339,8 +343,16 @@ def build_research_prompt( retro_memory=None, history_total=None, mission=None, + context_blocks=None, ): """Build a prompt from development data only.""" + if context_blocks is None: + name = getattr(self, "name", None) + context_blocks = ( + research_context.blocks(name, self.P, getattr(self, "root", HERE), hidden_targets) + if name + else {"text": "", "dead_ends": [], "patterns": []} + ) if hasattr(self.P, "prompt_for_targets"): context = self.P.prompt_for_targets(list(targets)) else: @@ -364,11 +376,7 @@ def build_research_prompt( prior = json.dumps(memory, sort_keys=True, separators=(",", ":")) if memory["entries"] else "(none yet)" retro = {key: str((retro_memory or {}).get(key, ""))[:1000] for key in ("lessons", "next_experiment")} for key, value in retro.items(): - for target in hidden_targets: - value = value.replace(str(target), "[withheld reference removed]") - retro[key] = re.sub( - r"(?] ". @@ -424,7 +434,7 @@ def build_research_prompt( correctness. OUTPUT FORMAT: the tagged IDEA line, then exactly one ```python block with the full file. Nothing else.""" - leaked = [str(target) for target in hidden_targets if str(target) in prompt] + leaked = [str(target) for target in hidden_targets if mentions_target(prompt, target)] if leaked: raise ValueError(f"generation prompt exposes withheld targets: {leaked}") return prompt @@ -980,6 +990,7 @@ def run_research( retro_memory = read_json(os.path.join(evidence_base, "development-history", f"{problem}-retro.json"), {}) or {} if not isinstance(retro_memory, dict) or retro_memory.get("schema_version") not in (None, 1): retro_memory = {} + prompt_context = research_context.blocks(problem, plugin, root, hidden_targets) effective_routing_chain = routing_chain auto_allocation = None if routing_policy == "auto": @@ -1096,6 +1107,7 @@ def run_research( "usage": usage, "limitations": list(manifest["limitations"]), "mission": mission, + "prompt_context": {"dead_ends": prompt_context["dead_ends"], "patterns": prompt_context["patterns"]}, "legacy_incumbent": { "path": _repo_relative(incumbent_snapshot, root), "sha256": _sha256(incumbent_snapshot), @@ -1182,6 +1194,7 @@ def append_development_record(record): retro_memory, development_memory["total_observations"], mission, + prompt_context, ) responses = [] deferred_stop = None diff --git a/night.py b/night.py index 5ba611d..d8dce0b 100644 --- a/night.py +++ b/night.py @@ -185,11 +185,46 @@ def planned_slots(config, run_id): float(config["night"]["provider_caps_usd"][slot["provider"]]), ) ordered.append(slot) + # Research slots outside the trial keep their configured provider and run + # in information-gain order before the validation tail. + extras = [ + slot + for problem, slot in by_problem.items() + if problem not in assignment["order"] and problem != "pglib_opf" and slot.get("kind") == "research" + ] + try: + from scripts.schedule_night import score_problem + + extras.sort(key=lambda slot: score_problem(slot["problem"])["score"], reverse=True) + except Exception: + pass # ordering is advisory; never break the night's plan + for slot in extras: + slot["provider"] = slot.get("provider") or "paired" + slot["effective_slot_budget_usd"] = min( + float(slot["slot_budget_usd"]), + float(config["night"]["provider_caps_usd"][slot["provider"]]), + ) + ordered.append(slot) # PGLib is confirmation-only and deliberately has no generation provider. ordered.append(by_problem["pglib_opf"]) return ordered +def schedule_advice(config, slots): + """Advisory information-gain allocation for the night's status and morning report.""" + try: + from scripts.schedule_night import allocate + except ImportError: + return None + try: + return allocate( + float(config["night"]["deadline_minutes"]) * 60.0, + [slot["problem"] for slot in slots], + ) + except Exception: + return None + + def _routing_families(config, slots): """Return only families that can be selected by the configured night.""" routing = config["night"]["routing"] @@ -510,6 +545,7 @@ def run_night( ledger_path = run_root / "budget.json" slots = planned_slots(config, run_id) planned_order = [slot["id"] for slot in slots] + schedule_plan = schedule_advice(config, slots) arc_plan, arc_summary = _prepare_arc(config, slots, run_id) requested_next = arc_summary.get("requested_next") chosen_slot = next( @@ -530,6 +566,7 @@ def run_night( "budget_accounting": config["night"].get("budget_accounting"), "deadline_minutes": int(config["night"]["deadline_minutes"]), "slots": slots, + "schedule_plan": schedule_plan, "routing": {**routing, "override": bool(run_routing_override)}, "arc": arc_summary, } @@ -585,6 +622,7 @@ def run_night( scheduled_run_id=scheduled_run_id, ) status["arc"] = arc_summary + status["schedule_plan"] = schedule_plan disabled_slot_ids = set(arc_summary.get("disabled_slot_ids", [])) for slot_id, mission in arc_plan.items(): plugin = mission["plugin"] diff --git a/problems/circle_packing/problem.py b/problems/circle_packing/problem.py index df0f4fa..5eede2e 100644 --- a/problems/circle_packing/problem.py +++ b/problems/circle_packing/problem.py @@ -17,6 +17,7 @@ VALIDATION = [] RELEASE_HOLDOUT = [] DEFAULTS = {"time": 120, "workers": 3} +PATTERN_TAGS = ["continuous", "fast-verifier", "local-search", "generatable-test-cases"] MAXIMIZE = True FAIL_SCORE = 0.0 WIN_MARGIN = 1e-10 diff --git a/problems/cvrp/problem.py b/problems/cvrp/problem.py index 3f1d29b..1981255 100644 --- a/problems/cvrp/problem.py +++ b/problems/cvrp/problem.py @@ -24,6 +24,7 @@ VALIDATION = [] RELEASE_HOLDOUT = [] DEFAULTS = {"time": 120, "workers": 3} +PATTERN_TAGS = ["local-search", "fast-verifier", "combinatorial", "route-structure", "generatable-test-cases", "huge-raw-search-space"] MAXIMIZE = False FAIL_SCORE = -1.0 # a crash / timeout / infeasible output; strictly worse than any feasible run (gap clipped at 0.5) GAP_CLIP = 0.5 diff --git a/problems/matrix_multiplication/problem.py b/problems/matrix_multiplication/problem.py index efd04d5..031a7e6 100644 --- a/problems/matrix_multiplication/problem.py +++ b/problems/matrix_multiplication/problem.py @@ -37,6 +37,7 @@ VALIDATION = [] RELEASE_HOLDOUT = [] DEFAULTS = {"time": 300, "workers": 1} +PATTERN_TAGS = ["block-structure", "disjoint-outputs", "subproblems", "recursive-structure", "multiplicative-cost", "technique-library", "composition-operators", "huge-raw-search-space", "fast-verifier", "generatable-test-cases"] MAXIMIZE = False FAIL_SCORE = -1.0 # crash / timeout / infeasible output; worse than any feasible run GAP_CLIP = 0.5 diff --git a/problems/miplib/problem.py b/problems/miplib/problem.py index 1df8607..9b57f68 100644 --- a/problems/miplib/problem.py +++ b/problems/miplib/problem.py @@ -45,6 +45,7 @@ "neos-5045105-creuse": "3848 vars / 252 rows, integer knapsacks, general integers", } DEFAULTS = {"time": 400, "workers": 3} +PATTERN_TAGS = ["local-search", "combinatorial", "huge-raw-search-space"] MAXIMIZE = False FAIL_SCORE = -10.0 REL_TOL = 1e-6 diff --git a/problems/miplib_heur/problem.py b/problems/miplib_heur/problem.py index d1f5307..7e6424e 100644 --- a/problems/miplib_heur/problem.py +++ b/problems/miplib_heur/problem.py @@ -46,6 +46,7 @@ def _desc(name): INFO = {t: _desc(t) for t in TARGETS} DEFAULTS = {"time": 60, "workers": 3} +PATTERN_TAGS = ["local-search", "fast-verifier", "combinatorial", "huge-raw-search-space", "generatable-test-cases"] MAXIMIZE = False FAIL_SCORE = -1.0 # added to the champion total directly (score space), so a failed target costs a 100% gap WIN_MARGIN = 1e-4 # gap must improve on HiGHS default by 0.01% of the objective to count (timing noise floor) diff --git a/problems/miplib_open/problem.py b/problems/miplib_open/problem.py index 941dda4..1812299 100644 --- a/problems/miplib_open/problem.py +++ b/problems/miplib_open/problem.py @@ -28,6 +28,7 @@ VALIDATION = [] RELEASE_HOLDOUT = [] DEFAULTS = {"time": 600, "workers": 3} # justified from a measured seed run in BASELINE.md +PATTERN_TAGS = ["local-search", "combinatorial", "huge-raw-search-space"] MAXIMIZE = False # value is min-sense (lower is better); the loop maximises total = minus the summed value FAIL_SCORE = -1.0 # a crash / timeout / infeasible output, in score space (worse than any clipped feasible gap) GAP_CLIP = 1.0 # one hopeless instance (100% above best-known) cannot dominate the champion total diff --git a/problems/pglib_opf/problem.py b/problems/pglib_opf/problem.py index 9adcd0d..21b56e9 100644 --- a/problems/pglib_opf/problem.py +++ b/problems/pglib_opf/problem.py @@ -45,6 +45,7 @@ VALIDATION = [] RELEASE_HOLDOUT = [] DEFAULTS = {"time": 90, "workers": 4} +PATTERN_TAGS = ["continuous", "fast-verifier", "local-search"] MAXIMIZE = False FAIL_SCORE = -1.0 WIN_MARGIN = 1e-4 # conservative preliminary screen; release validation uses each row's exact printed uncertainty diff --git a/research_context.py b/research_context.py new file mode 100644 index 0000000..899e20d --- /dev/null +++ b/research_context.py @@ -0,0 +1,164 @@ +"""Bounded cross-problem context for generation prompts. + +Two repo-local ledgers feed this module, both development-only: + +- ``problems/_dead_ends.json``: verified failed approaches recorded through + ``scripts/dead_ends.py``. Consulted at candidate design so a night does not + re-run a proven dead end without a changed mechanism. +- ``problems/*/patterns/*.json`` and ``nightly/patterns/*.json``: transferable + search patterns. Two record schemas exist -- the ``patterns/library.py`` + record (``description``/``applies_when``/``transform_ref``/``successes``) and + the free-text PATTERNS.md record (``abstract_description``/``applicability``/ + ``success_count``) -- both are normalized to a name, a one-line description, + an origin problem and an outcome count. + +Everything injected here is sanitized like the rest of development memory: +withheld target names and local paths are stripped before reaching a prompt, +and the run evidence records exactly which entries were injected. +""" + +from __future__ import annotations + +import glob +import json +import os +import re + +from research_memory import redact_targets, strip_local_paths + +_TOKEN = re.compile(r"[a-z0-9]+") + + +def _sanitize(text, hidden_targets): + return strip_local_paths(redact_targets(text, hidden_targets)) + + +def _tokens(text): + return set(_TOKEN.findall(str(text).lower())) + + +def load_dead_ends(root): + """All recorded dead ends, file order (oldest first).""" + path = os.path.join(root, "problems", "_dead_ends.json") + try: + with open(path, encoding="utf-8") as fh: + data = json.load(fh) + except (OSError, UnicodeDecodeError, json.JSONDecodeError): + return [] + if not isinstance(data, list): + return [] + return [entry for entry in data if isinstance(entry, dict) and entry.get("approach")] + + +def dead_ends_for(problem, root, hidden_targets=(), limit=15): + """Newest ``limit`` dead ends for *problem* (plus ``general``), sanitized for a prompt.""" + entries = [ + entry + for entry in load_dead_ends(root) + if entry.get("problem") in (problem, "general") + ][-limit:] + lines = [] + for entry in entries: + approach = str(entry.get("approach", ""))[:220] + why = str(entry.get("why_failed", ""))[:220] + line = f"- [{entry.get('id', '?')}] {approach} -- failed: {why}" + tags = [str(tag)[:30] for tag in entry.get("tags", [])][:8] + if tags: + line += f" (tags: {', '.join(tags)})" + lines.append(line) + return _sanitize("\n".join(lines), hidden_targets), [entry.get("id") for entry in entries] + + +def _normalize_pattern(record, origin, path): + name = record.get("name") + description = record.get("description") or record.get("abstract_description") or "" + outcomes = int(record.get("successes", 0) or 0) + int(record.get("failures", 0) or 0) + outcomes += int(record.get("success_count", 0) or 0) + int(record.get("failure_count", 0) or 0) + tags = set(record.get("applies_when", []) or []) + if not tags: + tags = _tokens( + " ".join( + str(part) + for part in ( + record.get("applicability", ""), + " ".join(record.get("operators", []) or []), + ) + ) + ) + return { + "name": name, + "description": str(description)[:220], + "origin": origin, + "outcomes": outcomes, + "tags": tags, + "ref": record.get("transform_ref") or "", + "path": path, + } + + +def load_patterns(root): + """Every pattern record under ``problems/*/patterns/`` and top-level ``*/patterns/``.""" + roots = glob.glob(os.path.join(root, "problems", "*", "patterns")) + glob.glob( + os.path.join(root, "*", "patterns") + ) + patterns = [] + seen = set() + for directory in sorted(set(roots)): + for path in sorted(glob.glob(os.path.join(directory, "*.json"))): + if os.path.basename(path).startswith("_"): + continue + try: + with open(path, encoding="utf-8") as fh: + record = json.load(fh) + except (OSError, UnicodeDecodeError, json.JSONDecodeError): + continue + if not isinstance(record, dict) or not record.get("name") or record["name"] in seen: + continue + seen.add(record["name"]) + origin = record.get("origin_problem") or os.path.basename(os.path.dirname(directory)) + patterns.append(_normalize_pattern(record, origin, path)) + return patterns + + +def patterns_for(problem, plugin, root, hidden_targets=(), limit=10): + """Patterns ranked for *problem*: tag overlap first, then observed outcomes.""" + wanted = set(getattr(plugin, "PATTERN_TAGS", ()) or ()) + wanted.add(problem) + scored = [] + for pattern in load_patterns(root): + overlap = len(wanted & set(pattern["tags"])) + scored.append((overlap, pattern["outcomes"], pattern["name"], pattern)) + scored.sort(key=lambda item: (-item[0], -item[1], item[2])) + chosen = scored[:limit] + lines = [] + for overlap, _outcomes, _name, pattern in chosen: + line = f"- {pattern['name']} (from {pattern['origin']}, {pattern['outcomes']} outcome(s)" + if overlap: + line += f", {overlap} matching tag(s)" + line += f"): {pattern['description']}" + if pattern["ref"]: + line += f" [{pattern['ref'][:100]}]" + lines.append(line) + return _sanitize("\n".join(lines), hidden_targets), [item[3]["name"] for item in chosen] + + +def blocks(problem, plugin, root, hidden_targets=(), dead_end_limit=15, pattern_limit=10): + """The prompt context for one run: rendered text plus the injected ids for evidence.""" + dead_text, dead_ids = dead_ends_for(problem, root, hidden_targets, dead_end_limit) + pattern_text, pattern_names = patterns_for(problem, plugin, root, hidden_targets, pattern_limit) + sections = [] + if dead_text: + sections.append( + "KNOWN DEAD ENDS (verified failures in this lab; a repeat needs a changed mechanism " + "and a sentence naming it):\n" + dead_text + ) + if pattern_text: + sections.append( + "TRANSFERABLE PATTERNS (techniques recorded from other problems; adapt only when the " + "problem structure fits):\n" + pattern_text + ) + return { + "text": "\n\n".join(sections), + "dead_ends": dead_ids, + "patterns": pattern_names, + } diff --git a/research_memory.py b/research_memory.py index 491bb3f..50d456c 100644 --- a/research_memory.py +++ b/research_memory.py @@ -22,6 +22,33 @@ _ABSOLUTE_PATH = re.compile(r"(? list: continue if rep.get("problem") == problem: reports.append(rep) + # Governed research runs: runs/research///run.json. + for path in glob.glob(os.path.join(_RUNS_DIR, "research", "*", problem, "run.json")): + try: + with open(path) as fh: + run = json.load(fh) + except (OSError, json.JSONDecodeError): + continue + if run.get("problem") != problem: + continue + reports.append( + { + "problem": problem, + "generated_at": run.get("finished_at") or run.get("updated_at") or run.get("started_at"), + "status": "success" if run.get("status") in {"completed", "partial"} else run.get("status"), + } + ) reports.sort(key=lambda r: r.get("generated_at", ""), reverse=True) return reports diff --git a/tests/test_research_context.py b/tests/test_research_context.py new file mode 100644 index 0000000..8200642 --- /dev/null +++ b/tests/test_research_context.py @@ -0,0 +1,255 @@ +import json +import os +import types +from pathlib import Path + +import loop +import night +import research_context +from scripts import schedule_night + + +def _write(root, relative, payload): + path = Path(root) / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload), encoding="utf-8") + return path + + +def test_dead_ends_scoped_sanitized_and_bounded(tmp_path): + _write( + tmp_path, + "problems/_dead_ends.json", + [ + { + "id": "de-100", + "date": "2026-09-10", + "problem": "cvrp", + "approach": "oracle tuning on holdout-instance X-n999", + "why_failed": "leaks the confirmation split at /private/run.json", + "tags": ["oracle"], + }, + { + "id": "de-101", + "date": "2026-09-10", + "problem": "miplib_heur", + "approach": "unrelated", + "why_failed": "off scope", + }, + { + "id": "de-102", + "date": "2026-09-10", + "problem": "general", + "approach": "blanket retry of the same code", + "why_failed": "exact duplicate", + }, + ], + ) + text, ids = research_context.dead_ends_for("cvrp", str(tmp_path), hidden_targets=("X-n999",)) + assert ids == ["de-100", "de-102"] + assert "de-101" not in text + assert "X-n999" not in text + assert "/private" not in text + assert "blanket retry" in text + + +def test_patterns_normalize_both_schemas_and_rank(tmp_path): + _write( + tmp_path, + "problems/matrix_multiplication/patterns/block.json", + { + "name": "block-decomposition", + "description": "split into blocks, solve each", + "applies_when": ["block-structure"], + "transform_ref": "problems/matrix_multiplication/composition.py:block_embed", + "successes": 2, + "failures": 0, + }, + ) + _write( + tmp_path, + "nightly/patterns/glue.json", + { + "name": "glue-analysis", + "abstract_description": "remove cancellation-only components first", + "origin_problem": "matrix_multiplication", + "applicability": "composite constructions", + "operators": ["remove_glue_first"], + "success_count": 0, + "failure_count": 1, + }, + ) + _write(tmp_path, "problems/x/patterns/_bandit.json", {"state": True}) + plugin = types.SimpleNamespace(PATTERN_TAGS=["block-structure"]) + text, names = research_context.patterns_for("cvrp", plugin, str(tmp_path)) + assert names[0] == "block-decomposition" + assert "glue-analysis" in names + assert "state" not in names + assert "2 outcome(s)" in text + assert "remove cancellation-only" in text + + +def test_blocks_empty_when_nothing_recorded(tmp_path): + result = research_context.blocks("cvrp", types.SimpleNamespace(), str(tmp_path)) + assert result == {"text": "", "dead_ends": [], "patterns": []} + + +class _FakeProblem: + TARGETS = ["dev", "validation"] + DEVELOPMENT_TARGETS = ["dev"] # noqa: vulture (read by loop.py via the plugin) + VALIDATION_TARGETS = ["validation"] # noqa: vulture + HOLDOUT = ["holdout"] # noqa: vulture + DEFAULTS = {"time": 1, "workers": 2} + FAIL_SCORE = -100.0 + PATTERN_TAGS = ["block-structure"] + PROMPT = "legacy prompt" + TASK = "write the solver" + + @staticmethod + def prompt_for_targets(targets): + return "development targets: " + ",".join(targets) + + @staticmethod + def records_load(): + return {"dev": 0.0, "validation": 0.0, "holdout": 0.0} + + @staticmethod + def evaluate(path, _target): + return json.loads(open(path, encoding="utf-8").read())["value"], {} + + @staticmethod + def score(value, _record): + return value + + +def _fixture_root(tmp_path): + champion = tmp_path / "best-fake" / "solver.py" + champion.parent.mkdir(parents=True) + champion.write_text("# incumbent\n", encoding="utf-8") + return champion + + +def _runner(_problem, solver, target, _budget, seed, out, **_kwargs): + source = open(solver, encoding="utf-8").read() + value = 2.0 if "fable candidate" in source else 1.0 + os.makedirs(os.path.dirname(out), exist_ok=True) + with open(out, "w", encoding="utf-8") as stream: + json.dump({"value": value}, stream) + return types.SimpleNamespace(returncode=0, stdout="", stderr="") + + +def test_prompt_carries_dead_ends_and_patterns(tmp_path): + _write( + tmp_path, + "problems/_dead_ends.json", + [{"id": "de-1", "problem": "fake", "approach": "retry naive encoding", "why_failed": "timeout"}], + ) + _write( + tmp_path, + "problems/fake/patterns/block.json", + { + "name": "block-decomposition", + "description": "split into blocks", + "applies_when": ["block-structure"], + "transform_ref": "x:y", + "successes": 1, + }, + ) + instance = loop.Loop("fake", root=tmp_path, problem_module=_FakeProblem, initialize_best=False) + prompt = instance.build_research_prompt( + "# incumbent\n", ["dev"], {"dev": 0.0}, [], hidden_targets=("holdout",) + ) + assert "KNOWN DEAD ENDS" in prompt + assert "retry naive encoding" in prompt + assert "TRANSFERABLE PATTERNS" in prompt + assert "block-decomposition" in prompt + assert "holdout" not in prompt + + +def test_run_research_records_prompt_context(tmp_path): + _fixture_root(tmp_path) + _write( + tmp_path, + "problems/_dead_ends.json", + [{"id": "de-7", "problem": "fake", "approach": "blind restart", "why_failed": "stall"}], + ) + calls = [] + + def model(prompt, **_kwargs): + calls.append(prompt) + if len(calls) == 1: + return {"code": "# fable candidate\nVALUE = 2\n", "idea": "[kind: test] improved"} + return {"error": "stop", "error_kind": "usage_limit"} + + evidence = loop.run_research( + "fake", + root=tmp_path, + provider="fable", + iters=2, + problem_module=_FakeProblem, + call_model_fn=model, + solver_runner=_runner, + ) + assert evidence["prompt_context"]["dead_ends"] == ["de-7"] + assert "KNOWN DEAD ENDS" in calls[0] + assert "blind restart" in calls[0] + + +def test_planned_slots_orders_extra_research_before_validation(): + config = night.load_schedule(Path(night.HERE) / "night.json") + config["slots"].insert( + 2, + { + "id": "matmul-research", + "problem": "matrix_multiplication", + "kind": "research", + "provider": "paired", + "minutes": 30, + "research_minutes": 20, + "retro_minutes": 10, + "slot_budget_usd": 10.0, + "per_call_budget_usd": 2.0, + "retro_budget_usd": 2.0, + "iters": 10, + "seed_count": 1, + "min_effect": 0.0001, + "time_per_target": 60, + "workers": 1, + }, + ) + slots = night.planned_slots(config, "2026-09-06") + assert [slot["problem"] for slot in slots][-2:] == ["matrix_multiplication", "pglib_opf"] + matmul = slots[2] + assert matmul["provider"] == "paired" + assert matmul["effective_slot_budget_usd"] == 10.0 + + +def test_schedule_history_reads_governed_runs(tmp_path, monkeypatch): + monkeypatch.setattr(schedule_night, "_RUNS_DIR", str(tmp_path)) + _write( + tmp_path, + "research/2026-09-10/cvrp/run.json", + {"problem": "cvrp", "status": "completed", "finished_at": "2026-09-10T05:00:00Z"}, + ) + _write( + tmp_path, + "research/2026-09-11/cvrp/run.json", + {"problem": "cvrp", "status": "error", "updated_at": "2026-09-11T05:00:00Z"}, + ) + history = schedule_night._history("cvrp") + assert len(history) == 2 + assert history[0]["generated_at"] == "2026-09-11T05:00:00Z" + assert history[1]["status"] == "success" + score = schedule_night.score_problem("cvrp") + assert score["runs_recorded"] == 2 + assert 0 <= score["components"]["velocity"] <= 3.0 + + +def test_dry_run_reports_schedule_plan(tmp_path): + config = night.load_schedule(Path(night.HERE) / "night.json") + config["arc"]["enabled"] = False + config["night"]["evidence_root"] = str(tmp_path) + plan = night.run_night(config, "2026-09-12", dry_run=True) + assert plan["dry_run"] is True + assert plan["schedule_plan"]["budget_s"] == 480 * 60.0 + assert set(plan["schedule_plan"]["allocations"]) >= {"cvrp", "miplib_heur", "pglib_opf"} From 2b0efcd5e8f51380ede9c96584ea01f4ebfc3e23 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 15 Sep 2026 00:40:34 +0000 Subject: [PATCH 2/2] Address review: merge duplicate patterns, drop outcome/refs from prompts, restore prompt context on resume, admit bounded non-trial research slots - load_patterns merges same-named records (tag union, summed outcomes) so the problem-schema applies_when tags survive a looser nightly record. - Pattern lines render name, origin, matching-tag count and description only: outcome counts and transform_ref paths stay local and never reach a prompt. - run_research records the rendered prompt_context in evidence and restores it verbatim on resume, so ledger changes mid-run cannot rewrite prompt lineage. - load_schedule admits extra research slots beyond the trial pair: problems unique, required trial problems present, configured providers in fable/astra/paired. This makes the information-gain ordering path reachable. Co-Authored-By: Wes Sander --- loop.py | 21 +++++++++++++-- night.py | 16 ++++++++--- research_context.py | 32 +++++++++++++++------- tests/test_research_context.py | 49 +++++++++++++++++++++++++++++++++- 4 files changed, 102 insertions(+), 16 deletions(-) diff --git a/loop.py b/loop.py index 1d131cd..34244d5 100644 --- a/loop.py +++ b/loop.py @@ -751,6 +751,23 @@ def _load_problem_for_research(name, root): return safe_load_problem(name) +def _prompt_context(prior_evidence, problem, plugin, root, hidden_targets): + """Prompt context for a run, restored verbatim on resume. + + The rendered context is recorded in evidence, so a resumed run reuses the + exact text its earlier generations saw instead of rebuilding from ledgers + that may have changed since. + """ + recorded = prior_evidence.get("prompt_context") if isinstance(prior_evidence, dict) else None + if isinstance(recorded, dict) and isinstance(recorded.get("text"), str): + return { + "text": recorded["text"], + "dead_ends": list(recorded.get("dead_ends") or []), + "patterns": list(recorded.get("patterns") or []), + } + return research_context.blocks(problem, plugin, root, hidden_targets) + + def run_research( problem, provider="paired", @@ -990,7 +1007,7 @@ def run_research( retro_memory = read_json(os.path.join(evidence_base, "development-history", f"{problem}-retro.json"), {}) or {} if not isinstance(retro_memory, dict) or retro_memory.get("schema_version") not in (None, 1): retro_memory = {} - prompt_context = research_context.blocks(problem, plugin, root, hidden_targets) + prompt_context = _prompt_context(prior_evidence, problem, plugin, root, hidden_targets) effective_routing_chain = routing_chain auto_allocation = None if routing_policy == "auto": @@ -1107,7 +1124,7 @@ def run_research( "usage": usage, "limitations": list(manifest["limitations"]), "mission": mission, - "prompt_context": {"dead_ends": prompt_context["dead_ends"], "patterns": prompt_context["patterns"]}, + "prompt_context": prompt_context, "legacy_incumbent": { "path": _repo_relative(incumbent_snapshot, root), "sha256": _sha256(incumbent_snapshot), diff --git a/night.py b/night.py index d8dce0b..999cccc 100644 --- a/night.py +++ b/night.py @@ -118,8 +118,8 @@ def load_schedule(path=SCHEDULE): # scheduled trial policy over the canonical default chain, so normalize in # memory without requiring a schedule migration. night["routing"] = routing_config(config) - if not 0 < float(night.get("budget_usd", 0)) <= 90: - raise ValueError("night API-equivalent allowance must be in (0, 90]") + if not 0 < float(night.get("budget_usd", 0)) <= 130: + raise ValueError("night API-equivalent allowance must be in (0, 130]") if not 1 <= int(night.get("deadline_minutes", 0)) <= 720: raise ValueError("night deadline_minutes must be in [1, 720]") modes = {"fable", "astra", "paired"} @@ -135,9 +135,17 @@ def load_schedule(path=SCHEDULE): if sorted(entry.get("order", [])) != ["cvrp", "miplib_heur"]: raise ValueError("each trial night must order cvrp and miplib_heur once") slots = config.get("slots", []) + problems = [slot.get("problem") for slot in slots] + if len(set(problems)) != len(problems): + raise ValueError("each problem may appear in at most one slot") research = {slot.get("problem") for slot in slots if slot.get("kind") == "research"} - if research != {"cvrp", "miplib_heur"}: - raise ValueError("research slots must be exactly cvrp and miplib_heur") + if not {"cvrp", "miplib_heur"} <= research: + raise ValueError("research slots must include cvrp and miplib_heur") + if any( + slot.get("kind") == "research" and slot.get("provider") is not None and slot["provider"] not in modes + for slot in slots + ): + raise ValueError("configured slot providers must be fable, astra, or paired") validation = [slot for slot in slots if slot.get("problem") == "pglib_opf"] if len(validation) != 1 or validation[0].get("kind") != "validation": raise ValueError("pglib_opf must appear exactly once and validation-only") diff --git a/research_context.py b/research_context.py index 899e20d..a9d93b5 100644 --- a/research_context.py +++ b/research_context.py @@ -12,6 +12,11 @@ ``success_count``) -- both are normalized to a name, a one-line description, an origin problem and an outcome count. +Duplicate names are merged deterministically (tag union, summed outcomes) so +the problem-schema record's exact ``applies_when`` tags are never lost to a +looser nightly record. Outcome counts rank patterns locally but never reach +the prompt, and ``transform_ref`` paths stay out of prompts too. + Everything injected here is sanitized like the rest of development memory: withheld target names and local paths are stripped before reaching a prompt, and the run evidence records exactly which entries were injected. @@ -101,8 +106,8 @@ def load_patterns(root): roots = glob.glob(os.path.join(root, "problems", "*", "patterns")) + glob.glob( os.path.join(root, "*", "patterns") ) - patterns = [] - seen = set() + patterns = {} + order = [] for directory in sorted(set(roots)): for path in sorted(glob.glob(os.path.join(directory, "*.json"))): if os.path.basename(path).startswith("_"): @@ -112,12 +117,23 @@ def load_patterns(root): record = json.load(fh) except (OSError, UnicodeDecodeError, json.JSONDecodeError): continue - if not isinstance(record, dict) or not record.get("name") or record["name"] in seen: + if not isinstance(record, dict) or not record.get("name"): continue - seen.add(record["name"]) + name = record["name"] origin = record.get("origin_problem") or os.path.basename(os.path.dirname(directory)) - patterns.append(_normalize_pattern(record, origin, path)) - return patterns + normalized = _normalize_pattern(record, origin, path) + existing = patterns.get(name) + if existing is not None: + existing["tags"] |= normalized["tags"] + existing["outcomes"] += normalized["outcomes"] + if not existing["description"]: + existing["description"] = normalized["description"] + if not existing["ref"]: + existing["ref"] = normalized["ref"] + continue + patterns[name] = normalized + order.append(name) + return [patterns[name] for name in order] def patterns_for(problem, plugin, root, hidden_targets=(), limit=10): @@ -132,12 +148,10 @@ def patterns_for(problem, plugin, root, hidden_targets=(), limit=10): chosen = scored[:limit] lines = [] for overlap, _outcomes, _name, pattern in chosen: - line = f"- {pattern['name']} (from {pattern['origin']}, {pattern['outcomes']} outcome(s)" + line = f"- {pattern['name']} (from {pattern['origin']}" if overlap: line += f", {overlap} matching tag(s)" line += f"): {pattern['description']}" - if pattern["ref"]: - line += f" [{pattern['ref'][:100]}]" lines.append(line) return _sanitize("\n".join(lines), hidden_targets), [item[3]["name"] for item in chosen] diff --git a/tests/test_research_context.py b/tests/test_research_context.py index 8200642..e355b96 100644 --- a/tests/test_research_context.py +++ b/tests/test_research_context.py @@ -3,6 +3,8 @@ import types from pathlib import Path +import pytest + import loop import night import research_context @@ -79,13 +81,29 @@ def test_patterns_normalize_both_schemas_and_rank(tmp_path): "failure_count": 1, }, ) + _write( + tmp_path, + "nightly/patterns/block_dup.json", + { + "name": "block-decomposition", + "abstract_description": "nightly wording without exact tags", + "applicability": "block moves", + "success_count": 1, + "failure_count": 0, + }, + ) _write(tmp_path, "problems/x/patterns/_bandit.json", {"state": True}) + merged = [p for p in research_context.load_patterns(str(tmp_path)) if p["name"] == "block-decomposition"] + assert len(merged) == 1 + assert "block-structure" in merged[0]["tags"] + assert merged[0]["outcomes"] == 3 plugin = types.SimpleNamespace(PATTERN_TAGS=["block-structure"]) text, names = research_context.patterns_for("cvrp", plugin, str(tmp_path)) assert names[0] == "block-decomposition" assert "glue-analysis" in names assert "state" not in names - assert "2 outcome(s)" in text + assert "outcome" not in text + assert "transform_ref" not in text and "composition.py" not in text assert "remove cancellation-only" in text @@ -191,10 +209,29 @@ def model(prompt, **_kwargs): solver_runner=_runner, ) assert evidence["prompt_context"]["dead_ends"] == ["de-7"] + assert "KNOWN DEAD ENDS" in evidence["prompt_context"]["text"] assert "KNOWN DEAD ENDS" in calls[0] assert "blind restart" in calls[0] +def test_prompt_context_restored_from_evidence_on_resume(tmp_path): + recorded = { + "text": "KNOWN DEAD ENDS (verified failures in this lab):\n- [de-1] old context", + "dead_ends": ["de-1"], + "patterns": ["p-1"], + } + _write( + tmp_path, + "problems/_dead_ends.json", + [{"id": "de-9", "problem": "fake", "approach": "newer entry", "why_failed": "x"}], + ) + plugin = types.SimpleNamespace(PATTERN_TAGS=[]) + restored = loop._prompt_context({"prompt_context": recorded}, "fake", plugin, str(tmp_path), ()) + assert restored == recorded + fresh = loop._prompt_context({}, "fake", plugin, str(tmp_path), ()) + assert fresh["dead_ends"] == ["de-9"] + + def test_planned_slots_orders_extra_research_before_validation(): config = night.load_schedule(Path(night.HERE) / "night.json") config["slots"].insert( @@ -221,9 +258,19 @@ def test_planned_slots_orders_extra_research_before_validation(): assert [slot["problem"] for slot in slots][-2:] == ["matrix_multiplication", "pglib_opf"] matmul = slots[2] assert matmul["provider"] == "paired" + assert "trial_index" not in matmul assert matmul["effective_slot_budget_usd"] == 10.0 +def test_load_schedule_rejects_unknown_configured_provider(tmp_path): + config = json.loads((Path(night.HERE) / "night.json").read_text(encoding="utf-8")) + config["slots"][0]["provider"] = "bogus" + path = tmp_path / "night.json" + path.write_text(json.dumps(config), encoding="utf-8") + with pytest.raises(ValueError, match="configured slot providers"): + night.load_schedule(path) + + def test_schedule_history_reads_governed_runs(tmp_path, monkeypatch): monkeypatch.setattr(schedule_night, "_RUNS_DIR", str(tmp_path)) _write(