diff --git a/skills/evaluator-patterns/SKILL.md b/skills/evaluator-patterns/SKILL.md index cfe2891..edc57eb 100644 --- a/skills/evaluator-patterns/SKILL.md +++ b/skills/evaluator-patterns/SKILL.md @@ -11,6 +11,13 @@ Use this skill to generate evaluator scripts that follow the optimize-anything c - Read one JSON object from stdin: `{"candidate": "..."}` - Write one JSON object to stdout on a single line: `{"score": , ...diagnostics...}` - Return a numeric `score` (recommended in `0.0..1.0`) +- **Preflight guard** (recommended): detect `"__optimize_anything_preflight__"` candidate and fast-return + ```python + candidate = str(payload.get("candidate", "")) + if candidate == "__optimize_anything_preflight__": + print(json.dumps({"score": 0.5}, separators=(",", ":"))) + return 0 + ``` --- @@ -36,6 +43,11 @@ def main() -> int: payload = json.load(sys.stdin) candidate = str(payload.get("candidate", "")) + # Preflight guard for optimize-anything CLI + if candidate == "__optimize_anything_preflight__": + print(json.dumps({"score": 0.5}, separators=(",", ":"))) + return 0 + dimensions = [ {"name": "clarity", "weight": 0.35, "guide": "Clear, specific, easy to follow"}, {"name": "constraint_following", "weight": 0.35, "guide": "Respects constraints and boundaries"}, @@ -130,6 +142,12 @@ set -euo pipefail payload="$(cat)" candidate="$(printf '%s' "$payload" | python3 -c 'import json,sys; print(json.load(sys.stdin).get("candidate",""))')" +# Preflight guard for optimize-anything CLI +if [ "$candidate" = "__optimize_anything_preflight__" ]; then + printf '{"score":0.5}\n' + exit 0 +fi + workdir="$(mktemp -d)" trap 'rm -rf "$workdir"' EXIT @@ -222,6 +240,12 @@ def clamp01(x: float) -> float: def main() -> int: payload = json.load(sys.stdin) candidate = str(payload.get("candidate", "")) + + # Preflight guard for optimize-anything CLI + if candidate == "__optimize_anything_preflight__": + print(json.dumps({"score": 0.5}, separators=(",", ":"))) + return 0 + text = candidate.strip() words = re.findall(r"\w+", text) @@ -286,6 +310,11 @@ def main() -> int: payload = json.load(sys.stdin) candidate = str(payload.get("candidate", "")).lower() + # Preflight guard for optimize-anything CLI (check raw payload, not lowercased) + if payload.get("candidate", "") == "__optimize_anything_preflight__": + print(json.dumps({"score": 0.5}, separators=(",", ":"))) + return 0 + scenarios = { "ambiguous_request": ["clarifying question", "assumption"], "tool_failure": ["retry", "fallback", "error message"], diff --git a/skills/generate-evaluator/SKILL.md b/skills/generate-evaluator/SKILL.md index dae7e77..7ec2e1f 100644 --- a/skills/generate-evaluator/SKILL.md +++ b/skills/generate-evaluator/SKILL.md @@ -14,6 +14,13 @@ Generate an evaluator that scores candidate artifacts for optimization with gepa - Default payload: `{"candidate": ""}` - Dataset-aware payload (`--dataset`): `{"candidate": "", "example": {...}}` - Output JSON must include `score` (float, usually in `[0,1]`), plus optional side-info fields. +- **Preflight detection** (command evaluators only): The CLI sends `"__optimize_anything_preflight__"` as the candidate text before optimization starts. Your evaluator should detect this and return immediately: + ```python + if candidate == "__optimize_anything_preflight__": + print(json.dumps({"score": 0.5})) + sys.exit(0) + ``` + This avoids the 10-second preflight timeout for evaluators that make slow API calls. ## Choose an Evaluator Pattern @@ -83,3 +90,4 @@ echo '{"candidate":"text","example":{"input":"q","expected":"a"}}' | python3 eva 4. Customize scoring logic and side-info fields. 5. Test with stdin payloads. You should see JSON with `score` plus diagnostic fields. 6. Validate score range: a good seed should score between 0.3-0.7. If above 0.85, the evaluator lacks discrimination. +7. Test preflight: `echo '{"candidate":"__optimize_anything_preflight__"}' | python3 your_evaluator.py` — should return `{"score": 0.5}` instantly. diff --git a/skills/optimization-guide/SKILL.md b/skills/optimization-guide/SKILL.md index a798851..9073ad5 100644 --- a/skills/optimization-guide/SKILL.md +++ b/skills/optimization-guide/SKILL.md @@ -17,6 +17,16 @@ Start with your current best version of the artifact. `gepa` evolves from here. ### 2. Create an Evaluator Use the **generate-evaluator** skill to create one matched to your objective. The evaluator is the most critical piece—`gepa`'s optimization quality is bounded by your evaluator's feedback quality. +### 2b. Choose Your Evaluator Interface + +1. Use the **Python API** for in-process Python evaluators. Pass a function that returns a score or `(score, diagnostics)`. +2. Use `--evaluator-command` for standalone scripts or binaries. Read `{"candidate":"..."}` from stdin and write `{"score":0.5}` to stdout. +3. Use `--evaluator-url` for remote services that accept request JSON and return score JSON. + +Prefer the Python API. For command templates, use the **generate-evaluator** and **evaluator-patterns** skills. + +**Command preflight and timeouts:** Before optimization, the CLI sends `{"_protocol_version":2,"candidate":"__optimize_anything_preflight__"}` and waits 10 seconds. Detect that sentinel and immediately return `{"score":0.5}`. Normal command evaluations time out after 30 seconds; use the Python API for slower work. + ### 3. Choose Optimization Mode **Single-task** (no dataset) — optimize one artifact against one evaluator: @@ -144,4 +154,4 @@ The result contains: 3. Clarify the objective: Set the `objective` string that is injected into `gepa`'s reflection prompt and specify constraints like token limits or format requirements. 4. Add background context: Use `background` for domain knowledge, constraints, or strategies such as "Target audience is non-technical users. Never use jargon." 5. Iterate on the evaluator: Improve the evaluator before increasing `budget` if optimization results on `seed.txt` are poor. -6. Set evaluator working directory: Pass `evaluator_cwd` as an absolute project path next to `seed.txt` and `evaluators/eval.sh` when `evaluators/eval.sh` or other evaluator commands use repo-relative files or scripts. \ No newline at end of file +6. Set evaluator working directory: Pass `evaluator_cwd` as an absolute project path next to `seed.txt` and `evaluators/eval.sh` when `evaluators/eval.sh` or other evaluator commands use repo-relative files or scripts. diff --git a/tests/test_doc_contract.py b/tests/test_doc_contract.py index c0db1ef..463bbca 100644 --- a/tests/test_doc_contract.py +++ b/tests/test_doc_contract.py @@ -129,6 +129,22 @@ def test_all_command_markdown_files_have_name_and_description_frontmatter(): ) +def test_bash_evaluator_template_fast_returns_for_preflight(): + skill_text = _read_text(Path("skills/evaluator-patterns/SKILL.md")) + pattern_2 = skill_text.split("## Pattern 2:", maxsplit=1)[1].split( + "## Pattern 3:", maxsplit=1 + )[0] + guard = ( + 'if [ "$candidate" = "__optimize_anything_preflight__" ]; then\n' + " printf '{\"score\":0.5}\\n'\n" + " exit 0\n" + "fi" + ) + + assert guard in pattern_2 + assert pattern_2.index(guard) < pattern_2.index('workdir="$(mktemp -d)"') + + def test_concepts_mentions_multi_task_and_generalization(): concepts_text = _read_text(Path("CONCEPTS.md")) _assert_contains_terms(