From 98daf93d4c8bb4b2d8feb6b2688ef5eadf88ab1f Mon Sep 17 00:00:00 2001 From: Ahmad Ragab Date: Sat, 1 Aug 2026 09:35:47 -0700 Subject: [PATCH 1/2] docs: design subscription-backed LLM integrations --- ...dex-claude-subscription-backends-design.md | 576 ++++++++++++++++++ 1 file changed, 576 insertions(+) create mode 100644 docs/superpowers/specs/2026-08-01-codex-claude-subscription-backends-design.md diff --git a/docs/superpowers/specs/2026-08-01-codex-claude-subscription-backends-design.md b/docs/superpowers/specs/2026-08-01-codex-claude-subscription-backends-design.md new file mode 100644 index 0000000..0d11c39 --- /dev/null +++ b/docs/superpowers/specs/2026-08-01-codex-claude-subscription-backends-design.md @@ -0,0 +1,576 @@ +# Codex and Claude Subscription Backends Design + +**Status:** Approved design, ready for implementation planning + +**Date:** 2026-08-01 + +**Scope:** Local `optimize-anything` CLI, Python runtime, generated evaluators, and packaged coding-agent skills + +## Summary + +Add a provider-neutral completion layer so LLM work can run through one of three backends: + +- the existing LiteLLM/API-key path; +- a local Codex subscription through the official Codex Python SDK; or +- a local Claude subscription through the installed Claude Code CLI. + +Direct CLI and Python use remain API-backed by default. Packaged skills instead select the subscription belonging to their host coding agent: Codex-hosted skills select Codex, and Claude-hosted skills select Claude. This selection is explicit in the skill's generated command, not inferred by the core CLI from ambient environment variables. + +A subscription backend covers every built-in LLM role in a run: GEPA proposal/reflection, built-in judging, dimension analysis, scoring, and validation. Deterministic command and HTTP evaluators remain unchanged. Subscription calls are serialized per provider by default. On an eligible subscription failure, the run automatically switches that role to the corresponding vendor API if a usable key and fallback model are available, while warning immediately and recording the billing/provenance change. + +The deliberately asymmetric implementation is: + +- Codex uses the purpose-built `openai-codex` Python SDK and reuses the user's saved Codex authentication. +- Claude uses a tightly sandboxed `claude -p` subprocess because Anthropic's Agent SDK does not permit third-party products to offer Claude.ai subscription login. The Claude bridge is therefore local-only, opt-in, experimental, and carries an explicit policy caveat. + +## Context + +Today, proposer selection resolves to a model string and passes that string to GEPA's `ReflectionConfig.reflection_lm`. Built-in judging and dimension analysis call LiteLLM directly. CLI configuration exposes model names and API base URLs, and generated judge/composite evaluators also call LiteLLM directly. + +That arrangement assumes every LLM operation is an API request. It misses a useful local workflow: the person invoking an optimization skill may already have an authenticated Codex or Claude Code subscription in the host agent. It also means subscription support cannot be added only to the proposer; doing so would leave judging, scoring, and analysis dependent on API keys. + +GEPA already accepts a callable reflection LM, so the proposer does not need a GEPA fork. The new completion layer can supply that callable while also replacing the direct LiteLLM calls in the built-in evaluator paths. + +## Research conclusions + +### Codex + +OpenAI documents both ChatGPT subscription login and API-key login for Codex. Non-interactive `codex exec` normally reuses saved CLI authentication, and the official Python SDK likewise reuses existing Codex authentication. The SDK exposes the app-server/thread model directly and is a better fit than parsing CLI output for an embedded Python application. + +Relevant official documentation: + +- [Codex authentication](https://developers.openai.com/codex/auth/) +- [Codex non-interactive mode](https://developers.openai.com/codex/noninteractive/) +- [Codex SDK](https://developers.openai.com/codex/sdk/) +- [Codex Python SDK README](https://github.com/openai/codex/blob/main/sdk/python/README.md) +- [Codex Python SDK API reference](https://github.com/openai/codex/blob/main/sdk/python/docs/api-reference.md) + +The adapter must reuse an existing ChatGPT login. It must not initiate login, accept tokens, copy credentials, or silently use an API-key-authenticated Codex session when the user requested a subscription. + +### Claude + +Anthropic documents `claude -p` for non-interactive use, JSON Schema output, `claude auth status`, and `--safe-mode`. Safe mode is important because it disables project instructions, skills, plugins, hooks, MCP servers, and other customizations while preserving authentication. The CLI also supports disabling tools and session persistence. + +Anthropic's authentication precedence gives API tokens, API keys, and cloud-provider credentials priority over subscription login. A Claude subscription adapter must therefore perform auth preflight and execution with those overrides removed; otherwise a run advertised as subscription-backed could incur API or cloud charges. + +Relevant official documentation: + +- [Claude Code headless/programmatic mode](https://code.claude.com/docs/en/headless) +- [Claude Code CLI reference](https://code.claude.com/docs/en/cli-reference) +- [Claude Code authentication](https://code.claude.com/docs/en/authentication) +- [Claude Code environment variables](https://code.claude.com/docs/en/env-vars) +- [Claude Code legal and compliance](https://code.claude.com/docs/en/legal-and-compliance) +- [Claude Agent SDK overview](https://code.claude.com/docs/en/agent-sdk/overview) + +Anthropic states that third-party developers should use API keys and may not offer Claude.ai login or route requests through Free, Pro, or Max credentials on behalf of users without approval. This design does not claim such approval. It limits the integration to a local subprocess chosen by the logged-in user, never handles login or credentials, labels the feature experimental, and excludes hosted/service use. That reduces credential and delegation risk but does not eliminate the policy risk; distribution documentation must disclose it plainly. + +### GEPA + +GEPA's reflection configuration accepts a callable language model. Its documentation also describes using a Claude CLI callable as a proposer. The backend abstraction can therefore present `complete_text(request)` as the GEPA reflection callable without modifying GEPA. + +- [GEPA ReflectionConfig](https://gepa-ai.github.io/gepa/api/optimize_anything/ReflectionConfig/) +- [GEPA FAQ](https://gepa-ai.github.io/gepa/guides/faq/) +- [Using Claude Code as a proposer](https://gepa-ai.github.io/gepa/guides/claude-cli-as-proposer/) + +## Goals + +- Let local users run all built-in LLM roles with an existing Codex or Claude subscription. +- Preserve existing API-backed behavior for direct CLI and Python callers. +- Make coding-agent skills default to their host agent's subscription. +- Keep credentials outside `optimize-anything` and prove which auth class actually handled each call. +- Provide consistent structured-output, timeout, error, fallback, concurrency, cache, and provenance behavior across backends. +- Keep deterministic command/HTTP evaluators and their concurrency behavior unchanged. +- Make generated judge/composite evaluators use the same installed runtime rather than embedding a separate API-only implementation. + +## Non-goals + +- Implementing OAuth, device-code login, logout, token refresh, or credential storage. +- Supporting subscriptions from a hosted service, CI runner, shared daemon, or remote multi-user process. +- Making subscription execution the default for ordinary CLI or Python calls. +- Normalizing subscription limits into API token budgets or promising equivalent model availability across auth classes. +- Giving Codex or Claude access to the repository, MCP servers, shell tools, network tools, skills, hooks, or host-agent conversation state. +- Changing command/HTTP evaluator transport or parallelism. +- Preserving standalone portability for generated LLM judge/composite scripts; they will require the installed `optimize-anything` runtime. + +## Design decisions + +1. Use one completion contract and three adapters: LiteLLM API, Codex SDK, and Claude CLI. +2. Use the Codex Python SDK rather than `codex exec`; the SDK is purpose-built for embedding and exposes account, thread, sandbox, and structured-result concepts without CLI parsing. +3. Use `claude -p`, not the Claude Agent SDK, for the local Claude experiment. Do not expose or automate Claude login. +4. Keep direct invocation API-first. Host-agent behavior lives in skill instructions that pass explicit backend flags. +5. Use a selected subscription backend for every built-in LLM role, not only GEPA proposal. +6. Serialize subscription calls per provider by default. Allow an explicit concurrency override, with a warning, for advanced users. +7. Automatically fail over only for a narrow set of eligible failures. Once a role fails over, keep that role on API for the rest of the run. +8. Do not fail over on timeout because the subscription request may still have consumed usage, creating duplicate work and billing. +9. Route generated LLM evaluators through a shared installed runtime. Keep non-LLM verification and simulation templates self-contained. + +## Architecture + +### Package layout + +The implementation should introduce a focused backend package rather than adding provider conditionals to CLI and judge modules: + +```text +src/optimize_anything/ + llm_backends/ + __init__.py + base.py # contracts, capabilities, typed errors + litellm_backend.py # existing API path + codex_backend.py # optional Codex SDK adapter + claude_backend.py # local Claude CLI adapter + fallback.py # eligible-error circuit breaker + factory.py # config resolution and preflight + provenance.py # call and run summaries + evaluator_runtime.py # stable entry point for generated LLM evaluators +``` + +`cli_optimize.py`, `llm_judge.py`, validation, score, and analysis code should depend on the completion contract or factory, never on a concrete subscription adapter. LiteLLM imports should move behind `LiteLLMBackend`, apart from compatibility shims being removed in the same change. + +### Completion contract + +The central contract is deliberately stateless. A backend receives a fully assembled prompt and returns one completion; it does not inherit the host conversation or retain provider sessions between calls. + +```python +@dataclass(frozen=True) +class CompletionRequest: + prompt: str + role: Literal["proposer", "judge", "analysis", "score", "validation"] + model: str | None = None + output_schema: Mapping[str, object] | None = None + timeout_seconds: float | None = None + sampling: SamplingOptions | None = None + +@dataclass(frozen=True) +class CompletionResult: + text: str + structured: JsonValue | None + requested_backend: str + actual_backend: str + requested_model: str | None + actual_model: str | None + auth_class: Literal["api", "subscription"] + auth_source: Literal[ + "chatgpt", "claude_subscription", "openai_api", "anthropic_api", "other_api" + ] + usage: Usage | None + fallback: FallbackRecord | None + +class CompletionBackend(Protocol): + def preflight(self) -> BackendStatus: ... + def complete(self, request: CompletionRequest) -> CompletionResult: ... +``` + +`JsonValue` covers any JSON Schema result, although the initial built-in roles use objects. `Usage` contains provider-reported input, output, and total tokens when available. Cost remains nullable because subscription calls do not have a meaningful per-call API price. `FallbackRecord` contains the original backend, normalized failure category, and switch timestamp, but never raw credential-bearing provider payloads. `auth_source` identifies only the billing/auth mechanism, never an account or credential. + +Each adapter publishes capabilities for structured output, model override, sampling controls, cancellation, and usage reporting. Unsupported requested controls cause `ConfigurationError` before dispatch; adapters must not silently ignore them. The initial optimization integration needs plain text and JSON Schema output. Chat history and multi-turn provider sessions are intentionally absent. + +### Backend specification + +Resolved configuration becomes an immutable role-specific `BackendSpec`: + +```python +@dataclass(frozen=True) +class BackendSpec: + backend: Literal["api", "codex", "claude"] + model: str | None + api_base: str | None + api_fallback: bool + api_fallback_model: str | None + max_concurrency: int +``` + +The factory validates the whole run once, resolves fallback models, performs auth/capability preflight, and returns shared backend instances. It does not preflight on every candidate evaluation. + +When fallback is enabled, the factory returns a `FallbackBackend` that wraps the subscription primary and corresponding `LiteLLMBackend`. The wrapper owns role circuits, eligible-error classification, billing warnings, and requested-versus-actual provenance; concrete provider adapters do not decide whether to switch transports. + +### LiteLLM API backend + +`LiteLLMBackend` preserves current behavior: + +- it accepts existing provider/model strings and `api_base`; +- it calls LiteLLM for text and structured completions; +- it maps LiteLLM/provider failures to the shared typed errors; +- it reports `auth_class="api"`; and +- it remains the default when no backend is specified by a direct caller. + +No existing API key is copied into a `BackendSpec`, result, log, or cache key. LiteLLM continues to discover credentials through its supported environment/config mechanisms. + +### Codex SDK backend + +`CodexSdkBackend` is installed through the optional extra: + +```text +optimize-anything[codex] +``` + +The extra pins `openai-codex` to the tested compatible range in the project lockfile. The release gate must verify the required account-status, ephemeral-thread, sandbox, cancellation, and structured-output APIs against that range before enabling the adapter. + +Preflight uses the SDK's account-status surface and succeeds only when the saved session is authenticated through ChatGPT. An API-key-authenticated Codex session does not count as a subscription backend. The adapter reports a direct remediation command such as `codex login`, but never runs it. + +Every request uses a fresh thread in a newly created private temporary workspace. The thread configuration must enforce: + +- read-only filesystem access to the empty temporary workspace; +- no repository working directory or inherited project instructions; +- no MCP servers; +- no shell, file mutation, or other agent tools; +- no model-accessible network tools; +- no user/project configuration except the minimum needed to reuse authentication; and +- no session reuse after the completion. + +The SDK/app-server transport may reach OpenAI; “no network tools” means the model cannot browse or initiate arbitrary network access. If the tested SDK cannot enforce every isolation property, the Codex adapter is release-blocked rather than weakened silently. + +The prompt is submitted through the SDK request body, never a process argument. JSON Schema requests use the SDK's structured-output support and are validated again locally before returning. + +### Claude CLI backend + +`ClaudeCliBackend` depends on an external `claude` executable and introduces no Python package extra. Preflight checks the executable's version/capabilities and runs `claude auth status` under the same scrubbed environment used for completions. It succeeds only for a logged-in Claude subscription, not an API token, API key, Bedrock, Vertex, or Foundry configuration. + +The subprocess shape is capability-equivalent to: + +```text +claude -p \ + --safe-mode \ + --tools "" \ + --disable-slash-commands \ + --strict-mcp-config \ + --mcp-config \ + --no-session-persistence \ + --max-turns 1 \ + --output-format json \ + --json-schema +``` + +The prompt is written to stdin, never appended to the argument vector. Anthropic's documented interface requires the JSON Schema value as the `--json-schema` argument. Before dispatch, the adapter removes non-structural annotations such as `title`, `description`, `$comment`, `examples`, and `default`, and rejects schemas containing non-static external references. This keeps objective, candidate, and other user artifact text out of the process list while retaining validation keywords. The original schema is still applied locally to the returned value. The empty MCP file is created with user-only permissions in a private temporary directory and removed after the call. stdout and stderr are captured separately with bounded size. + +The child environment removes the documented API/cloud authentication overrides and the `CLAUDECODE` parent-agent marker. At minimum this covers Anthropic API/auth variables, Claude OAuth-token overrides, Bedrock/AWS selectors and credentials, Vertex/Google selectors and credentials, Foundry/Azure selectors and credentials, and any configured API-key helper. The denylist is maintained beside the adapter with tests derived from Anthropic's authentication precedence documentation. The parent environment remains intact so an API fallback can still use its API key after the subscription subprocess exits. + +`--bare` must not be used because it bypasses the credential stores needed for subscription authentication. Safe mode plus explicit tool/MCP/session restrictions provides isolation while preserving the user's saved login. + +The implementation publishes a tested minimum Claude CLI version whose documented flags satisfy this contract. A missing flag or an auth-status response that cannot distinguish subscription from paid API/cloud auth is `BackendUnavailable`, not a reason to run with weaker isolation. + +### GEPA proposer integration + +GEPA receives a callable instead of a model string when the resolved proposer backend is Codex or Claude. The callable builds a `CompletionRequest(role="proposer")`, calls the shared backend, and returns the result text expected by GEPA. + +The API path also uses the callable backed by `LiteLLMBackend`, so error mapping and provenance are uniform across proposer backends. Regression tests must prove that this internal plumbing change preserves existing GEPA proposal behavior and defaults. + +### Built-in judge, analysis, score, and validation + +Direct `litellm.completion` calls in built-in judging and dimension analysis move behind `CompletionBackend.complete`. JSON-producing operations provide a schema and reject invalid local validation rather than trying to recover provider-specific output in each caller. + +The same mechanism covers: + +- optimize's built-in LLM judge; +- `score`; +- `analyze` and dimension discovery; +- analysis performed while generating an evaluator; and +- each LLM provider entry used by `validate`. + +The role field remains distinct even when multiple roles share one configured judge backend. This preserves independent fallback circuits and useful provenance. + +### Generated evaluator runtime + +Judge and composite evaluator templates stop importing LiteLLM or embedding provider calls. They become thin wrappers around a stable installed entry point in `optimize_anything.evaluator_runtime`. + +The wrapper still obeys the existing evaluator JSON-lines contract: it reads an input containing `candidate` and writes an object containing numeric `score`. It passes template configuration and the candidate to the shared runtime, which resolves the configured backend and performs structured judging. + +Consequences: + +- generated LLM evaluators require a compatible installed `optimize-anything` package; +- their generated metadata records that runtime requirement and minimum compatible contract version; +- deterministic verification/simulation templates remain standalone; and +- preflight probes must remain local and must not trigger an LLM call. + +## Configuration and selection + +### CLI + +The optimize command gains explicit role-specific backend flags: + +```text +optimize-anything optimize seed.txt \ + --proposer-backend codex \ + --judge-backend codex \ + --objective "Improve clarity" +``` + +Rules: + +- `--proposer-backend` and `--judge-backend` accept `api`, `codex`, or `claude`. +- Omitted backend flags resolve to `api` for direct CLI calls. +- Existing `--model`, `--judge-model`, and `--api-base` behavior remains unchanged for API backends. +- A subscription model may be omitted to use the coding agent's default model. An explicit model is passed through only if the adapter reports model-override support. +- Selecting `--judge-backend` selects the built-in judge and remains mutually exclusive with command and HTTP evaluator options. +- Commands that already expose a single LLM role gain the corresponding backend flag: `score` uses `--judge-backend`; `analyze` and evaluator analysis use `--analysis-backend` while retaining their existing model flags. +- All subscription-using commands expose `--subscription-concurrency`, default `1`. Values above `1` emit a warning and are applied per provider, not to command/HTTP evaluators. +- `--no-api-fallback` disables automatic fallback. Fallback is otherwise enabled for explicit subscription selection when a corresponding API key and resolvable fallback model exist. +- `--openai-api-fallback-model` and `--anthropic-api-fallback-model` override every matching role for that invocation. When proposer and judge need different API fallback models for the same vendor, configure their separate TOML role tables and omit the provider-wide CLI override. +- An explicit `--api-base` is used by an API fallback as well as a primary API backend; the backend plan prints that a custom base is active without printing credentials. + +`validate --providers` continues accepting existing API model strings and additionally accepts reserved subscription selectors: + +```text +codex +codex: +claude +claude: +``` + +Reserved selectors are parsed before LiteLLM model strings. API validation providers keep their current behavior. + +### TOML + +Legacy scalar model configuration remains valid. Structured role tables add backend configuration: + +```toml +[model.proposer] +backend = "codex" +model = "gpt-5.6-sol" # optional for subscriptions +api_fallback = true +api_fallback_model = "openai/gpt-5.6-sol" + +[model.judge] +backend = "codex" +api_fallback = true +api_fallback_model = "openai/gpt-5.6-luna" +``` + +Analysis and score commands use the judge role configuration unless explicitly overridden on their command line. This keeps the configuration model compact while preserving distinct runtime role labels and circuits. + +For a subscription role, omitted `api_fallback` defaults to `true`; for an API role it is ignored. Fallback still requires ready corresponding-vendor credentials and a resolvable model. The CLI's `--no-api-fallback` overrides all role tables for that invocation. + +Resolution precedence is: + +1. command-line role option; +2. structured role table; +3. legacy scalar role value, interpreted as `backend="api"`; +4. existing API default. + +Mixing a legacy scalar and a structured table for the same role is a configuration error with the conflicting keys named in the diagnostic. + +### Python + +Existing Python entry points keep API defaults. Advanced callers may construct a `BackendSpec` through the public backend factory and pass it to supported optimization/judge entry points. Concrete SDK/CLI clients stay internal so the project can evolve adapter details without expanding the compatibility surface. + +### Host-agent skills + +Core code does not guess its host from `CLAUDECODE`, terminal metadata, or other ambient markers. Packaged skill instructions know which agent is executing them and pass explicit flags: + +- in Codex: `--proposer-backend codex --judge-backend codex`; +- in Claude Code: `--proposer-backend claude --judge-backend claude`; +- in an unknown or unsupported host: omit the flags and retain API behavior. + +The skill passes a judge backend only when the workflow actually uses the built-in judge; an external command or HTTP evaluator does not need one. Before execution, the skill announces that it will use the host's logged-in subscription, that calls are serialized, and that eligible failures can switch to billed API usage when a corresponding key is available. The CLI's own preflight remains authoritative. + +## Preflight and runtime flow + +For each distinct requested backend, startup performs one preflight: + +1. Validate configuration and optional dependency/executable availability. +2. Check required structured-output and isolation capabilities. +3. Inspect auth status without reading a credential. +4. Confirm the auth class is the requested subscription. +5. Resolve the corresponding vendor API fallback and test only for the presence of supported credential configuration without retaining or logging its value. +6. Create a provider-wide semaphore and independent role circuits. +7. Print a concise backend plan before any paid or quota-consuming request. + +Preflight does not send a model request. If subscription preflight fails with an eligible error and fallback is ready, the affected role starts directly on API, emits the fallback warning, and records the preflight cause. If no fallback is ready, the command exits with remediation that names the missing executable, extra, login, API key, or fallback model as appropriate. + +During execution, a subscription call acquires its provider semaphore, dispatches one isolated request, validates the result, records provenance, and releases the semaphore in `finally`. Codex and Claude have separate semaphores; command/HTTP evaluator concurrency does not acquire them. + +## Error model and fallback + +All adapters normalize failures into: + +- `BackendUnavailable` +- `AuthenticationError` +- `RateLimitError` +- `QuotaExceeded` +- `Timeout` +- `InvalidResponse` +- `ConfigurationError` + +Fallback behavior is deliberately conservative: + +| Failure | API fallback? | Reason | +|---|---:|---| +| Missing optional SDK or executable | Yes | Subscription backend is unavailable before dispatch | +| Logged out or wrong auth class | Yes | Subscription cannot serve the request | +| Subscription quota exhausted | Yes | API is the configured continuity path | +| Provider rate limit / temporary availability | Yes | Retry would otherwise block the run | +| Timeout or cancellation | No | The original request may still consume usage; avoid duplicate work/billing | +| Invalid request or unsupported control | No | Switching transport will not repair caller/configuration defects | +| Malformed or schema-invalid response | No | Preserve deterministic failure and expose adapter/provider defect | +| Internal programming/configuration error | No | Never hide implementation defects behind paid fallback | + +Codex falls back only to an OpenAI API model; Claude falls back only to an Anthropic API model. Fallback requires both a resolvable API model and a positive, non-secret readiness probe for the corresponding LiteLLM credentials. Initially, readiness means a non-empty canonical `OPENAI_API_KEY` or `ANTHROPIC_API_KEY` is present; future credential resolvers may participate only through a boolean/status interface that does not expose the value. The model may be explicit or come from the project's centralized provider-and-role fallback defaults. If either requirement is absent, fallback is unavailable and the original normalized error is raised with remediation. + +The first eligible failure opens a circuit for that backend and role for the rest of the run. Later calls for that role go directly to API. Proposer, judge, analysis, score, and validation circuits are independent, so a judge failure does not move the proposer. A run that mixes backends is marked `mixed_backend=true`; this is especially visible for judges because a backend change can affect score comparability. + +There is no silent fallback. The first switch produces an immediate stderr warning containing role, source backend, target API model, reason category, and the fact that API billing may apply. Final summaries repeat the switch and counts. + +## Concurrency and cancellation + +Each subscription provider has an in-process semaphore with default capacity one. Every proposer, judge, analysis, score, and validation call through that provider shares it. This avoids bursts against interactive subscription limits and prevents multiple local agent sessions from contending for credential/session state. + +An explicit concurrency override changes the semaphore capacity and emits a warning. It does not change GEPA worker count, command evaluator concurrency, HTTP evaluator concurrency, or cross-provider concurrency. + +On provider timeout, the backend attempts provider-supported cancellation, terminates a Claude child process with a bounded graceful period, cleans private temporary files, and raises `Timeout` without API fallback. A user interruption propagates through the CLI's existing interruption path, also without fallback, after the same cleanup. Cleanup failures are reported without masking the primary error. + +## Security and credential handling + +The feature follows these invariants: + +- Never read credential file contents or parse, copy, return, cache, or log OAuth tokens, API keys, cookies, or authorization headers. Fallback readiness may test only whether a canonical credential environment variable is present and non-empty. +- Inspect only provider-supported account/auth status metadata. +- Never initiate login or open a browser/device-code flow. +- Never put prompts, user artifact text, or secrets on a process command line. Claude receives only a stripped structural schema in its documented `--json-schema` argument. +- Use user-only private temporary directories and deterministic cleanup. +- Bound captured stdout/stderr and redact known credential-shaped values before diagnostics. +- Exclude repository contents, host-agent conversation state, project instructions, skills, plugins, hooks, MCP configuration, and user tools from subscription requests. +- Preserve the minimum environment needed for the provider's saved local authentication while removing overrides that would change the billed auth class. +- Record `subscription` versus `api` auth class, never credential identity. + +Security tests must inspect the actual child argument vector, environment keys, temporary permissions, backend configuration, and produced logs/artifacts. A release is blocked if isolation depends only on prompt instructions. + +## Provenance, summaries, and cache identity + +Every completion records: + +- run and call identifier; +- semantic role; +- requested and actual backend; +- requested and actual model, when the provider reports it; +- auth class (`subscription` or `api`); +- auth source (`chatgpt`, `claude_subscription`, or the resolved API provider class); +- start time and duration; +- provider-reported usage, when available; +- retry count; +- fallback source and normalized reason; and +- schema/prompt contract version. + +Run summaries aggregate calls and usage by backend, model, auth class, and role; list fallback causes; show whether the run mixed backends; and keep cost nullable for subscription calls. The canonical optimize summary contract should gain additive optional fields so existing consumers remain valid. + +Cache fingerprints include: + +- actual backend; +- actual model or a stable `provider-default` marker; +- auth class; +- role; +- prompt/schema contract version; and +- candidate/input identity. + +Fallback output is written under the actual API fingerprint, not the requested subscription fingerprint. Auth class is part of identity because provider defaults, system behavior, and limits can differ even when model names appear equal. + +## Backward compatibility + +- No backend option means API, exactly as today. +- Existing model strings, API bases, environment keys, evaluator command/URL options, and legacy TOML scalar values retain their meaning. +- Existing command/HTTP evaluator JSON contracts and concurrency remain unchanged. +- New summary data is additive and optional. +- Existing generated deterministic evaluators remain standalone. +- Generated LLM evaluator portability changes intentionally and is called out in generated comments, CLI output, migration notes, and release notes. +- The Codex dependency remains optional; ordinary installs do not gain the SDK. +- Missing Claude CLI has no effect unless Claude is selected explicitly or by a host skill. + +## Verification strategy + +### Offline unit and contract tests + +All default project gates remain network-free. Shared backend contract tests use fakes and cover: + +- text and schema completions; +- schema validation and invalid responses; +- timeout and cancellation; +- every normalized error category; +- unsupported capabilities; +- provenance completeness; +- bounded/redacted diagnostics; and +- absence of secrets in results, logs, cache keys, and artifacts. + +Codex adapter tests use a fake SDK/app-server and verify: + +- optional-extra diagnostics; +- account status accepts ChatGPT auth and rejects API-key auth; +- one ephemeral thread and private empty workspace per request; +- read-only sandbox, no tools, no MCP, no project/user instructions, and no model network tools; +- prompt and schema request placement; +- structured result and local schema validation; +- cleanup and cancellation; and +- failure when any isolation capability cannot be enforced. + +Claude adapter tests use a fake executable and verify: + +- exact safe-mode/tool/MCP/session/schema flags; +- prompt arrives through stdin and is absent from argv; +- auth preflight uses the same scrubbed environment as completion; +- API, OAuth override, Bedrock, Vertex, Foundry, and `CLAUDECODE` variables are absent; +- necessary saved-login environment remains available; +- JSON/stdout parsing, bounded stderr, exit-code mapping, timeout termination, and cleanup; and +- missing required CLI capabilities block execution. + +Fallback tests verify: + +- only eligible categories switch; +- missing auth, quota, rate limit, and backend-unavailable paths; +- no fallback on timeout, cancellation, invalid request, invalid response, or configuration error; +- corresponding-vendor key/model requirements; +- an immediate billing warning; +- independent per-role circuits; +- no second subscription attempt after a circuit opens; and +- correct mixed-backend summary and cache identity. + +Concurrency tests prove the default maximum is one active request per subscription provider, an override changes only that provider's semaphore, and command/HTTP evaluators remain unaffected. + +Compatibility tests cover: + +- unchanged CLI defaults and help behavior; +- existing `--model`, `--judge-model`, and API-base paths; +- legacy and structured TOML parsing plus conflict diagnostics; +- reserved `validate --providers` selectors; +- current Python entry-point defaults; +- canonical result-contract compatibility; +- generated judge/composite wrappers contain no direct LiteLLM call; +- generated wrapper/runtime contract versioning; and +- Codex-, Claude-, and unknown-host skill command fixtures. + +### Opt-in live integration gates + +Live gates are separate from `scripts/check.py`, skipped unless explicitly enabled, and never run in normal CI: + +1. `integration_codex_subscription` performs one JSON Schema completion using existing ChatGPT Codex auth, then asserts subscription provenance and isolation evidence. +2. `integration_claude_subscription` performs one safe-mode JSON Schema completion after API/cloud overrides are removed, then asserts subscription provenance and the expected CLI mode. +3. Each provider runs one minimal end-to-end optimization with budget `1` and a deterministic command evaluator, proving the GEPA callable path without introducing an LLM judge variable. +4. API fallback is tested with fakes by default. A paid live fallback requires a separate explicit marker and prints a billing warning before dispatch. + +Live evidence records compatible SDK/CLI versions, operating system, auth class without account identity, requested/actual model, and pass/fail isolation assertions. It must not retain prompts that contain user artifacts unless the user explicitly requests artifact retention. + +### Release gates + +The feature may ship only when: + +- the complete existing test suite and `scripts/check.py` pass offline; +- all backend, fallback, concurrency, compatibility, and skill fixture tests pass; +- no credential, prompt, or user artifact text appears in argv, logs, cache keys, or retained temporary artifacts; +- tested Codex SDK and Claude CLI versions are pinned/documented; +- Codex and Claude isolation capabilities are verified rather than inferred; +- direct CLI/Python behavior is demonstrably backward-compatible; +- subscription calls serialize by default; +- opt-in live gates provide passing evidence on supported platforms; and +- Claude documentation prominently includes the local-only experimental scope and Anthropic policy caveat. + +Failure to enforce isolation, distinguish subscription auth from API/cloud auth, or validate structured output blocks the affected adapter. The implementation must not fall back to a less isolated invocation. + +## Documentation and rollout + +Ship in three layers: + +1. Introduce the completion contract and migrate existing LiteLLM behavior with no user-visible default change. +2. Add Codex behind the optional extra and opt-in flags, then update Codex-hosted skills after live verification. +3. Add Claude as an experimental local adapter and update Claude-hosted skills only after the CLI capability, auth-class, isolation, and policy disclosures are verified. + +Documentation must include installation, auth preflight, backend/model selection, API fallback and billing behavior, concurrency, generated evaluator runtime requirements, troubleshooting, supported versions/platforms, data handling, and removal/disable instructions. Examples must never instruct users to paste tokens into `optimize-anything`. + +## Acceptance criteria + +The design is implemented when a local, already logged-in Codex or Claude user can invoke the corresponding host skill and complete a budget-1 optimization without an API key, including any built-in LLM role selected by that workflow; direct CLI calls still use the API by default; eligible failures switch only to a ready corresponding-vendor API fallback with an immediate billing warning; all actual backend/auth/model choices are visible in the final summary; subscription calls serialize; generated LLM evaluators use the shared runtime; and the isolation, compatibility, offline, and opt-in live gates above pass. From 9ddfe82505f7a7c4c5c1ac80c4c7260e374e0104 Mon Sep 17 00:00:00 2001 From: Ahmad Ragab Date: Tue, 22 Sep 2026 20:15:12 -0700 Subject: [PATCH 2/2] feat: add subscription-backed LLM roles --- README.md | 58 +++- SKILL.md | 13 + commands/analyze.md | 7 + commands/compare.md | 4 + commands/optimize.md | 9 +- commands/quick.md | 10 +- commands/score.md | 4 + commands/validate.md | 4 + install.md | 21 ++ pyproject.toml | 4 + skills/generate-evaluator/SKILL.md | 3 + skills/optimization-guide/SKILL.md | 16 + src/optimize_anything/cli.py | 98 +++++- src/optimize_anything/cli_optimize.py | 171 ++++++++- src/optimize_anything/cli_tools.py | 164 ++++++--- src/optimize_anything/evaluator_generator.py | 88 ++++- src/optimize_anything/evaluator_runtime.py | 187 ++++++++++ .../llm_backends/__init__.py | 39 +++ src/optimize_anything/llm_backends/base.py | 217 ++++++++++++ .../llm_backends/claude_backend.py | 326 ++++++++++++++++++ .../llm_backends/codex_backend.py | 252 ++++++++++++++ .../llm_backends/coordination.py | 247 +++++++++++++ src/optimize_anything/llm_backends/factory.py | 137 ++++++++ .../llm_backends/fallback.py | 170 +++++++++ .../llm_backends/litellm_backend.py | 234 +++++++++++++ .../llm_backends/provenance.py | 76 ++++ src/optimize_anything/llm_backends/schema.py | 53 +++ src/optimize_anything/llm_judge.py | 173 ++++++---- src/optimize_anything/spec_loader.py | 37 +- tests/test_claude_backend.py | 151 ++++++++ tests/test_codex_backend.py | 151 ++++++++ tests/test_evaluator_generator.py | 102 +++++- tests/test_evaluator_runtime.py | 138 ++++++++ tests/test_llm_backend_contract.py | 107 ++++++ tests/test_llm_coordination.py | 64 ++++ tests/test_llm_factory.py | 55 +++ tests/test_llm_fallback.py | 133 +++++++ tests/test_spec_loader.py | 25 ++ tests/test_subscription_live.py | 116 +++++++ uv.lock | 46 ++- 40 files changed, 3746 insertions(+), 164 deletions(-) create mode 100644 src/optimize_anything/evaluator_runtime.py create mode 100644 src/optimize_anything/llm_backends/__init__.py create mode 100644 src/optimize_anything/llm_backends/base.py create mode 100644 src/optimize_anything/llm_backends/claude_backend.py create mode 100644 src/optimize_anything/llm_backends/codex_backend.py create mode 100644 src/optimize_anything/llm_backends/coordination.py create mode 100644 src/optimize_anything/llm_backends/factory.py create mode 100644 src/optimize_anything/llm_backends/fallback.py create mode 100644 src/optimize_anything/llm_backends/litellm_backend.py create mode 100644 src/optimize_anything/llm_backends/provenance.py create mode 100644 src/optimize_anything/llm_backends/schema.py create mode 100644 tests/test_claude_backend.py create mode 100644 tests/test_codex_backend.py create mode 100644 tests/test_evaluator_runtime.py create mode 100644 tests/test_llm_backend_contract.py create mode 100644 tests/test_llm_coordination.py create mode 100644 tests/test_llm_factory.py create mode 100644 tests/test_llm_fallback.py create mode 100644 tests/test_subscription_live.py diff --git a/README.md b/README.md index d92ce97..6d06ff9 100644 --- a/README.md +++ b/README.md @@ -178,6 +178,56 @@ commands, expected report fields, acceptance criteria, and troubleshooting. - `analyze` - `validate` +## Codex and Claude subscription backends + +Local Codex and Claude Code logins can power proposer and built-in evaluator +roles without API keys. Selection is explicit; omitting backend flags preserves +the existing LiteLLM API behavior. + +```bash +# Install the pinned Codex SDK adapter, then authenticate with ChatGPT. +uv sync --extra codex +codex login + +# Codex subscription: proposer plus built-in judge. +optimize-anything optimize seed.txt \ + --proposer-backend codex --judge-backend codex \ + --objective "Improve clarity" + +# Claude Code subscription. Install/sign in to the claude CLI first. +optimize-anything optimize seed.txt \ + --proposer-backend claude --judge-backend claude \ + --objective "Improve clarity" +``` + +Subscription calls are serialized per provider by default. The CLI may switch +an eligible failure to a same-vendor API only when a matching key and fallback +model are available, and prints a billing warning first. Use +`--no-api-fallback` to prohibit that switch. Claude support is local-only and +experimental; neither adapter initiates login or copies credential contents. + +Single-role commands use `--analysis-backend` (`analyze`) or +`--judge-backend` (`score`). `validate --providers` also accepts `codex`, +`codex:`, `claude`, and `claude:`. Structured TOML uses role +tables such as: + +```toml +[model.proposer] +backend = "codex" +api_fallback = false + +[model.judge] +backend = "claude" +api_fallback_model = "anthropic/claude-sonnet-5" +``` + +Opt-in live gates consume local subscription quota: + +```bash +OPTIMIZE_ANYTHING_RUN_SUBSCRIPTION_LIVE=1 \ + uv run pytest tests/test_subscription_live.py +``` + ## Agent Plugins The Claude Code plugin and Codex plugin share the same `skills/` tree and @@ -380,7 +430,7 @@ optimize-anything optimize seed.txt \ ### `optimize` flags (complete) -Exactly one evaluator source is required: `--evaluator-command` OR `--evaluator-url` OR `--judge-model`. +Exactly one evaluator source is required: `--evaluator-command` OR `--evaluator-url` OR a built-in judge selected by `--judge-model`/`--judge-backend`. | Flag | Description | Default | |---|---|---| @@ -397,7 +447,13 @@ Exactly one evaluator source is required: `--evaluator-command` OR `--evaluator- | `--budget ` | Max evaluator calls | `100` | | `--output, -o ` | Write best artifact to file | -- | | `--model ` | Proposer model (or env fallback) | `OPTIMIZE_ANYTHING_MODEL`, then `openai/gpt-5.6-sol` | +| `--proposer-backend api\|codex\|claude` | Proposal backend | `api` | | `--judge-model ` | Built-in LLM judge evaluator model | -- | +| `--judge-backend api\|codex\|claude` | Built-in judge backend | `api` | +| `--subscription-concurrency ` | Concurrent calls allowed per subscription provider | `1` | +| `--no-api-fallback` | Prohibit subscription-to-API fallback | `false` | +| `--openai-api-fallback-model ` | Same-vendor API fallback for Codex | -- | +| `--anthropic-api-fallback-model ` | Same-vendor API fallback for Claude | -- | | `--judge-objective ` | Judge objective override | falls back to `--objective` | | `--api-base ` | Override LiteLLM API base | -- | | `--diff` | Print unified diff (seed vs best) to stderr | `false` | diff --git a/SKILL.md b/SKILL.md index 9500a94..d799e1d 100644 --- a/SKILL.md +++ b/SKILL.md @@ -32,6 +32,19 @@ evaluation, invoke `$optimize-prompt` in Claude Code or the namespaced `$optimize-anything:optimize-prompt` in Codex. It uses the bundled runtime, keeps search output outside the source, and applies only an accepted result. +## Reuse the Host Subscription + +When this skill runs in Codex, pass `--proposer-backend codex` and use +`--judge-backend codex` for built-in judging. When it runs in Claude Code, use +the corresponding `claude` values. For `analyze`, pass `--analysis-backend`; +for `validate`, use the reserved provider selector `codex` or `claude`. +Do not infer a backend in the Python runtime or for an unknown host. + +Tell the user before running that subscription calls are serialized by default. +Eligible availability, authentication, rate-limit, or quota failures may switch +to a same-vendor API model only when a matching key and fallback model are +available; pass `--no-api-fallback` when billed fallback is not acceptable. + ## Available Skills - **optimize-prompt** — Optimize inline prompts, files, embedded regions, or independent batches with fast prompt-quality or rigorous task-output evidence diff --git a/commands/analyze.md b/commands/analyze.md index 48411c9..937deb6 100644 --- a/commands/analyze.md +++ b/commands/analyze.md @@ -2,6 +2,10 @@ name: analyze description: Discover quality dimensions for an artifact and objective --- +When running in Codex, pass `--analysis-backend codex`; in Claude Code, pass +`--analysis-backend claude`. A subscription backend may omit `--judge-model`. +Unknown hosts use the existing API model. Announce possible same-vendor billed +API fallback, or pass `--no-api-fallback` when it is not acceptable. # analyze @@ -21,6 +25,9 @@ Use an LLM to discover relevant quality dimensions for a given artifact and opti --objective "Score for clarity and persuasiveness" ``` +In Codex or Claude Code, replace `--judge-model ...` with +`--analysis-backend codex` or `--analysis-backend claude`, respectively. + ## Next Step: optimize with discovered dimensions After dimension discovery, proceed directly to optimization using the returned `intake_json`: diff --git a/commands/compare.md b/commands/compare.md index 4129321..d72a65a 100644 --- a/commands/compare.md +++ b/commands/compare.md @@ -2,6 +2,10 @@ name: compare description: Side-by-side comparison of two artifacts using composed score calls --- +When comparison uses LLM scoring, select the current host explicitly: +`--judge-backend codex` in Codex or `--judge-backend claude` in Claude Code. +Unknown hosts keep API defaults. Announce possible same-vendor billed API +fallback, or add `--no-api-fallback` to prohibit it. Compare two artifacts with the same scoring setup by composing existing `score` calls. ## Usage diff --git a/commands/optimize.md b/commands/optimize.md index a11f1ae..cd914c7 100644 --- a/commands/optimize.md +++ b/commands/optimize.md @@ -8,6 +8,12 @@ For inline prompts, embedded prompt regions, independent prompt batches, or task-output evaluation with representative examples, invoke the shared `$optimize-prompt` skill instead. +When executing this Claude Code command, add `--proposer-backend claude` and, +when using the built-in judge, `--judge-backend claude`. Announce that +subscription calls are serialized and that +eligible failures may use billed same-vendor API fallback; honor a request to +disable it with `--no-api-fallback`. + ## Step 1: Identify the artifact - If the user provided a file argument, use it directly. - Otherwise ask: **"What file should I optimize?"** @@ -33,7 +39,8 @@ Present these options and ask the user to choose one unless they already specifi - If evaluator is already specified, proceed. - If no evaluator is specified: 1. Run `analyze` first: - - `"${CLAUDE_PLUGIN_ROOT}/scripts/run-optimize-anything" analyze --judge-model --objective ""` + - Subscription mode: `"${CLAUDE_PLUGIN_ROOT}/scripts/run-optimize-anything" analyze --analysis-backend claude --objective ""` + - API mode: `"${CLAUDE_PLUGIN_ROOT}/scripts/run-optimize-anything" analyze --judge-model --objective ""` 2. If analyze fails (API key missing, model unavailable): ask the user for their preferred model, or suggest using `--evaluator-command` with a custom script instead. 3. Ask: **"Should we use LLM judge directly, or do you want a custom evaluator?"** 4. If custom evaluator is needed, invoke evaluator generation workflow. diff --git a/commands/quick.md b/commands/quick.md index 73db47f..83e1c16 100644 --- a/commands/quick.md +++ b/commands/quick.md @@ -4,15 +4,21 @@ description: Zero-config one-shot optimization for fast improvements --- Run a no-questions-asked fast optimization. +Use Claude Code's subscription explicitly with `--analysis-backend claude`, +`--proposer-backend claude`, and `--judge-backend claude`. Announce serialized +subscription use and possible same-vendor billed API fallback before running. + ## Usage `/optimize-anything:quick ""` ## Behavior (do not ask follow-up questions) 1. Run analysis to discover dimensions: - - `"${CLAUDE_PLUGIN_ROOT}/scripts/run-optimize-anything" analyze --judge-model openai/gpt-5.6-luna --objective ""` + - Subscription mode: `"${CLAUDE_PLUGIN_ROOT}/scripts/run-optimize-anything" analyze --analysis-backend claude --objective ""` + - API mode: `"${CLAUDE_PLUGIN_ROOT}/scripts/run-optimize-anything" analyze --judge-model openai/gpt-5.6-luna --objective ""` - If analyze fails, skip dimension discovery and run optimize with `--judge-model` directly using the objective as-is. 2. Run optimization using LLM judge with: - - `--judge-model openai/gpt-5.6-luna` + - Subscription mode: `--proposer-backend claude --judge-backend claude` (omit API model strings) + - API mode: `--judge-model openai/gpt-5.6-luna` - `--budget 50` - `--diff` - `--early-stop` diff --git a/commands/score.md b/commands/score.md index 1f9e7a1..2af4066 100644 --- a/commands/score.md +++ b/commands/score.md @@ -2,6 +2,10 @@ name: score description: Score a single artifact with an evaluator --- +For an LLM judge, use `--judge-backend codex` in Codex or +`--judge-backend claude` in Claude Code; the subscription model may be omitted. +Do not add a judge backend for command or HTTP evaluators. Announce possible +same-vendor billed API fallback, or add `--no-api-fallback` to prohibit it. # score diff --git a/commands/validate.md b/commands/validate.md index 98beeb3..3cab49d 100644 --- a/commands/validate.md +++ b/commands/validate.md @@ -2,6 +2,10 @@ name: validate description: Cross-validate an artifact with multiple LLM judge providers --- +The `--providers` list accepts `codex`, `codex:`, `claude`, and +`claude:` before ordinary LiteLLM model strings. Prefer the selector for +the current host when subscription reuse is requested. Announce possible +same-vendor billed API fallback, or add `--no-api-fallback` to prohibit it. Use multiple LLM judges to verify that a quality improvement is not provider-specific. ## When to use diff --git a/install.md b/install.md index db18585..adafe33 100644 --- a/install.md +++ b/install.md @@ -1,5 +1,26 @@ # Installation Guide +## Optional local subscription backends + +The default install keeps LiteLLM/API behavior. To reuse a Codex login made +through ChatGPT, install the pinned optional SDK and log in with the provider's +CLI: + +```bash +uv sync --extra codex +codex login +codex login status +``` + +Claude subscription support uses the locally installed `claude` executable; +install Claude Code and run `claude auth login`. The adapter requires Claude +Code 2.1.278 or newer and verifies `claude.ai` first-party authentication. + +Neither backend accepts tokens in optimize-anything configuration. Codex uses +a private temporary home linked to the provider-owned saved auth file; Claude +runs with API/cloud environment overrides removed. Both fail closed if the +required isolation or auth class cannot be verified. + Choose the host integration or standalone runtime you need: | Method | Skills | Slash commands | Runtime | Prerequisites | diff --git a/pyproject.toml b/pyproject.toml index e80859b..c78d7b1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -16,8 +16,12 @@ dependencies = [ "httpx", "litellm>=1.83.0,<1.92,!=1.82.7,!=1.82.8", "cloudpickle", + "jsonschema>=4.26,<5", ] +[project.optional-dependencies] +codex = ["openai-codex>=0.156.0,<0.157.0"] + [project.urls] Homepage = "https://github.com/ASRagab/optimize-anything" Repository = "https://github.com/ASRagab/optimize-anything" diff --git a/skills/generate-evaluator/SKILL.md b/skills/generate-evaluator/SKILL.md index ecd9f10..e19a669 100644 --- a/skills/generate-evaluator/SKILL.md +++ b/skills/generate-evaluator/SKILL.md @@ -50,6 +50,9 @@ Generate an evaluator that scores candidate artifacts for optimization with gepa ## Generation Flags - `--evaluator-type judge|command|http|composite` - `--model `: hardcodes judge model into judge/composite scripts. +- `--judge-backend api|codex|claude`: configures the installed evaluator runtime. +- In Codex use `--judge-backend codex`; in Claude Code use `--judge-backend claude`. + Add `--no-api-fallback` when billed API fallback is not acceptable. - `--dataset`: generate dataset-aware templates that read `example` and show how to use it in scoring. - `--intake-json` / `--intake-file`: embed rubric/quality dimensions. diff --git a/skills/optimization-guide/SKILL.md b/skills/optimization-guide/SKILL.md index b80ac82..c4e5edb 100644 --- a/skills/optimization-guide/SKILL.md +++ b/skills/optimization-guide/SKILL.md @@ -65,6 +65,22 @@ single proposal. ### 5. Run Optimization +If the workflow is executing inside Codex or Claude Code, reuse that host's +logged-in subscription explicitly: + +```bash +# Codex host +optimize-anything optimize seed.txt --proposer-backend codex --judge-backend codex ... + +# Claude Code host +optimize-anything optimize seed.txt --proposer-backend claude --judge-backend claude ... +``` + +Subscription calls default to one concurrent call per provider. Announce the +backend and possible billed API fallback before execution; add +`--no-api-fallback` to prohibit it. Unknown hosts omit backend flags and keep +the API defaults. + **Via CLI:** ```bash optimize-anything optimize seed.txt --evaluator-command bash evaluators/eval.sh --budget 100 --objective "maximize clarity" -o result.txt diff --git a/src/optimize_anything/cli.py b/src/optimize_anything/cli.py index 905fe05..a7d5b25 100644 --- a/src/optimize_anything/cli.py +++ b/src/optimize_anything/cli.py @@ -15,11 +15,35 @@ _preflight_http_evaluator, ) from optimize_anything.model_defaults import DEFAULT_EVALUATOR_MODEL +from optimize_anything.llm_backends.base import Role EvaluatorFn = Callable[..., tuple[float, dict[str, Any]]] EvaluatorFactory = Callable[..., EvaluatorFn] +def _add_subscription_options(parser: argparse.ArgumentParser) -> None: + parser.add_argument( + "--subscription-concurrency", + type=int, + default=1, + help="Maximum concurrent calls per subscription provider (default: 1).", + ) + parser.add_argument( + "--no-api-fallback", + action="store_true", + default=False, + help="Never switch a subscription request to a billed API call.", + ) + parser.add_argument( + "--openai-api-fallback-model", + help="OpenAI API model used if an eligible Codex subscription call fails.", + ) + parser.add_argument( + "--anthropic-api-fallback-model", + help="Anthropic API model used if an eligible Claude subscription call fails.", + ) + + def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser( prog="optimize-anything", @@ -66,6 +90,12 @@ def main(argv: list[str] | None = None) -> int: "Falls back to OPTIMIZE_ANYTHING_MODEL env var." ), ) + opt_parser.add_argument( + "--proposer-backend", + choices=["api", "codex", "claude"], + default=None, + help="Backend for proposal generation (default: api).", + ) opt_parser.add_argument( "--judge-model", help=( @@ -74,6 +104,13 @@ def main(argv: list[str] | None = None) -> int: "Mutually exclusive with --evaluator-command and --evaluator-url." ), ) + opt_parser.add_argument( + "--judge-backend", + choices=["api", "codex", "claude"], + default=None, + help="Backend for the built-in LLM judge (default: api).", + ) + _add_subscription_options(opt_parser) opt_parser.add_argument( "--judge-objective", help="Objective for the LLM judge. Falls back to --objective if not set.", @@ -188,12 +225,20 @@ def main(argv: list[str] | None = None) -> int: ) gen_parser.add_argument( "--model", - default=DEFAULT_EVALUATOR_MODEL, + default=None, help=( "LiteLLM model to hardcode in generated judge/composite evaluators " - f"(default: {DEFAULT_EVALUATOR_MODEL})" + f"(API default: {DEFAULT_EVALUATOR_MODEL}; subscription default: provider choice)" ), ) + gen_parser.add_argument( + "--judge-backend", + choices=["api", "codex", "claude"], + default=None, + help="Backend configured in generated judge/composite evaluators.", + ) + gen_parser.add_argument("--api-base", help="API base for the generated evaluator runtime.") + _add_subscription_options(gen_parser) gen_parser.add_argument( "--dataset", action="store_true", @@ -263,6 +308,13 @@ def main(argv: list[str] | None = None) -> int: "Mutually exclusive with --evaluator-command and --evaluator-url." ), ) + score_parser.add_argument( + "--judge-backend", + choices=["api", "codex", "claude"], + default=None, + help="Backend for LLM scoring (default: api).", + ) + _add_subscription_options(score_parser) score_parser.add_argument( "--objective", help="Objective for the LLM judge (required with --judge-model).", @@ -308,6 +360,7 @@ def main(argv: list[str] | None = None) -> int: "gemini/gemini-3.6-flash" ), ) + _add_subscription_options(validate_parser) validate_parser.add_argument( "--objective", required=True, @@ -334,9 +387,16 @@ def main(argv: list[str] | None = None) -> int: ) analyze_parser.add_argument( "--judge-model", - required=True, + required=False, help="LiteLLM model string for the LLM judge. Judge: openai/gpt-5.6-luna.", ) + analyze_parser.add_argument( + "--analysis-backend", + choices=["api", "codex", "claude"], + default=None, + help="Backend for scoring and dimension discovery (default: api).", + ) + _add_subscription_options(analyze_parser) analyze_parser.add_argument( "--objective", required=True, @@ -363,6 +423,12 @@ def main(argv: list[str] | None = None) -> int: ) args = parser.parse_args(argv) + if ( + args.command == "analyze" + and (args.analysis_backend or "api") == "api" + and not args.judge_model + ): + parser.error("analyze with the API backend requires --judge-model") if args.command == "optimize": from optimize_anything.cli_optimize import _cmd_optimize @@ -437,6 +503,9 @@ def _resolve_evaluator( api_base: str | None = None, task_model: str | None = None, score_range: str = "unit", + judge_backend: str = "api", + completion_backend: object | None = None, + completion_role: Role = "judge", ) -> tuple[EvaluatorFn | None, str]: """Resolve evaluator from command, URL, judge model, or intake spec. @@ -452,10 +521,10 @@ def _resolve_evaluator( evaluator_sources = sum([ bool(evaluator_command), bool(evaluator_url), - bool(judge_model), + bool(judge_model) or judge_backend != "api", ]) if evaluator_sources > 1: - return (None, "Error: provide only one of --evaluator-command, --evaluator-url, or --judge-model") + return (None, "Error: provide only one of --evaluator-command, --evaluator-url, --judge-model, or --judge-backend") if evaluator_sources == 0: if allow_intake_fallback and intake_spec is not None: @@ -464,7 +533,7 @@ def _resolve_evaluator( return (None, "Error: intake execution_mode='command' requires --evaluator-command") else: return (None, "Error: intake execution_mode='http' requires --evaluator-url") - return (None, "Error: provide --evaluator-command, --evaluator-url, or --judge-model") + return (None, "Error: provide --evaluator-command, --evaluator-url, --judge-model, or a subscription --judge-backend") if evaluator_command: return _resolve_command_evaluator_source( @@ -490,6 +559,9 @@ def _resolve_evaluator( intake_spec=intake_spec, api_base=api_base, task_model=task_model, + judge_backend=judge_backend, + completion_backend=completion_backend, + completion_role=completion_role, ) @@ -562,10 +634,15 @@ def _resolve_judge_evaluator_source( intake_spec: dict[str, Any] | None, api_base: str | None, task_model: str | None, + judge_backend: str = "api", + completion_backend: object | None = None, + completion_role: Role = "judge", ) -> tuple[EvaluatorFn | None, str]: judge_obj = judge_objective or objective if not judge_obj: - return (None, "Error: --judge-model requires --objective or --judge-objective") + if judge_backend == "api": + return (None, "Error: --judge-model requires --objective or --judge-objective") + return (None, "Error: the built-in judge requires --objective or --judge-objective") quality_dimensions = None hard_constraints = None @@ -580,8 +657,13 @@ def _resolve_judge_evaluator_source( hard_constraints=hard_constraints, api_base=api_base, task_model=task_model, + backend=completion_backend, + role=completion_role, ) - return eval_fn, f"LLM judge ({judge_model})" + model_label = judge_model or "provider default" + if judge_backend == "api": + return eval_fn, f"LLM judge ({model_label})" + return eval_fn, f"LLM judge ({judge_backend}:{model_label})" def _read_seed(path: str) -> str | None: diff --git a/src/optimize_anything/cli_optimize.py b/src/optimize_anything/cli_optimize.py index 05140cc..dd375c6 100644 --- a/src/optimize_anything/cli_optimize.py +++ b/src/optimize_anything/cli_optimize.py @@ -6,8 +6,12 @@ import copy import json import sys +from contextlib import contextmanager, nullcontext +from dataclasses import dataclass from pathlib import Path -from typing import Any +from typing import Any, Iterator, cast + +from optimize_anything.llm_backends.base import BackendName from optimize_anything.persist import ( _copy_cache_from_run, @@ -27,14 +31,40 @@ ] -def _cmd_optimize(args: argparse.Namespace) -> int: - from optimize_anything.cli import _resolve_evaluator +@dataclass(frozen=True) +class OptimizationBackends: + proposer_lm: Any + proposer_model: str | None + judge: Any + judge_name: str + coordinator: Any + plan: dict[str, Any] + +def _cmd_optimize(args: argparse.Namespace) -> int: prepared = _prepare_optimize_inputs(args) if prepared is None: return 1 args, seed, dataset, valset, intake_spec = prepared + try: + with _optimization_backends(args) as backend_state: + return _run_optimize(args, seed, dataset, valset, intake_spec, backend_state) + except Exception as exc: + print(f"Error: backend preflight failed: {exc}", file=sys.stderr) + return 1 + + +def _run_optimize( + args: argparse.Namespace, + seed: str | None, + dataset: list[dict] | None, + valset: list[dict] | None, + intake_spec: dict[str, Any] | None, + backend_state: OptimizationBackends, +) -> int: + from optimize_anything.cli import _resolve_evaluator + from optimize_anything.result_contract import build_optimize_summary from gepa.optimize_anything import optimize_anything @@ -50,6 +80,8 @@ def _cmd_optimize(args: argparse.Namespace) -> int: api_base=args.api_base, task_model=args.task_model, score_range=args.score_range, + judge_backend=backend_state.judge_name, + completion_backend=backend_state.judge, ) if eval_fn is None: print(evaluator_label, file=sys.stderr) @@ -67,7 +99,7 @@ def _cmd_optimize(args: argparse.Namespace) -> int: file=sys.stderr, ) - model = resolve_proposer_model(args.model) + model = backend_state.proposer_lm config, gepa_run_dir, early_stop_active, runtime_error = _build_optimize_runtime( args, model=model, @@ -104,6 +136,10 @@ def _cmd_optimize(args: argparse.Namespace) -> int: requested_budget=args.budget, early_stop_active=early_stop_active, ) + summary["backend_plan"] = backend_state.plan + coordinator = backend_state.coordinator + if coordinator is not None: + summary["llm_provenance"] = coordinator.events() best = summary["best_artifact"] persist_error = _persist_optimize_outputs( args=args, @@ -122,7 +158,7 @@ def _cmd_optimize(args: argparse.Namespace) -> int: if summary.get("plateau_detected") and args.judge_model: _print_judge_plateau_advisory( judge_model=args.judge_model, - proposer_model=model, + proposer_model=backend_state.proposer_model, has_intake=intake_spec is not None, file=sys.stderr, ) @@ -131,10 +167,116 @@ def _cmd_optimize(args: argparse.Namespace) -> int: return 0 +@contextmanager +def _optimization_backends(args: argparse.Namespace) -> Iterator[OptimizationBackends]: + from optimize_anything.llm_backends.coordination import RunCoordinator + + proposer_name = getattr(args, "proposer_backend", None) or "api" + judge_name = getattr(args, "judge_backend", None) or "api" + proposer_model = ( + resolve_proposer_model(args.model) if proposer_name == "api" else args.model + ) + judge_selected = bool(args.judge_model) or judge_name != "api" + concurrency = getattr(args, "subscription_concurrency", 1) + if concurrency < 1: + raise ValueError("--subscription-concurrency must be at least 1") + # Generated evaluator processes may select either subscription backend even + # when the parent proposer and judge use API models. + coordinator = RunCoordinator.create({"codex": concurrency, "claude": concurrency}) + try: + environment = coordinator.exported_environment() + with environment: + yield from _configured_optimization_backends( + args, + proposer_name=proposer_name, + judge_name=judge_name, + proposer_model=proposer_model, + judge_selected=judge_selected, + concurrency=concurrency, + coordinator=coordinator, + ) + finally: + coordinator.close() + + +def _configured_optimization_backends( + args: argparse.Namespace, + *, + proposer_name: str, + judge_name: str, + proposer_model: str | None, + judge_selected: bool, + concurrency: int, + coordinator: Any, +) -> Iterator[OptimizationBackends]: + from optimize_anything.llm_backends.factory import ( + BackendLanguageModel, + create_backend, + resolve_backend_spec, + ) + + def role_spec(role: str, backend: str, model: str | None): + role_fallback = getattr(args, f"{role}_api_fallback", None) + role_fallback_model = getattr(args, f"{role}_api_fallback_model", None) + openai_fallback = getattr(args, "openai_api_fallback_model", None) + anthropic_fallback = getattr(args, "anthropic_api_fallback_model", None) + if role_fallback_model and backend == "codex" and openai_fallback is None: + openai_fallback = role_fallback_model + if role_fallback_model and backend == "claude" and anthropic_fallback is None: + anthropic_fallback = role_fallback_model + return resolve_backend_spec( + backend=cast(BackendName, backend), + model=model, + api_base=args.api_base, + no_api_fallback=( + getattr(args, "no_api_fallback", False) or role_fallback is False + ), + openai_api_fallback_model=openai_fallback, + anthropic_api_fallback_model=anthropic_fallback, + max_concurrency=concurrency, + ) + + proposer_spec = role_spec("proposer", proposer_name, proposer_model) + proposer_backend = create_backend( + proposer_spec, role="proposer", coordinator=coordinator + ) + proposer_lm: Any = proposer_model + if proposer_name != "api": + proposer_backend.preflight() + proposer_lm = BackendLanguageModel(proposer_backend, model=proposer_model) + + judge_backend = None + if judge_selected: + judge_spec = role_spec("judge", judge_name, args.judge_model) + judge_backend = create_backend(judge_spec, role="judge", coordinator=coordinator) + if judge_name != "api": + judge_backend.preflight() + + plan = { + "proposer": {"backend": proposer_name, "model": proposer_model}, + "judge": { + "backend": judge_name if judge_selected else None, + "model": args.judge_model, + }, + "api_fallback": not getattr(args, "no_api_fallback", False), + "subscription_concurrency": concurrency, + "custom_api_base": bool(args.api_base), + } + print(f"Backend plan: {json.dumps(plan, sort_keys=True)}", file=sys.stderr) + yield OptimizationBackends( + proposer_lm=proposer_lm, + proposer_model=proposer_model, + judge=judge_backend, + judge_name=judge_name, + coordinator=coordinator, + plan=plan, + ) + + def _build_optimize_runtime( args: argparse.Namespace, *, - model: str, + model: Any, ) -> tuple[Any, str | None, bool, str | None]: """Build GEPA runtime config plus run-dir state for optimize.""" from optimize_anything.stop import plateau_stop_callback @@ -277,9 +419,16 @@ def _resolve_optimize_seed(args: argparse.Namespace) -> tuple[bool, str | None]: if not args.no_seed: print("Error: provide seed_file or pass --no-seed", file=sys.stderr) return False, None - if not args.objective or not args.model: + if not args.objective or ( + (getattr(args, "proposer_backend", None) or "api") == "api" and not args.model + ): + message = ( + "Error: seedless mode (--no-seed) requires both --objective and --model" + if (getattr(args, "proposer_backend", None) or "api") == "api" + else "Error: seedless mode (--no-seed) requires --objective" + ) print( - "Error: seedless mode (--no-seed) requires both --objective and --model", + message, file=sys.stderr, ) return False, None @@ -366,6 +515,12 @@ def _apply_spec_to_args( "budget", "proposals_per_iteration", "judge_model", + "proposer_backend", + "judge_backend", + "proposer_api_fallback", + "judge_api_fallback", + "proposer_api_fallback_model", + "judge_api_fallback_model", ) _apply_spec_alias_if_missing(args, spec, arg_key="model", spec_key="proposer_model") _apply_parallel_from_spec(args, spec) diff --git a/src/optimize_anything/cli_tools.py b/src/optimize_anything/cli_tools.py index 352936e..30999e3 100644 --- a/src/optimize_anything/cli_tools.py +++ b/src/optimize_anything/cli_tools.py @@ -6,7 +6,8 @@ import json import statistics import sys -from typing import Any, cast +from contextlib import contextmanager +from typing import Any, Iterator, cast from optimize_anything.cli import ( EvaluatorFn, @@ -14,6 +15,7 @@ _read_seed, _resolve_evaluator, ) +from optimize_anything.llm_backends.base import BackendName, Role def _cmd_intake(args: argparse.Namespace) -> int: @@ -83,13 +85,26 @@ def _cmd_generate_evaluator(args: argparse.Namespace) -> int: from optimize_anything.evaluator_generator import generate_evaluator_script + from optimize_anything.model_defaults import DEFAULT_EVALUATOR_MODEL + + backend_name = args.judge_backend or "api" + model = args.model or (DEFAULT_EVALUATOR_MODEL if backend_name == "api" else None) script = generate_evaluator_script( seed=seed, objective=args.objective, evaluator_type=args.evaluator_type, intake=intake_spec, - model=args.model, + model=model, dataset=args.dataset, + backend=backend_name, + api_base=args.api_base, + api_fallback=not args.no_api_fallback, + api_fallback_model=( + args.openai_api_fallback_model + if args.judge_backend == "codex" + else args.anthropic_api_fallback_model + ), + max_concurrency=args.subscription_concurrency, ) print(script, end="") return 0 @@ -145,32 +160,37 @@ def _cmd_score(args: argparse.Namespace) -> int: if intake_requested and intake_spec is None: return 1 - eval_fn, error = _resolve_evaluator( - evaluator_command=args.evaluator_command, - evaluator_url=args.evaluator_url, - judge_model=args.judge_model, - judge_objective=args.judge_objective, - objective=args.objective, - evaluator_cwd=args.evaluator_cwd, - intake_spec=intake_spec, - allow_intake_fallback=False, - api_base=args.api_base, - task_model=args.task_model, - score_range=args.score_range, - ) - if eval_fn is None: - print(error, file=sys.stderr) - return 1 - - if args.evaluator_url and args.evaluator_cwd: - print( - "Warning: --evaluator-cwd has no effect when using --evaluator-url. " - "The HTTP evaluator runs in the server's own working directory.", - file=sys.stderr, - ) - + judge_name = cast(BackendName, args.judge_backend or "api") try: - score, side_info = eval_fn(artifact) + with _completion_backend(args, judge_name, args.judge_model, "score") as backend: + eval_fn, error = _resolve_evaluator( + evaluator_command=args.evaluator_command, + evaluator_url=args.evaluator_url, + judge_model=args.judge_model, + judge_objective=args.judge_objective, + objective=args.objective, + evaluator_cwd=args.evaluator_cwd, + intake_spec=intake_spec, + allow_intake_fallback=False, + api_base=args.api_base, + task_model=args.task_model, + score_range=args.score_range, + judge_backend=judge_name, + completion_backend=backend, + completion_role="score", + ) + if eval_fn is None: + print(error, file=sys.stderr) + return 1 + + if args.evaluator_url and args.evaluator_cwd: + print( + "Warning: --evaluator-cwd has no effect when using --evaluator-url. " + "The HTTP evaluator runs in the server's own working directory.", + file=sys.stderr, + ) + + score, side_info = eval_fn(artifact) except Exception as exc: print(f"Error: evaluator call failed: {exc}", file=sys.stderr) return 1 @@ -203,13 +223,17 @@ def _cmd_validate(args: argparse.Namespace) -> int: provider_results: list[dict[str, object]] = [] successful_scores: list[float] = [] for provider in args.providers: + backend_name, model = _parse_validation_provider(provider) result, numeric_score = _validate_provider( artifact=artifact, provider=provider, + backend_name=backend_name, + model=model, objective=args.objective, quality_dimensions=quality_dimensions, hard_constraints=hard_constraints, api_base=args.api_base, + args=args, ) provider_results.append(result) if numeric_score is not None: @@ -259,20 +283,23 @@ def _cmd_analyze(args: argparse.Namespace) -> int: from optimize_anything.llm_judge import analyze_for_dimensions print( - f"Analyzing artifact with {args.judge_model}...", + f"Analyzing artifact with {args.analysis_backend or 'api'}:{args.judge_model or 'provider default'}...", file=sys.stderr, ) try: - result = analyze_for_dimensions( - artifact=artifact, - objective=args.objective, - model=args.judge_model, - api_base=args.api_base, - timeout=args.timeout, - temperature=args.temperature, - ) - except (ValueError, RuntimeError) as exc: + backend_name = cast(BackendName, args.analysis_backend or "api") + with _completion_backend(args, backend_name, args.judge_model, "analysis") as backend: + result = analyze_for_dimensions( + artifact=artifact, + objective=args.objective, + model=args.judge_model, + api_base=args.api_base, + timeout=args.timeout, + temperature=args.temperature, + backend=backend, + ) + except Exception as exc: print(f"Error: {exc}", file=sys.stderr) return 1 @@ -284,25 +311,31 @@ def _validate_provider( *, artifact: str, provider: str, + backend_name: BackendName, + model: str | None, objective: str, quality_dimensions: Any, hard_constraints: Any, api_base: str | None, + args: argparse.Namespace, ) -> tuple[dict[str, object], float | None]: from optimize_anything.llm_judge import llm_judge_evaluator try: - evaluator = cast( - EvaluatorFn, - llm_judge_evaluator( - objective, - model=provider, - quality_dimensions=quality_dimensions, - hard_constraints=hard_constraints, - api_base=api_base, - ), - ) - score, side_info = evaluator(artifact) + with _completion_backend(args, backend_name, model, "validation") as backend: + evaluator = cast( + EvaluatorFn, + llm_judge_evaluator( + objective, + model=model, + quality_dimensions=quality_dimensions, + hard_constraints=hard_constraints, + api_base=api_base, + backend=backend, + role="validation", + ), + ) + score, side_info = evaluator(artifact) numeric_score = float(score) except Exception as exc: return { @@ -318,3 +351,38 @@ def _validate_provider( if isinstance(side_info, dict): result.update(side_info) return result, numeric_score + + +def _parse_validation_provider(provider: str) -> tuple[BackendName, str | None]: + if provider == "codex" or provider.startswith("codex:"): + return "codex", provider.partition(":")[2] or None + if provider == "claude" or provider.startswith("claude:"): + return "claude", provider.partition(":")[2] or None + return "api", provider + + +@contextmanager +def _completion_backend( + args: argparse.Namespace, + backend_name: BackendName, + model: str | None, + role: Role, +) -> Iterator[Any]: + from optimize_anything.llm_backends.factory import create_backend, resolve_backend_spec + + concurrency = getattr(args, "subscription_concurrency", 1) + if concurrency < 1: + raise ValueError("--subscription-concurrency must be at least 1") + spec = resolve_backend_spec( + backend=backend_name, + model=model, + api_base=getattr(args, "api_base", None), + no_api_fallback=getattr(args, "no_api_fallback", False), + openai_api_fallback_model=getattr(args, "openai_api_fallback_model", None), + anthropic_api_fallback_model=getattr(args, "anthropic_api_fallback_model", None), + max_concurrency=concurrency, + ) + backend = create_backend(spec, role=role) + if backend_name != "api": + backend.preflight() + yield backend diff --git a/src/optimize_anything/evaluator_generator.py b/src/optimize_anything/evaluator_generator.py index 5472d54..60ea0eb 100644 --- a/src/optimize_anything/evaluator_generator.py +++ b/src/optimize_anything/evaluator_generator.py @@ -15,8 +15,13 @@ def generate_evaluator_script( objective: str, evaluator_type: str | None = None, intake: Mapping[str, Any] | None = None, - model: str = DEFAULT_EVALUATOR_MODEL, + model: str | None = DEFAULT_EVALUATOR_MODEL, dataset: bool = False, + backend: str = "api", + api_base: str | None = None, + api_fallback: bool = False, + api_fallback_model: str | None = None, + max_concurrency: int = 1, ) -> str: """Generate an evaluator script that reads input JSON and outputs score JSON.""" normalized_intake = _normalize_intake_if_provided(intake) @@ -37,6 +42,24 @@ def generate_evaluator_script( quality_dimensions=quality_dimensions, dataset=dataset, ) + if resolved_evaluator_type in {"judge", "composite"} and backend != "api": + return _generate_runtime_evaluator( + objective, + evaluator_type=resolved_evaluator_type, + template_family=template_family, + rubric_summary=rubric_summary, + quality_dimensions=quality_dimensions, + hard_constraints=list((normalized_intake or {}).get("hard_constraints", [])), + model=model, + dataset=dataset, + backend=backend, + api_base=api_base, + api_fallback=api_fallback, + api_fallback_model=api_fallback_model, + max_concurrency=max_concurrency, + ) + if model is None: + raise ValueError("API judge evaluators require a model") if resolved_evaluator_type == "judge": return _generate_judge_evaluator( seed, @@ -631,6 +654,69 @@ def main() -> int: """).lstrip() +def _generate_runtime_evaluator( + objective: str, + *, + evaluator_type: str, + template_family: str, + rubric_summary: str, + quality_dimensions: list[tuple[str, float]], + hard_constraints: list[str], + model: str | None, + dataset: bool, + backend: str, + api_base: str | None, + api_fallback: bool, + api_fallback_model: str | None, + max_concurrency: int, +) -> str: + """Generate a configuration wrapper for a subscription evaluator runtime.""" + from optimize_anything.evaluator_runtime import RUNTIME_CONTRACT_VERSION + + config = { + "min_runtime_contract_version": RUNTIME_CONTRACT_VERSION, + "evaluator_type": evaluator_type, + "objective": objective, + "template_family": template_family, + "rubric_summary": rubric_summary, + "quality_dimensions": quality_dimensions, + "hard_constraints": hard_constraints, + "model": model, + "dataset": dataset, + "backend": backend, + "api_base": api_base, + "api_fallback": api_fallback, + "api_fallback_model": api_fallback_model, + "max_concurrency": max_concurrency, + } + return textwrap.dedent(f"""\ + #!/usr/bin/env python3 + import json + import sys + + MODEL = {model!r} + EVALUATOR_METADATA = {{"min_runtime_contract_version": {RUNTIME_CONTRACT_VERSION}}} + CONFIG = {config!r} + + def main() -> int: + try: + from optimize_anything.evaluator_runtime import run_generated_evaluator + except ImportError: + for line in sys.stdin: + if line.strip(): + print(json.dumps({{ + "score": 0.0, + "error": "runtime_unavailable", + "reasoning": "Install or upgrade optimize-anything to run this evaluator.", + }})) + return 0 + return run_generated_evaluator(CONFIG) + + if __name__ == "__main__": + raise SystemExit(main()) + """).lstrip() + + def _indent(text: str, spaces: int) -> str: pad = " " * spaces return "\n".join(pad + line if line else line for line in text.splitlines()) diff --git a/src/optimize_anything/evaluator_runtime.py b/src/optimize_anything/evaluator_runtime.py new file mode 100644 index 0000000..f739f7e --- /dev/null +++ b/src/optimize_anything/evaluator_runtime.py @@ -0,0 +1,187 @@ +"""Installed runtime for versioned generated LLM evaluator scripts.""" + +from __future__ import annotations + +import json +import math +import re +import sys +from collections.abc import Callable, Mapping +from typing import Any, TextIO + +from optimize_anything.llm_backends.base import Role +from optimize_anything.llm_backends.schema import score_output_schema, strip_code_fences + +RUNTIME_CONTRACT_VERSION = 1 +def _resolve_backend(config: Mapping[str, Any], *, role: Role) -> Any: + """Keep the installed backend factory boundary in one replaceable place.""" + from optimize_anything.llm_backends.base import BackendSpec + from optimize_anything.llm_backends.factory import create_backend + + spec = BackendSpec( + backend=config.get("backend", "api"), + model=config.get("model"), + api_base=config.get("api_base"), + api_fallback=config.get("api_fallback", False), + api_fallback_model=config.get("api_fallback_model"), + max_concurrency=config.get("max_concurrency", 1), + ) + return create_backend(spec, role=role) + + +def run_generated_evaluator( + config: Mapping[str, Any], + *, + backend_resolver: Callable[..., Any] | None = None, + input_stream: TextIO | None = None, + output_stream: TextIO | None = None, +) -> int: + """Read JSON lines and emit one canonical numeric-score line per input line.""" + source = input_stream if input_stream is not None else sys.stdin + destination = output_stream if output_stream is not None else sys.stdout + resolver = backend_resolver or _resolve_backend + backend = None + + for line in source: + if not line.strip(): + continue + required_version = config.get("min_runtime_contract_version") + if (not isinstance(required_version, int) or isinstance(required_version, bool) + or required_version < 1 or required_version > RUNTIME_CONTRACT_VERSION): + _emit(destination, { + "score": 0.0, "error": "incompatible_runtime", + "reasoning": "Generated evaluator runtime contract is incompatible; upgrade optimize-anything.", + }) + continue + try: + payload = json.loads(line) + except json.JSONDecodeError: + _emit(destination, {"score": 0.0, "error": "invalid_input", "reasoning": "Input must be valid JSON."}) + continue + if not isinstance(payload, dict): + _emit(destination, {"score": 0.0, "error": "invalid_input", "reasoning": "Input must be a JSON object."}) + continue + + candidate = str(payload.get("candidate", "")) + if candidate == "__optimize_anything_preflight__": + _emit(destination, {"score": 1.0, "reasoning": "Evaluator runtime is ready."}) + continue + is_composite = config.get("evaluator_type") == "composite" + if is_composite: + failures = _local_constraint_failures(candidate) + if failures: + _emit(destination, { + "score": 0.0, "reasoning": "Hard constraints failed", + "hard_constraint_failures": failures, + "hard_constraints_satisfied": False, + }) + continue + + try: + if backend is None: + backend = resolver(config, role="judge") + from optimize_anything.llm_backends.base import CompletionRequest + + request = CompletionRequest( + prompt=_build_prompt(candidate, payload.get("example") if config.get("dataset") else None, config), + role="judge", + model=config.get("model"), + output_schema=score_output_schema( + [str(name) for name, _ in config.get("quality_dimensions", ())], + include_hard_constraints=bool(config.get("hard_constraints")), + ), + timeout_seconds=60.0, + prompt_contract_version="generated-evaluator-v1", + schema_contract_version="generated-evaluator-v1", + ) + completion = backend.complete(request) + parsed = completion.structured + if parsed is None: + parsed = json.loads(strip_code_fences(completion.text)) + result = _score_result(parsed, config, is_composite=is_composite) + from optimize_anything.llm_backends.provenance import completion_event + + try: + result["llm_provenance"] = completion_event(completion) + except (AttributeError, TypeError): + # Injectable test/third-party backends may implement only text/structured. + pass + except (ImportError, ModuleNotFoundError): + result = { + "score": 0.0, "error": "runtime_backend_unavailable", + "reasoning": "Install a compatible optimize-anything runtime and backend extra.", + } + except Exception as exc: + result = { + "score": 0.0, "error": "evaluator_failed", + "reasoning": f"Evaluator failed: {type(exc).__name__}.", + } + _emit(destination, result) + return 0 + + +def _build_prompt(candidate: str, example: Any, config: Mapping[str, Any]) -> str: + dimensions = "\n".join( + f"- {name} (weight={weight})" for name, weight in config.get("quality_dimensions", ()) + ) + constraints = "\n".join(f"- {item}" for item in config.get("hard_constraints", ())) or "(none)" + example_text = json.dumps(example, ensure_ascii=False, indent=2) if example is not None else "(none)" + return ( + "You are a careful, objective evaluator. Return only a JSON object.\n" + f"## Objective\n{config['objective']}\n\n" + f"## Template Family\n{config['template_family']}\n\n" + f"## Rubric Summary\n{config['rubric_summary']}\n\n" + f"## Quality Dimensions\n{dimensions}\n\n" + f"## Hard Constraints\n{constraints}\n\n" + f"## Example Context (optional)\n{example_text}\n\n" + f"## Artifact to Evaluate\n```\n{candidate}\n```\n\n" + "Return JSON with score, reasoning, one numeric key per quality dimension, and " + "hard_constraints_satisfied when constraints are present. score must be in [0,1]." + ) + + +def _local_constraint_failures(candidate: str) -> list[str]: + failures = [] + if not candidate.strip(): + failures.append("candidate must not be empty") + if len(candidate) > 12000: + failures.append("candidate exceeds max_len=12000") + if re.search(r"TODO|TBD|\[FILL\]", candidate): + failures.append("candidate contains placeholder tokens") + return failures + + +def _score_result(parsed: Any, config: Mapping[str, Any], *, is_composite: bool) -> dict[str, Any]: + if not isinstance(parsed, Mapping): + return {"score": 0.0, "error": "invalid_response", "reasoning": "Judge returned non-object JSON."} + raw_score = _unit_float(parsed.get("score")) + result: dict[str, Any] = { + "score": raw_score, + "reasoning": str(parsed.get("reasoning", "No reasoning provided.")), + "dimension_scores": {}, + } + for name, _ in config.get("quality_dimensions", ()): + value = _unit_float(parsed.get(name)) + result["dimension_scores"][name] = value + if name not in {"score", "reasoning", "dimension_scores", "error", "hard_constraints_satisfied", "hard_constraint_failures"}: + result[name] = value + if config.get("hard_constraints") and parsed.get("hard_constraints_satisfied") is False: + result["score"] = 0.0 + result["hard_constraints_satisfied"] = False + elif is_composite: + result["hard_constraints_satisfied"] = True + return result + + +def _unit_float(value: Any) -> float: + if isinstance(value, bool): + return 0.0 + try: + number = float(value) + except (TypeError, ValueError): + return 0.0 + return max(0.0, min(1.0, number)) if math.isfinite(number) else 0.0 + + +def _emit(stream: TextIO, result: Mapping[str, Any]) -> None: + stream.write(json.dumps(result, ensure_ascii=False) + "\n") diff --git a/src/optimize_anything/llm_backends/__init__.py b/src/optimize_anything/llm_backends/__init__.py new file mode 100644 index 0000000..47bdb20 --- /dev/null +++ b/src/optimize_anything/llm_backends/__init__.py @@ -0,0 +1,39 @@ +"""Completion backends and run-scoped execution primitives.""" + +from .base import ( + AuthenticationError, + BackendCapabilities, + BackendError, + BackendSpec, + BackendStatus, + BackendUnavailable, + Cancelled, + CompletionBackend, + CompletionRequest, + CompletionResult, + ConfigurationError, + FallbackRecord, + InvalidResponse, + QuotaExceeded, + RateLimitError, + SamplingOptions, + Timeout, + Usage, + thaw_json, + validate_capabilities, +) +from .coordination import RunCoordinator +from .fallback import FallbackBackend, fallback_ready +from .factory import BackendLanguageModel, create_backend, resolve_backend_spec +from .litellm_backend import LiteLLMBackend, validate_structured +from .provenance import aggregate_provenance, cache_fingerprint, completion_event + +__all__ = [ + "AuthenticationError", "BackendCapabilities", "BackendError", "BackendSpec", "BackendStatus", + "BackendUnavailable", "Cancelled", "CompletionBackend", "CompletionRequest", "CompletionResult", + "ConfigurationError", "FallbackBackend", "FallbackRecord", "InvalidResponse", "LiteLLMBackend", + "QuotaExceeded", "RateLimitError", "RunCoordinator", "SamplingOptions", "Timeout", "Usage", + "aggregate_provenance", "cache_fingerprint", "completion_event", "fallback_ready", "thaw_json", + "validate_capabilities", "validate_structured", "BackendLanguageModel", "create_backend", + "resolve_backend_spec", +] diff --git a/src/optimize_anything/llm_backends/base.py b/src/optimize_anything/llm_backends/base.py new file mode 100644 index 0000000..38032bc --- /dev/null +++ b/src/optimize_anything/llm_backends/base.py @@ -0,0 +1,217 @@ +"""Provider-neutral, immutable single-completion contract.""" + +from __future__ import annotations + +import math +from dataclasses import dataclass +from types import MappingProxyType +from typing import Any, Literal, Mapping, Protocol + +Role = Literal["proposer", "judge", "analysis", "score", "validation"] +BackendName = Literal["api", "codex", "claude"] +AuthClass = Literal["api", "subscription"] +AuthSource = Literal["chatgpt", "claude_subscription", "openai_api", "anthropic_api", "other_api"] +JsonValue = Any + + +def _freeze(value: Any) -> Any: + if isinstance(value, Mapping): + return MappingProxyType({str(key): _freeze(item) for key, item in value.items()}) + if isinstance(value, (list, tuple)): + return tuple(_freeze(item) for item in value) + return value + + +def thaw_json(value: Any) -> Any: + """Return plain JSON-compatible containers from an immutable contract value.""" + if isinstance(value, Mapping): + return {key: thaw_json(item) for key, item in value.items()} + if isinstance(value, tuple): + return [thaw_json(item) for item in value] + return value + + +@dataclass(frozen=True) +class SamplingOptions: + temperature: float | None = None + top_p: float | None = None + max_output_tokens: int | None = None + + def __post_init__(self) -> None: + if self.temperature is not None and ( + not math.isfinite(self.temperature) or self.temperature < 0 + ): + raise ConfigurationError("temperature must be finite and nonnegative") + if self.top_p is not None and ( + not math.isfinite(self.top_p) or not 0 < self.top_p <= 1 + ): + raise ConfigurationError("top_p must be between zero and one") + if self.max_output_tokens is not None and ( + not isinstance(self.max_output_tokens, int) or self.max_output_tokens < 1 + ): + raise ConfigurationError("max_output_tokens must be positive") + + +@dataclass(frozen=True) +class CompletionRequest: + prompt: str + role: Role + model: str | None = None + output_schema: Mapping[str, object] | None = None + json_mode: bool = False + timeout_seconds: float | None = None + sampling: SamplingOptions | None = None + system_prompt: str | None = None + prompt_contract_version: str = "1" + schema_contract_version: str = "1" + + def __post_init__(self) -> None: + if not isinstance(self.prompt, str) or not self.prompt: + raise ConfigurationError("prompt must be a non-empty string") + if self.role not in ("proposer", "judge", "analysis", "score", "validation"): + raise ConfigurationError("unsupported completion role") + if self.model is not None and (not isinstance(self.model, str) or not self.model.strip()): + raise ConfigurationError("model must be a non-empty string") + if self.timeout_seconds is not None and ( + not math.isfinite(self.timeout_seconds) or self.timeout_seconds <= 0 + ): + raise ConfigurationError("timeout must be finite and positive") + if self.output_schema is not None: + if not isinstance(self.output_schema, Mapping): + raise ConfigurationError("output schema must be a mapping") + object.__setattr__(self, "output_schema", _freeze(self.output_schema)) + if not isinstance(self.json_mode, bool): + raise ConfigurationError("json_mode must be a boolean") + if self.sampling is not None and not isinstance(self.sampling, SamplingOptions): + raise ConfigurationError("sampling must be SamplingOptions") + + +@dataclass(frozen=True) +class Usage: + input_tokens: int | None = None + output_tokens: int | None = None + total_tokens: int | None = None + cost: float | None = None + + +@dataclass(frozen=True) +class FallbackRecord: + source_backend: str + reason: str + switched_at: str + + +@dataclass(frozen=True) +class CompletionResult: + text: str + structured: JsonValue | None + requested_backend: str + actual_backend: str + requested_model: str | None + actual_model: str | None + auth_class: AuthClass + auth_source: AuthSource + usage: Usage | None = None + fallback: FallbackRecord | None = None + role: Role | None = None + started_at: str | None = None + duration_seconds: float | None = None + retry_count: int = 0 + prompt_contract_version: str = "1" + schema_contract_version: str = "1" + + def __post_init__(self) -> None: + if self.structured is not None: + object.__setattr__(self, "structured", _freeze(self.structured)) + + +@dataclass(frozen=True) +class BackendSpec: + backend: BackendName = "api" + model: str | None = None + api_base: str | None = None + api_fallback: bool = False + api_fallback_model: str | None = None + max_concurrency: int = 1 + + def __post_init__(self) -> None: + if self.backend not in ("api", "codex", "claude"): + raise ConfigurationError("unsupported backend") + if self.model is not None and (not isinstance(self.model, str) or not self.model.strip()): + raise ConfigurationError("model must be a non-empty string") + if self.max_concurrency < 1: + raise ConfigurationError("max_concurrency must be positive") + + +@dataclass(frozen=True) +class BackendCapabilities: + structured_output: bool = True + model_override: bool = True + sampling: bool = True + cancellation: bool = False + usage_reporting: bool = False + + +@dataclass(frozen=True) +class BackendStatus: + ready: bool + backend: str + auth_class: AuthClass + auth_source: AuthSource | None = None + detail: str | None = None + + +class BackendError(Exception): + """A normalized backend failure; never attach provider exception payloads.""" + + category = "backend_error" + + +class BackendUnavailable(BackendError): + category = "backend_unavailable" + + +class AuthenticationError(BackendError): + category = "authentication" + + +class RateLimitError(BackendError): + category = "rate_limit" + + +class QuotaExceeded(BackendError): + category = "quota_exceeded" + + +class Timeout(BackendError): + category = "timeout" + + +class Cancelled(BackendError): + category = "cancelled" + + +class InvalidResponse(BackendError): + category = "invalid_response" + + +class ConfigurationError(BackendError): + category = "configuration" + + +class CompletionBackend(Protocol): + capabilities: BackendCapabilities + + def preflight(self) -> BackendStatus: ... + + def complete(self, request: CompletionRequest) -> CompletionResult: ... + + +def validate_capabilities(request: CompletionRequest, capabilities: BackendCapabilities) -> None: + """Reject unsupported controls before any provider dispatch.""" + if request.output_schema is not None and not capabilities.structured_output: + raise ConfigurationError("backend does not support structured output") + if request.model is not None and not capabilities.model_override: + raise ConfigurationError("backend does not support model override") + if request.sampling is not None and not capabilities.sampling: + raise ConfigurationError("backend does not support sampling controls") diff --git a/src/optimize_anything/llm_backends/claude_backend.py b/src/optimize_anything/llm_backends/claude_backend.py new file mode 100644 index 0000000..af99943 --- /dev/null +++ b/src/optimize_anything/llm_backends/claude_backend.py @@ -0,0 +1,326 @@ +"""Local, isolated Claude Code subscription completion adapter.""" + +from __future__ import annotations + +import json +import os +import re +import selectors +import signal +import shutil +import subprocess +import tempfile +import time +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import BinaryIO, Callable, Mapping, Sequence, cast + +from jsonschema import Draft202012Validator, SchemaError, ValidationError # type: ignore[import-untyped] + +from .base import ( + AuthenticationError, + BackendCapabilities, + BackendStatus, + BackendUnavailable, + CompletionRequest, + CompletionResult, + ConfigurationError, + InvalidResponse, + Timeout, + Usage, + thaw_json, + validate_capabilities, +) +from .schema import reject_external_refs + +_MIN_VERSION = (2, 1, 278) +_MAX_OUTPUT_BYTES = 1_048_576 +_MAX_PROMPT_BYTES = 9_000_000 +_TRANSPORT_SCHEMA = json.dumps( + { + "type": "object", + "properties": {"payload": {"type": "string"}}, + "required": ["payload"], + "additionalProperties": False, + }, + separators=(",", ":"), +) +_REQUIRED_FLAGS = ( + "--safe-mode", + "--tools", + "--disable-slash-commands", + "--strict-mcp-config", + "--mcp-config", + "--no-session-persistence", + "--output-format", + "--json-schema", + "--permission-prompts", +) +_BLOCKED_ENV_NAMES = { + "CLAUDECODE", + "CLAUDE_CODE_OAUTH_TOKEN", + "CLAUDE_CODE_API_KEY_HELPER", + "CLAUDE_CODE_USE_BEDROCK", + "CLAUDE_CODE_USE_VERTEX", + "CLAUDE_CODE_USE_FOUNDRY", + "CLAUDE_CODE_USE_ANTHROPIC_API", + "API_KEY_HELPER", +} +_BLOCKED_ENV_PREFIXES = ( + "ANTHROPIC_", + "CLAUDE_CODE_", + "AWS_", + "GOOGLE_", + "CLOUDSDK_", + "GCP_", + "VERTEX_", + "BEDROCK_", + "AZURE_", + "FOUNDRY_", +) + + +@dataclass(frozen=True) +class _ProcessOutput: + returncode: int + stdout: bytes + stderr: bytes + + +def _subscription_env(source: Mapping[str, str]) -> dict[str, str]: + """Retain saved-login discovery while removing paid-auth overrides.""" + return { + key: value + for key, value in source.items() + if key.upper() not in _BLOCKED_ENV_NAMES + and not key.upper().startswith(_BLOCKED_ENV_PREFIXES) + } + + +def _terminate(process: subprocess.Popen[bytes]) -> None: + try: + os.killpg(process.pid, signal.SIGTERM) + except (AttributeError, ProcessLookupError): + if process.poll() is None: + process.terminate() + try: + if process.poll() is None: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + pass + try: + os.killpg(process.pid, signal.SIGKILL) + except (AttributeError, ProcessLookupError): + if process.poll() is None: + process.kill() + if process.poll() is None: + process.wait(timeout=2) + + +def _run_bounded( + argv: Sequence[str], *, stdin: bytes, env: Mapping[str, str], cwd: str, + timeout: float, max_output: int, +) -> _ProcessOutput: + """Drain pipes incrementally and stop the child before buffers exceed the cap.""" + process = subprocess.Popen( + argv, cwd=cwd, env=dict(env), stdin=subprocess.PIPE, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True, + ) + assert process.stdin and process.stdout and process.stderr + selector = selectors.DefaultSelector() + buffers = {process.stdout: bytearray(), process.stderr: bytearray()} + deadline = time.monotonic() + timeout + written = 0 + try: + for stream in (process.stdin, process.stdout, process.stderr): + os.set_blocking(stream.fileno(), False) + selector.register(process.stdout, selectors.EVENT_READ) + selector.register(process.stderr, selectors.EVENT_READ) + if stdin: + selector.register(process.stdin, selectors.EVENT_WRITE) + else: + process.stdin.close() + while selector.get_map(): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise Timeout("Claude completion timed out") + for key, _ in selector.select(min(remaining, 0.25)): + stream = cast(BinaryIO, key.fileobj) + if stream is process.stdin: + try: + written += os.write(stream.fileno(), stdin[written:written + 65536]) + except BrokenPipeError: + written = len(stdin) + if written >= len(stdin): + selector.unregister(stream) + stream.close() + else: + chunk = os.read(stream.fileno(), 65536) + if not chunk: + selector.unregister(stream) + stream.close() + continue + buffers[stream].extend(chunk) + if sum(map(len, buffers.values())) > max_output: + raise InvalidResponse("Claude output exceeded the configured limit") + remaining = deadline - time.monotonic() + if remaining <= 0: + raise Timeout("Claude completion timed out") + try: + returncode = process.wait(timeout=remaining) + except subprocess.TimeoutExpired: + raise Timeout("Claude completion timed out") from None + return _ProcessOutput(returncode, bytes(buffers[process.stdout]), bytes(buffers[process.stderr])) + finally: + selector.close() + _terminate(process) + + +class ClaudeCliBackend: + """Runs one no-tools Claude print request using the user's saved login.""" + + capabilities = BackendCapabilities( + structured_output=True, model_override=True, sampling=False, + cancellation=True, usage_reporting=True, + ) + + def __init__( + self, *, executable: str = "claude", model: str | None = None, + runner: Callable[..., _ProcessOutput] = _run_bounded, + environ: Mapping[str, str] | None = None, + ) -> None: + self.executable = executable + self.model = model + self._runner = runner + self._environ = environ + + def _env(self) -> dict[str, str]: + return _subscription_env(os.environ if self._environ is None else self._environ) + + def _run(self, argv: Sequence[str], *, stdin: bytes = b"", cwd: str, timeout: float, + max_output: int = 65536) -> _ProcessOutput: + try: + return self._runner( + argv, stdin=stdin, env=self._env(), cwd=cwd, + timeout=timeout, max_output=max_output, + ) + except (Timeout, InvalidResponse): + raise + except (OSError, subprocess.SubprocessError): + raise BackendUnavailable("Claude Code could not be started") from None + + def preflight(self) -> BackendStatus: + executable = shutil.which(self.executable) if os.sep not in self.executable else self.executable + if not executable or not Path(executable).is_file(): + raise BackendUnavailable("Install Claude Code and sign in with `claude auth login`") + with tempfile.TemporaryDirectory(prefix="optimize-claude-preflight-") as workspace: + version = self._run([executable, "--version"], cwd=workspace, timeout=10) + match = re.search(rb"(\d+)\.(\d+)\.(\d+)", version.stdout) + if version.returncode or not match or tuple(map(int, match.groups())) < _MIN_VERSION: + raise BackendUnavailable("Claude Code 2.1.278 or newer is required") + help_output = self._run([executable, "--help"], cwd=workspace, timeout=10) + help_text = help_output.stdout.decode("utf-8", "replace") + # --safe-mode is documented but hidden from some --help versions. + required = tuple(flag for flag in _REQUIRED_FLAGS if flag != "--safe-mode") + if help_output.returncode or any(flag not in help_text for flag in required): + raise BackendUnavailable("Claude Code lacks required isolation flags") + auth = self._run([executable, "auth", "status"], cwd=workspace, timeout=10) + if auth.returncode: + raise AuthenticationError("Sign in to Claude Code with a Claude subscription") + try: + status = json.loads(auth.stdout) + except (ValueError, UnicodeDecodeError): + raise BackendUnavailable("Claude auth status was not valid JSON") from None + if not isinstance(status, dict) or not status.get("loggedIn"): + raise AuthenticationError("Sign in to Claude Code with a Claude subscription") + if status.get("authMethod") != "claude.ai" or status.get("apiProvider") != "firstParty": + raise AuthenticationError("Claude Code is not using a Claude subscription") + return BackendStatus(True, "claude", "subscription", "claude_subscription") + + def complete(self, request: CompletionRequest) -> CompletionResult: + validate_capabilities(request, self.capabilities) + if request.sampling is not None: + raise ConfigurationError("Claude subscription does not support sampling controls") + model = request.model or self.model + if model is not None and not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._:-]{0,127}", model): + raise ConfigurationError("invalid Claude model name") + schema = thaw_json(request.output_schema) if request.output_schema is not None else None + if schema is not None: + reject_external_refs(schema) + try: + Draft202012Validator.check_schema(schema) + except SchemaError: + raise ConfigurationError("invalid output schema") from None + prompt = request.prompt + if request.system_prompt: + prompt = f"System instruction:\n{request.system_prompt}\n\nTask:\n{prompt}" + if schema is not None: + prompt += "\n\nReturn a JSON value matching this schema inside the payload string:\n" + prompt += json.dumps(schema, ensure_ascii=False) + prompt_bytes = prompt.encode("utf-8") + if len(prompt_bytes) > _MAX_PROMPT_BYTES: + raise ConfigurationError("Claude input exceeds the supported size") + self.preflight() + started = datetime.now(timezone.utc) + begin = time.monotonic() + with tempfile.TemporaryDirectory(prefix="optimize-claude-") as workspace: + mcp_path = Path(workspace, "mcp.json") + fd = os.open(mcp_path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "w", encoding="utf-8") as file: + file.write('{"mcpServers":{}}') + argv = [ + self.executable, "-p", "--safe-mode", "--tools", "", + "--disable-slash-commands", + "--strict-mcp-config", "--mcp-config", str(mcp_path), + "--no-session-persistence", "--permission-mode", "dontAsk", + "--permission-prompts", "none", "--no-chrome", + "--output-format", "json", + ] + if model: + argv.extend(["--model", model]) + if schema is not None: + argv.extend(["--json-schema", _TRANSPORT_SCHEMA]) + output = self._run( + argv, stdin=prompt_bytes, cwd=workspace, + timeout=request.timeout_seconds or 120, max_output=_MAX_OUTPUT_BYTES, + ) + if output.returncode: + raise BackendUnavailable("Claude completion failed") + try: + response = json.loads(output.stdout) + except (UnicodeDecodeError, ValueError): + raise InvalidResponse("Claude returned malformed JSON") from None + if not isinstance(response, dict) or response.get("is_error"): + raise InvalidResponse("Claude returned an invalid completion") + structured = None + if schema is not None: + envelope = response.get("structured_output") + if not isinstance(envelope, dict) or not isinstance(envelope.get("payload"), str): + raise InvalidResponse("Claude omitted structured output") + try: + structured = json.loads(envelope["payload"]) + Draft202012Validator(schema).validate(structured) + except (ValueError, ValidationError, SchemaError): + raise InvalidResponse("Claude output did not match the requested schema") from None + text = json.dumps(structured, ensure_ascii=False) + else: + raw_text = response.get("result") + if not isinstance(raw_text, str) or not raw_text: + raise InvalidResponse("Claude omitted completion text") + text = raw_text + usage_data = response.get("usage") or {} + usage = Usage( + input_tokens=usage_data.get("input_tokens"), + output_tokens=usage_data.get("output_tokens"), + ) if isinstance(usage_data, dict) else None + return CompletionResult( + text=text, structured=structured, requested_backend="claude", + actual_backend="claude", requested_model=model, + actual_model=response.get("model") or model, + auth_class="subscription", auth_source="claude_subscription", + usage=usage, role=request.role, started_at=started.isoformat(), + duration_seconds=time.monotonic() - begin, + prompt_contract_version=request.prompt_contract_version, + schema_contract_version=request.schema_contract_version, + ) diff --git a/src/optimize_anything/llm_backends/codex_backend.py b/src/optimize_anything/llm_backends/codex_backend.py new file mode 100644 index 0000000..cd51457 --- /dev/null +++ b/src/optimize_anything/llm_backends/codex_backend.py @@ -0,0 +1,252 @@ +"""Isolated single-turn completion through the optional Codex Python SDK.""" + +from __future__ import annotations + +import importlib +import os +import re +import tempfile +import time +from concurrent.futures import ThreadPoolExecutor, TimeoutError as FutureTimeout +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +from jsonschema import Draft202012Validator, SchemaError, ValidationError # type: ignore[import-untyped] + +from .base import ( + AuthenticationError, + BackendCapabilities, + BackendStatus, + BackendUnavailable, + Cancelled, + CompletionRequest, + CompletionResult, + ConfigurationError, + InvalidResponse, + RateLimitError, + Timeout, + Usage, + thaw_json, + validate_capabilities, +) +from .schema import reject_external_refs + +_SDK_VERSION = "0.156.0" +_MAX_OUTPUT_BYTES = 1_048_576 +_MAX_PROMPT_BYTES = 9_000_000 +_ISOLATION_OVERRIDES = ( + "project_doc_max_bytes=0", + "project_doc_fallback_filenames=[]", + "web_search=\"disabled\"", + "model_max_output_tokens=8192", + "features.shell_tool=false", + "features.unified_exec=false", + "features.code_mode_host=false", + "features.view_image=false", + "features.multi_agent_v2=false", + "features.multi_agent=false", + "features.apps=false", + "features.plugins=false", + "features.plugin_sharing=false", + "features.remote_plugin=false", + "features.computer_use=false", + "features.image_generation=false", + "features.browser_use=false", + "features.browser_use_external=false", + "features.in_app_browser=false", + "features.workspace_dependencies=false", + "features.skill_mcp_dependency_install=false", + "include_apply_patch_tool=false", + "mcp_servers={}", +) + + +def _sdk() -> Any: + try: + return importlib.import_module("openai_codex") + except ImportError: + raise BackendUnavailable( + "Install the Codex backend with `pip install 'optimize-anything[codex]'`" + ) from None + + +def _account_type(status: Any) -> str | None: + account = getattr(status, "account", None) + if account is None: + return None + root = getattr(account, "root", account) + return getattr(root, "type", None) + + +class CodexSdkBackend: + """Creates a private, read-only ephemeral Codex thread for each request.""" + + capabilities = BackendCapabilities( + structured_output=True, model_override=True, sampling=False, + cancellation=True, usage_reporting=True, + ) + + def __init__( + self, *, model: str | None = None, sdk_module: Any = None, + auth_path: Path | None = None, + ) -> None: + self.model = model + self._sdk_module = sdk_module + saved_home = Path(os.environ.get("CODEX_HOME", str(Path.home() / ".codex"))) + self._auth_path = auth_path or saved_home / "auth.json" + + def _module(self) -> Any: + sdk = self._sdk_module if self._sdk_module is not None else _sdk() + if getattr(sdk, "__version__", None) != _SDK_VERSION: + raise BackendUnavailable("Codex SDK 0.156.0 is required for the isolation profile") + for name in ("Codex", "CodexConfig", "Sandbox", "ApprovalMode"): + if not hasattr(sdk, name): + raise BackendUnavailable("Codex SDK lacks required isolation controls") + if not hasattr(sdk.Sandbox, "read_only") or not hasattr(sdk.ApprovalMode, "deny_all"): + raise BackendUnavailable("Codex SDK lacks read-only or deny-all controls") + return sdk + + def _client(self, sdk: Any, workspace: str, private_home: str) -> Any: + config = sdk.CodexConfig( + cwd=workspace, config_overrides=_ISOLATION_OVERRIDES, + env={ + "CODEX_HOME": private_home, + "OPENAI_API_KEY": "", + "CODEX_API_KEY": "", + }, + ) + return sdk.Codex(config=config) + + def _prepare_private_home(self, private_home: str) -> None: + if not self._auth_path.is_file() or self._auth_path.is_symlink(): + raise BackendUnavailable("Saved Codex login was not found; run `codex login`") + (Path(private_home) / "auth.json").symlink_to(self._auth_path) + + @staticmethod + def _check_account(client: Any) -> None: + try: + status = client.account() + except Exception: + raise BackendUnavailable("Could not verify saved Codex authentication") from None + if _account_type(status) != "chatgpt": + raise AuthenticationError("Sign in to Codex through ChatGPT with `codex login`") + + def preflight(self) -> BackendStatus: + sdk = self._module() + try: + with tempfile.TemporaryDirectory(prefix="optimize-codex-preflight-") as workspace, tempfile.TemporaryDirectory(prefix="optimize-codex-home-") as private_home: + self._prepare_private_home(private_home) + with self._client(sdk, workspace, private_home) as client: + self._check_account(client) + except (AuthenticationError, BackendUnavailable): + raise + except Exception: + raise BackendUnavailable("Codex isolation preflight failed") from None + return BackendStatus(True, "codex", "subscription", "chatgpt") + + def complete(self, request: CompletionRequest) -> CompletionResult: + validate_capabilities(request, self.capabilities) + if request.sampling is not None: + raise ConfigurationError("Codex subscription does not support sampling controls") + sdk = self._module() + model = request.model or self.model + if model is not None and not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._:-]{0,127}", model): + raise ConfigurationError("invalid Codex model name") + schema = thaw_json(request.output_schema) if request.output_schema is not None else None + if schema is not None: + reject_external_refs(schema) + try: + Draft202012Validator.check_schema(schema) + except SchemaError: + raise ConfigurationError("invalid output schema") from None + prompt = request.prompt + if request.system_prompt: + prompt = f"System instruction:\n{request.system_prompt}\n\nTask:\n{prompt}" + if len(prompt.encode("utf-8")) > _MAX_PROMPT_BYTES: + raise ConfigurationError("Codex input exceeds the supported size") + started = datetime.now(timezone.utc) + begin = time.monotonic() + actual_model = model + try: + with tempfile.TemporaryDirectory(prefix="optimize-codex-") as workspace, tempfile.TemporaryDirectory(prefix="optimize-codex-home-") as private_home: + if list(Path(workspace).iterdir()): + raise BackendUnavailable("Codex workspace was not empty") + self._prepare_private_home(private_home) + with self._client(sdk, workspace, private_home) as client: + self._check_account(client) + thread = client.thread_start( + cwd=workspace, ephemeral=True, + sandbox=sdk.Sandbox.read_only, + approval_mode=sdk.ApprovalMode.deny_all, + base_instructions="Return only the requested answer. Do not use tools.", + developer_instructions="Do not access files, network tools, or external services.", + config={"project_doc_max_bytes": 0, "web_search": "disabled"}, + model=model, + ) + handle = thread.turn( + prompt, output_schema=schema, + sandbox=sdk.Sandbox.read_only, + approval_mode=sdk.ApprovalMode.deny_all, + ) + executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="codex-turn") + future = executor.submit(handle.run) + try: + result = future.result(timeout=request.timeout_seconds or 120) + except FutureTimeout: + try: + handle.interrupt() + except Exception: + pass + try: + future.result(timeout=5) + except Exception: + pass + raise Timeout("Codex completion timed out") from None + finally: + executor.shutdown(wait=False, cancel_futures=True) + try: + actual_model = getattr(thread.read().thread, "model", None) or model + except Exception: + pass + except (AuthenticationError, BackendUnavailable, Timeout): + raise + except Exception as exc: + if hasattr(sdk, "ServerBusyError") and isinstance(exc, sdk.ServerBusyError): + raise RateLimitError("Codex is temporarily rate limited") from None + raise BackendUnavailable("Codex completion failed") from None + status = getattr(result, "status", None) + status_value = getattr(status, "value", status) + if status_value == "interrupted": + raise Cancelled("Codex turn was cancelled") + if getattr(result, "error", None) is not None or status_value != "completed": + raise BackendUnavailable("Codex turn did not complete") + text = getattr(result, "final_response", None) + if not isinstance(text, str) or not text: + raise InvalidResponse("Codex omitted completion text") + if len(text.encode("utf-8")) > _MAX_OUTPUT_BYTES: + raise InvalidResponse("Codex output exceeded the configured limit") + structured = None + if schema is not None: + import json + + try: + structured = json.loads(text) + Draft202012Validator(schema).validate(structured) + except (ValueError, ValidationError, SchemaError): + raise InvalidResponse("Codex output did not match the requested schema") from None + raw_usage = getattr(getattr(result, "usage", None), "total", None) + usage = Usage( + input_tokens=getattr(raw_usage, "input_tokens", None), + output_tokens=getattr(raw_usage, "output_tokens", None), + total_tokens=getattr(raw_usage, "total_tokens", None), + ) if raw_usage is not None else None + return CompletionResult( + text=text, structured=structured, requested_backend="codex", + actual_backend="codex", requested_model=model, actual_model=actual_model, + auth_class="subscription", auth_source="chatgpt", usage=usage, + role=request.role, started_at=started.isoformat(), + duration_seconds=time.monotonic() - begin, + prompt_contract_version=request.prompt_contract_version, + schema_contract_version=request.schema_contract_version, + ) diff --git a/src/optimize_anything/llm_backends/coordination.py b/src/optimize_anything/llm_backends/coordination.py new file mode 100644 index 0000000..73bc48c --- /dev/null +++ b/src/optimize_anything/llm_backends/coordination.py @@ -0,0 +1,247 @@ +"""Private run-scoped coordination shared by optimization child processes.""" + +from __future__ import annotations + +import fcntl +import json +import math +import os +import re +import shutil +import sys +import tempfile +import time +import uuid +from contextlib import contextmanager +from pathlib import Path +from typing import Any, Iterator, Mapping + +from .base import ConfigurationError, Timeout + +COORDINATION_DIR_ENV = "OPTIMIZE_ANYTHING_COORDINATION_DIR" +COORDINATION_ID_ENV = "OPTIMIZE_ANYTHING_COORDINATION_ID" +_PROVIDERS = ("codex", "claude") +_ROLES = ("proposer", "judge", "analysis", "score", "validation") +_EVENT_KEYS = frozenset({ + "run_id", "call_id", "role", "requested_backend", "actual_backend", + "requested_model", "actual_model", "auth_class", "auth_source", "started_at", + "duration_seconds", "input_tokens", "output_tokens", "total_tokens", "retry_count", + "fallback_source", "fallback_reason", "prompt_contract_version", "schema_contract_version", +}) +_SAFE_VALUE = re.compile(r"^[A-Za-z0-9_./:+@-]{1,128}$") +_MAX_EVENT_BYTES = 4096 +_MAX_LOG_BYTES = 1_048_576 + + +class _SlotLease: + def __init__(self, fd: int) -> None: + self._fd = fd + + def release(self) -> None: + if self._fd >= 0: + fcntl.flock(self._fd, fcntl.LOCK_UN) + os.close(self._fd) + self._fd = -1 + + def __enter__(self) -> _SlotLease: + return self + + def __exit__(self, *_args: object) -> None: + self.release() + + +class RunCoordinator: + """OS-lock-backed provider slots, sticky role circuits, and safe events.""" + + def __init__(self, path: Path, *, owner: bool = False) -> None: + self.path = path + self._owner = owner + self._verify_private_path(path) + try: + metadata = json.loads((path / "run.json").read_text()) + except (OSError, ValueError) as exc: + raise ConfigurationError("invalid coordination directory") from exc + self.run_id: str = metadata["run_id"] + self.capacities: dict[str, int] = metadata["capacities"] + + @staticmethod + def _verify_private_path(path: Path) -> None: + try: + info = path.lstat() + except OSError as exc: + raise ConfigurationError("coordination directory is unavailable") from exc + if not path.is_dir() or path.is_symlink() or info.st_uid != os.getuid() or info.st_mode & 0o077: + raise ConfigurationError("coordination directory is not private") + + @classmethod + def create( + cls, capacities: Mapping[str, int] | None = None, *, + max_concurrency: Mapping[str, int] | None = None, + ) -> RunCoordinator: + selected = capacities or max_concurrency or {} + if any(key not in _PROVIDERS or not isinstance(value, int) or value < 1 + for key, value in selected.items()): + raise ConfigurationError("invalid provider capacity") + capacity = {provider: selected.get(provider, 1) for provider in _PROVIDERS} + for provider, count in capacity.items(): + if count != 1: + print(f"Warning: {provider} subscription concurrency set to {count}.", file=sys.stderr) + path = Path(tempfile.mkdtemp(prefix="optimize-anything-run-")) + os.chmod(path, 0o700) + run_id = uuid.uuid4().hex + (path / "run.json").write_text(json.dumps({"run_id": run_id, "capacities": capacity})) + (path / "circuits.json").write_text("{}") + (path / "events.jsonl").touch(mode=0o600) + (path / "state.lock").touch(mode=0o600) + (path / "events.lock").touch(mode=0o600) + for provider, count in capacity.items(): + for index in range(count): + (path / f"slot-{provider}-{index}.lock").touch(mode=0o600) + return cls(path, owner=True) + + @classmethod + def attach(cls, path: str | os.PathLike[str], *, expected_run_id: str | None = None) -> RunCoordinator: + coordinator = cls(Path(path)) + if expected_run_id is not None and coordinator.run_id != expected_run_id: + raise ConfigurationError("coordination run identifier mismatch") + return coordinator + + @classmethod + def from_environment(cls) -> RunCoordinator | None: + path = os.environ.get(COORDINATION_DIR_ENV) + run_id = os.environ.get(COORDINATION_ID_ENV) + if not path and not run_id: + return None + if not path or not run_id: + raise ConfigurationError("incomplete coordination environment") + return cls.attach(path, expected_run_id=run_id) + + def child_environment(self) -> dict[str, str]: + return {COORDINATION_DIR_ENV: str(self.path), COORDINATION_ID_ENV: self.run_id} + + @contextmanager + def exported_environment(self) -> Iterator[None]: + """Temporarily expose this run to evaluator child processes.""" + previous: dict[str, str | None] = {} + for key, value in self.child_environment().items(): + previous[key] = os.environ.get(key) + os.environ[key] = value + try: + yield + finally: + for key, previous_value in previous.items(): + if previous_value is None: + os.environ.pop(key, None) + else: + os.environ[key] = previous_value + + def __enter__(self) -> RunCoordinator: + return self + + def __exit__(self, *_args: object) -> None: + self.close() + + def close(self) -> None: + if self._owner: + shutil.rmtree(self.path) + self._owner = False + + def _provider(self, provider: str) -> None: + if provider not in _PROVIDERS: + raise ConfigurationError("unsupported subscription provider") + + def try_acquire_slot(self, provider: str) -> _SlotLease | None: + self._provider(provider) + for index in range(self.capacities[provider]): + fd = os.open(self.path / f"slot-{provider}-{index}.lock", os.O_RDWR | os.O_NOFOLLOW) + try: + fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + os.close(fd) + continue + return _SlotLease(fd) + return None + + @contextmanager + def slot(self, provider: str, *, timeout_seconds: float | None = None) -> Iterator[None]: + deadline = None if timeout_seconds is None else time.monotonic() + timeout_seconds + lease = self.try_acquire_slot(provider) + while lease is None: + if deadline is not None and time.monotonic() >= deadline: + raise Timeout("provider slot wait timed out") + time.sleep(0.02) + lease = self.try_acquire_slot(provider) + try: + yield + finally: + lease.release() + + @contextmanager + def _lock(self, name: str) -> Iterator[None]: + fd = os.open(self.path / name, os.O_RDWR | os.O_NOFOLLOW) + try: + fcntl.flock(fd, fcntl.LOCK_EX) + yield + finally: + fcntl.flock(fd, fcntl.LOCK_UN) + os.close(fd) + + def circuit_reason(self, provider: str, role: str) -> str | None: + self._provider(provider) + if role not in _ROLES: + raise ConfigurationError("unsupported completion role") + with self._lock("state.lock"): + state = json.loads((self.path / "circuits.json").read_text()) + return state.get(f"{provider}:{role}") + + def open_circuit(self, provider: str, role: str, reason: str) -> bool: + self._provider(provider) + if role not in _ROLES or not _SAFE_VALUE.fullmatch(reason): + raise ConfigurationError("invalid circuit state") + with self._lock("state.lock"): + state_path = self.path / "circuits.json" + state = json.loads(state_path.read_text()) + key = f"{provider}:{role}" + if key in state: + return False + state[key] = reason + fd, temporary = tempfile.mkstemp(prefix="circuits-", dir=self.path) + try: + with os.fdopen(fd, "w") as stream: + json.dump(state, stream) + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, state_path) + finally: + if os.path.exists(temporary): + os.unlink(temporary) + return True + + def record_event(self, event: Mapping[str, Any]) -> bool: + """Append whitelisted, bounded non-content metadata; drop unsafe values.""" + safe: dict[str, Any] = {"run_id": self.run_id} + for key in _EVENT_KEYS - {"run_id"}: + value = event.get(key) + if isinstance(value, str) and _SAFE_VALUE.fullmatch(value): + safe[key] = value + elif isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value): + safe[key] = value + line = json.dumps(safe, sort_keys=True, separators=(",", ":")) + "\n" + data = line.encode("utf-8") + if len(data) > _MAX_EVENT_BYTES: + return False + with self._lock("events.lock"): + path = self.path / "events.jsonl" + if path.stat().st_size + len(data) > _MAX_LOG_BYTES: + return False + fd = os.open(path, os.O_WRONLY | os.O_APPEND | os.O_NOFOLLOW) + try: + os.write(fd, data) + os.fsync(fd) + finally: + os.close(fd) + return True + + def events(self) -> list[dict[str, Any]]: + with self._lock("events.lock"): + return [json.loads(line) for line in (self.path / "events.jsonl").read_text().splitlines()] diff --git a/src/optimize_anything/llm_backends/factory.py b/src/optimize_anything/llm_backends/factory.py new file mode 100644 index 0000000..bc35960 --- /dev/null +++ b/src/optimize_anything/llm_backends/factory.py @@ -0,0 +1,137 @@ +"""Resolve immutable backend specifications into completion adapters.""" + +from __future__ import annotations + +import json +from dataclasses import dataclass, field +from typing import Any + +from .base import ( + BackendCapabilities, + BackendName, + BackendSpec, + BackendStatus, + CompletionBackend, + CompletionRequest, + CompletionResult, + Role, +) +from .coordination import RunCoordinator +from .fallback import FallbackBackend +from .litellm_backend import LiteLLMBackend +from .provenance import completion_event + + +def resolve_backend_spec( + *, + backend: BackendName, + model: str | None, + api_base: str | None = None, + no_api_fallback: bool = False, + openai_api_fallback_model: str | None = None, + anthropic_api_fallback_model: str | None = None, + max_concurrency: int = 1, +) -> BackendSpec: + """Resolve CLI-style provider-wide fallback options into a backend spec.""" + fallback_model = None + if backend == "codex": + fallback_model = openai_api_fallback_model + if fallback_model is None and model and model.startswith("openai/"): + fallback_model = model + elif backend == "claude": + fallback_model = anthropic_api_fallback_model + if fallback_model is None and model and model.startswith("anthropic/"): + fallback_model = model + return BackendSpec( + backend=backend, + model=model, + api_base=api_base, + api_fallback=backend != "api" and not no_api_fallback and fallback_model is not None, + api_fallback_model=fallback_model, + max_concurrency=max_concurrency, + ) + + +def create_backend( + spec: BackendSpec, + *, + role: Role, + coordinator: RunCoordinator | None = None, +) -> CompletionBackend: + """Create one role backend without probing or dispatching it.""" + if spec.backend == "api": + return LiteLLMBackend(model=spec.model, api_base=spec.api_base) + + if coordinator is None: + coordinator = RunCoordinator.from_environment() + if spec.backend == "codex": + from .codex_backend import CodexSdkBackend + + primary: CompletionBackend = CodexSdkBackend(model=spec.model) + else: + from .claude_backend import ClaudeCliBackend + + primary = ClaudeCliBackend(model=spec.model) + + if not spec.api_fallback: + return _CoordinatedBackend(primary, spec.backend, coordinator) + + fallback = LiteLLMBackend(model=spec.api_fallback_model, api_base=spec.api_base) + return FallbackBackend( + primary=primary, + fallback=fallback, + source_backend=spec.backend, + fallback_model=spec.api_fallback_model, + coordinator=coordinator, + ) + + +@dataclass +class _CoordinatedBackend: + """Apply provider slots even when API fallback is disabled.""" + + backend: CompletionBackend + provider: str + coordinator: RunCoordinator | None + capabilities: BackendCapabilities = field(init=False) + + def __post_init__(self) -> None: + self.capabilities = self.backend.capabilities + + def preflight(self) -> BackendStatus: + return self.backend.preflight() + + def complete(self, request: CompletionRequest) -> CompletionResult: + if self.coordinator is None: + return self.backend.complete(request) + with self.coordinator.slot(self.provider, timeout_seconds=request.timeout_seconds): + result = self.backend.complete(request) + self.coordinator.record_event(completion_event(result)) + return result + + +class BackendLanguageModel: + """Adapt a completion backend to GEPA's callable language-model protocol.""" + + def __init__( + self, + backend: CompletionBackend, + *, + model: str | None = None, + timeout_seconds: float = 120.0, + ) -> None: + self.backend = backend + self.model = model + self.timeout_seconds = timeout_seconds + + def __call__(self, prompt: str | list[dict[str, Any]]) -> str: + if not isinstance(prompt, str): + prompt = json.dumps(prompt, ensure_ascii=False) + return self.backend.complete( + CompletionRequest( + prompt=prompt, + role="proposer", + model=self.model, + timeout_seconds=self.timeout_seconds, + ) + ).text diff --git a/src/optimize_anything/llm_backends/fallback.py b/src/optimize_anything/llm_backends/fallback.py new file mode 100644 index 0000000..f46eaf5 --- /dev/null +++ b/src/optimize_anything/llm_backends/fallback.py @@ -0,0 +1,170 @@ +"""Conservative, visible subscription-to-API continuity policy.""" + +from __future__ import annotations + +import os +import sys +import threading +import time +from dataclasses import replace +from datetime import datetime, timezone +from typing import Callable + +from .base import ( + AuthenticationError, + BackendCapabilities, + BackendStatus, + BackendUnavailable, + CompletionBackend, + CompletionRequest, + CompletionResult, + ConfigurationError, + FallbackRecord, + QuotaExceeded, + RateLimitError, + Timeout, +) +from .coordination import RunCoordinator +from .provenance import completion_event + +_ELIGIBLE = (BackendUnavailable, AuthenticationError, RateLimitError, QuotaExceeded) + + +def fallback_ready(source_backend: str, model: str | None) -> bool: + """Check only canonical key presence, never read or retain its value.""" + if source_backend == "codex": + return bool(model and model.startswith("openai/") and os.environ.get("OPENAI_API_KEY")) + if source_backend == "claude": + return bool(model and model.startswith("anthropic/") and os.environ.get("ANTHROPIC_API_KEY")) + return False + + +_canonical_fallback_ready = fallback_ready + + +class FallbackBackend: + """Wrap a subscription adapter with per-role sticky same-vendor fallback.""" + + def __init__( + self, *, primary: CompletionBackend, fallback: CompletionBackend, + source_backend: str, fallback_model: str | None, + fallback_ready: Callable[[], bool] | None = None, + coordinator: RunCoordinator | None = None, + ) -> None: + if source_backend not in ("codex", "claude"): + raise ConfigurationError("fallback source must be a subscription backend") + self.primary = primary + self.fallback = fallback + self.source_backend = source_backend + self.fallback_model = fallback_model + self._ready = fallback_ready or (lambda: _canonical_fallback_ready(source_backend, fallback_model)) + self.coordinator = coordinator + self.capabilities = getattr(primary, "capabilities", BackendCapabilities()) + self._circuits: dict[str, str] = {} + self._lock = threading.Lock() + self._preflight_reason: str | None = None + + def _can_fallback(self) -> bool: + return self._same_vendor() and self._ready() + + def _same_vendor(self) -> bool: + prefix = "openai/" if self.source_backend == "codex" else "anthropic/" + return bool(self.fallback_model and self.fallback_model.startswith(prefix)) + + def _reason(self, role: str) -> str | None: + if self.coordinator: + return self.coordinator.circuit_reason(self.source_backend, role) + with self._lock: + return self._circuits.get(role) + + def _open(self, role: str, reason: str) -> bool: + if self.coordinator: + return self.coordinator.open_circuit(self.source_backend, role, reason) + with self._lock: + if role in self._circuits: + return False + self._circuits[role] = reason + return True + + def preflight(self) -> BackendStatus: + try: + return self.primary.preflight() + except _ELIGIBLE as exc: + if not self._can_fallback(): + raise + self._preflight_reason = exc.category + # Role-specific circuits are opened when each role is first used. + return BackendStatus(ready=True, backend=self.source_backend, auth_class="subscription", + detail=f"API fallback available after {exc.category}") + + def complete(self, request: CompletionRequest) -> CompletionResult: + reason = self._reason(request.role) + if reason is not None: + return self._complete_fallback(request, reason) + if self._preflight_reason is not None: + reason = self._preflight_reason + if self._open(request.role, reason): + self._warn(request.role, reason) + return self._complete_fallback(request, reason) + try: + if self.coordinator: + started = time.monotonic() + with self.coordinator.slot(self.source_backend, timeout_seconds=request.timeout_seconds): + reason = self._reason(request.role) + if reason is not None: + return self._complete_fallback(request, reason) + primary_request = request + if request.timeout_seconds is not None: + remaining = request.timeout_seconds - (time.monotonic() - started) + if remaining <= 0: + raise Timeout("subscription completion timed out waiting for a provider slot") + primary_request = replace(request, timeout_seconds=remaining) + result = self.primary.complete(primary_request) + else: + result = self.primary.complete(request) + if result.actual_backend != self.source_backend: + raise ConfigurationError("subscription adapter reported another backend") + result = replace( + result, role=request.role, + prompt_contract_version=request.prompt_contract_version, + schema_contract_version=request.schema_contract_version, + ) + if self.coordinator: + self.coordinator.record_event(completion_event(result)) + return result + except _ELIGIBLE as exc: + if not self._can_fallback(): + raise + opened = self._open(request.role, exc.category) + if opened: + self._warn(request.role, exc.category) + return self._complete_fallback(request, exc.category) + + def _warn(self, role: str, reason: str) -> None: + print( + f"Warning: {role} switched from {self.source_backend} to API model " + f"{self.fallback_model} after {reason}; API billing may apply.", + file=sys.stderr, + ) + + def _complete_fallback(self, request: CompletionRequest, reason: str) -> CompletionResult: + if not self._can_fallback(): + raise BackendUnavailable("API fallback is not ready") + api_request = replace(request, model=self.fallback_model) + result = self.fallback.complete(api_request) + if result.actual_backend != "api": + raise ConfigurationError("fallback adapter did not use API") + resolved = replace( + result, requested_backend=self.source_backend, + requested_model=request.model, + role=request.role, + prompt_contract_version=request.prompt_contract_version, + schema_contract_version=request.schema_contract_version, + fallback=FallbackRecord( + source_backend=self.source_backend, reason=reason, + switched_at=datetime.now(timezone.utc).isoformat(), + ), + ) + if self.coordinator: + self.coordinator.record_event(completion_event(resolved)) + return resolved diff --git a/src/optimize_anything/llm_backends/litellm_backend.py b/src/optimize_anything/llm_backends/litellm_backend.py new file mode 100644 index 0000000..11153cc --- /dev/null +++ b/src/optimize_anything/llm_backends/litellm_backend.py @@ -0,0 +1,234 @@ +"""API completion adapter with local JSON Schema validation.""" + +from __future__ import annotations + +import json +import math +import time +from datetime import datetime, timezone +from typing import Any, Callable, Mapping + +from .base import ( + AuthSource, + AuthenticationError, + BackendCapabilities, + BackendStatus, + BackendUnavailable, + Cancelled, + CompletionRequest, + CompletionResult, + ConfigurationError, + InvalidResponse, + QuotaExceeded, + RateLimitError, + Timeout, + Usage, + validate_capabilities, +) +from .schema import strip_code_fences + +_SCHEMA_KEYS = { + "type", "properties", "required", "additionalProperties", "items", "enum", "const", + "minimum", "maximum", "minLength", "maxLength", "minItems", "maxItems", "description", + "title", "default", "examples", "$schema", +} +_TYPES = {"object", "array", "string", "number", "integer", "boolean", "null"} + + +def _check_schema(schema: Mapping[str, Any]) -> None: + unknown = set(schema) - _SCHEMA_KEYS + if unknown: + raise ConfigurationError("unsupported JSON Schema keyword") + kind = schema.get("type") + if kind is not None and (not isinstance(kind, str) or kind not in _TYPES): + raise ConfigurationError("unsupported JSON Schema type") + properties = schema.get("properties", {}) + if not isinstance(properties, Mapping): + raise ConfigurationError("schema properties must be an object") + for child in properties.values(): + if not isinstance(child, Mapping): + raise ConfigurationError("schema property must be an object") + _check_schema(child) + items = schema.get("items") + if items is not None: + if not isinstance(items, Mapping): + raise ConfigurationError("schema items must be an object") + _check_schema(items) + required = schema.get("required", ()) + if not isinstance(required, (list, tuple)) or any(not isinstance(k, str) for k in required): + raise ConfigurationError("schema required must be a list of strings") + additional = schema.get("additionalProperties", True) + if not isinstance(additional, bool): + raise ConfigurationError("schema additionalProperties must be a boolean") + if "enum" in schema and not isinstance(schema["enum"], (list, tuple)): + raise ConfigurationError("schema enum must be an array") + for key in ("minimum", "maximum"): + value = schema.get(key) + if value is not None and (not isinstance(value, (int, float)) or isinstance(value, bool) + or not math.isfinite(value)): + raise ConfigurationError("schema numeric bound must be finite") + for key in ("minLength", "maxLength", "minItems", "maxItems"): + value = schema.get(key) + if value is not None and (not isinstance(value, int) or isinstance(value, bool) or value < 0): + raise ConfigurationError("schema size bound must be nonnegative") + + +def _valid(value: Any, schema: Mapping[str, Any]) -> bool: + kind = schema.get("type") + matches = { + "object": lambda: isinstance(value, dict), + "array": lambda: isinstance(value, list), + "string": lambda: isinstance(value, str), + "number": lambda: isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value), + "integer": lambda: isinstance(value, int) and not isinstance(value, bool), + "boolean": lambda: isinstance(value, bool), + "null": lambda: value is None, + } + if kind and not matches[kind](): + return False + if "const" in schema and value != schema["const"]: + return False + if "enum" in schema and value not in schema["enum"]: + return False + if isinstance(value, dict): + props = schema.get("properties", {}) + if any(key not in value for key in schema.get("required", ())): + return False + if schema.get("additionalProperties") is False and any(key not in props for key in value): + return False + if any(not _valid(item, props[key]) for key, item in value.items() if key in props): + return False + if isinstance(value, list): + if "items" in schema and any(not _valid(item, schema["items"]) for item in value): + return False + if len(value) < schema.get("minItems", 0) or len(value) > schema.get("maxItems", float("inf")): + return False + if isinstance(value, str): + if len(value) < schema.get("minLength", 0) or len(value) > schema.get("maxLength", float("inf")): + return False + if isinstance(value, (int, float)) and not isinstance(value, bool): + if value < schema.get("minimum", float("-inf")) or value > schema.get("maximum", float("inf")): + return False + return True + + +def validate_structured(text: str, schema: Mapping[str, Any]) -> Any: + """Parse and validate a response locally, without echoing provider content.""" + _check_schema(schema) + cleaned = strip_code_fences(text) + try: + parsed = json.loads(cleaned) + except (ValueError, TypeError): + raise InvalidResponse("provider returned malformed JSON") from None + if not _valid(parsed, schema): + raise InvalidResponse("provider response did not satisfy output schema") + return parsed + + +def _error_for(exc: Exception) -> Exception: + name = type(exc).__name__.lower() + if "cancel" in name: + return Cancelled("API completion cancelled") + if "timeout" in name: + return Timeout("API completion timed out") + if "authentication" in name or "permission" in name or "unauthorized" in name: + return AuthenticationError("API authentication failed") + if "quota" in name or "budget" in name: + return QuotaExceeded("API quota exceeded") + if "ratelimit" in name or "rate_limit" in name: + return RateLimitError("API rate limited") + if "badrequest" in name or "invalidrequest" in name: + return ConfigurationError("API rejected completion request") + return BackendUnavailable(f"API completion unavailable ({type(exc).__name__})") + + +def _usage(response: Any) -> Usage | None: + raw = getattr(response, "usage", None) + if raw is None: + return None + def get(name: str) -> Any: + return raw.get(name) if isinstance(raw, Mapping) else getattr(raw, name, None) + return Usage( + input_tokens=get("prompt_tokens"), + output_tokens=get("completion_tokens"), + total_tokens=get("total_tokens"), + ) + + +class LiteLLMBackend: + """Preserve LiteLLM model and API-base behavior behind the shared contract.""" + + capabilities = BackendCapabilities(usage_reporting=True) + + def __init__( + self, model: str | None = None, *, api_base: str | None = None, + completion: Callable[..., Any] | None = None, + ) -> None: + self.model = model + self.api_base = api_base + self._completion = completion + + def preflight(self) -> BackendStatus: + return BackendStatus(ready=True, backend="api", auth_class="api", + auth_source=self._auth_source(self.model)) + + @staticmethod + def _auth_source(model: str | None) -> AuthSource: + if model and (model.startswith("openai/") or model.startswith("azure/")): + return "openai_api" + if model and model.startswith("anthropic/"): + return "anthropic_api" + return "other_api" + + def complete(self, request: CompletionRequest) -> CompletionResult: + validate_capabilities(request, self.capabilities) + model = request.model or self.model + if not model: + raise ConfigurationError("API model is required") + if request.output_schema is not None: + _check_schema(request.output_schema) + messages = [] + if request.system_prompt is not None: + messages.append({"role": "system", "content": request.system_prompt}) + messages.append({"role": "user", "content": request.prompt}) + kwargs: dict[str, Any] = {"model": model, "messages": messages} + if request.timeout_seconds is not None: + kwargs["timeout"] = request.timeout_seconds + if request.output_schema is not None or request.json_mode: + kwargs["response_format"] = {"type": "json_object"} + if request.sampling is not None: + if request.sampling.temperature is not None: + kwargs["temperature"] = request.sampling.temperature + if request.sampling.top_p is not None: + kwargs["top_p"] = request.sampling.top_p + if request.sampling.max_output_tokens is not None: + kwargs["max_tokens"] = request.sampling.max_output_tokens + if self.api_base: + kwargs["base_url"] = self.api_base + completion = self._completion + if completion is None: + import litellm + completion = litellm.completion + started_at = datetime.now(timezone.utc).isoformat() + start = time.monotonic() + try: + response = completion(**kwargs) + except Exception as exc: + raise _error_for(exc) from None + try: + content = response.choices[0].message.content + except (AttributeError, IndexError, KeyError, TypeError): + raise InvalidResponse("API response did not include completion content") from None + if not isinstance(content, str) or not content: + raise InvalidResponse("API response content was empty or non-text") + structured = None + if request.output_schema is not None: + structured = validate_structured(content, request.output_schema) + return CompletionResult( + text=content, structured=structured, requested_backend="api", actual_backend="api", + requested_model=model, actual_model=getattr(response, "model", None) or model, + auth_class="api", auth_source=self._auth_source(model), usage=_usage(response), + role=request.role, started_at=started_at, duration_seconds=time.monotonic() - start, + prompt_contract_version=request.prompt_contract_version, + schema_contract_version=request.schema_contract_version, + ) diff --git a/src/optimize_anything/llm_backends/provenance.py b/src/optimize_anything/llm_backends/provenance.py new file mode 100644 index 0000000..ef87fde --- /dev/null +++ b/src/optimize_anything/llm_backends/provenance.py @@ -0,0 +1,76 @@ +"""Content-free completion provenance and cache identity primitives.""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter +from typing import Any, Iterable, Mapping + +from .base import CompletionRequest, CompletionResult + + +def completion_event(result: CompletionResult, *, call_id: str | None = None) -> dict[str, Any]: + """Select the safe, stable fields allowed in the run event stream.""" + event: dict[str, Any] = { + "role": result.role, + "requested_backend": result.requested_backend, + "actual_backend": result.actual_backend, + "requested_model": result.requested_model, + "actual_model": result.actual_model, + "auth_class": result.auth_class, + "auth_source": result.auth_source, + "started_at": result.started_at, + "duration_seconds": result.duration_seconds, + "retry_count": result.retry_count, + "prompt_contract_version": result.prompt_contract_version, + "schema_contract_version": result.schema_contract_version, + } + if call_id: + event["call_id"] = call_id + if result.usage: + event.update({ + "input_tokens": result.usage.input_tokens, + "output_tokens": result.usage.output_tokens, + "total_tokens": result.usage.total_tokens, + }) + if result.fallback: + event["fallback_source"] = result.fallback.source_backend + event["fallback_reason"] = result.fallback.reason + return {key: value for key, value in event.items() if value is not None} + + +def aggregate_provenance(events: Iterable[Mapping[str, Any]]) -> dict[str, Any]: + """Aggregate only provider metadata, with no prompts or account identifiers.""" + rows = list(events) + dimensions = ("actual_backend", "actual_model", "auth_class", "role") + counts = {dimension: dict(Counter(str(row.get(dimension, "unknown")) for row in rows)) + for dimension in dimensions} + usage = { + name: sum(value for row in rows if isinstance((value := row.get(name)), int)) + for name in ("input_tokens", "output_tokens", "total_tokens") + } + return { + "call_count": len(rows), + "counts": counts, + "usage": usage, + "fallback_causes": dict(Counter(str(row["fallback_reason"]) for row in rows + if "fallback_reason" in row)), + "mixed_backend": len({row.get("actual_backend") for row in rows}) > 1, + } + + +def cache_fingerprint( + request: CompletionRequest, result: CompletionResult, *, input_identity: str, +) -> str: + """Hash cache identity under the actual billing/backend route.""" + identity = { + "actual_backend": result.actual_backend, + "actual_model": result.actual_model or "provider-default", + "auth_class": result.auth_class, + "role": request.role, + "prompt_contract_version": request.prompt_contract_version, + "schema_contract_version": request.schema_contract_version, + "input_identity": input_identity, + } + return hashlib.sha256(json.dumps(identity, sort_keys=True, separators=(",", ":")).encode()).hexdigest() diff --git a/src/optimize_anything/llm_backends/schema.py b/src/optimize_anything/llm_backends/schema.py new file mode 100644 index 0000000..13ac40d --- /dev/null +++ b/src/optimize_anything/llm_backends/schema.py @@ -0,0 +1,53 @@ +"""Shared JSON Schema trust-boundary checks.""" + +from __future__ import annotations + +from .base import ConfigurationError + + +def score_output_schema( + dimension_names: list[str], *, include_hard_constraints: bool, +) -> dict[str, object]: + """Build the shared strict score response contract.""" + properties: dict[str, object] = { + "score": {"type": "number"}, + "reasoning": {"type": "string"}, + } + required = ["score", "reasoning"] + for name in dimension_names: + properties[name] = {"type": "number"} + if name not in required: + required.append(name) + if include_hard_constraints: + properties["hard_constraints_satisfied"] = {"type": "boolean"} + required.append("hard_constraints_satisfied") + return { + "type": "object", + "properties": properties, + "required": required, + "additionalProperties": False, + } + + +def reject_external_refs(value: object) -> None: + """Allow only document-local references; reject remote and rebasing identifiers.""" + if isinstance(value, dict): + for key, item in value.items(): + if key in ("$ref", "$dynamicRef", "$recursiveRef", "$id") and ( + not isinstance(item, str) or key == "$id" or not item.startswith("#/") + ): + raise ConfigurationError("output schema may only use local references") + reject_external_refs(item) + elif isinstance(value, list): + for item in value: + reject_external_refs(item) + + +def strip_code_fences(text: str) -> str: + """Remove one optional Markdown fence around provider JSON.""" + cleaned = text.strip() + if cleaned.startswith("```"): + cleaned = cleaned.partition("\n")[2] + if cleaned.rstrip().endswith("```"): + cleaned = cleaned.rstrip()[:-3].rstrip() + return cleaned diff --git a/src/optimize_anything/llm_judge.py b/src/optimize_anything/llm_judge.py index 23b4168..60c62c1 100644 --- a/src/optimize_anything/llm_judge.py +++ b/src/optimize_anything/llm_judge.py @@ -1,6 +1,6 @@ """LLM-as-Judge evaluator factory and analysis tools. -Uses litellm to call a language model as an evaluator. The model receives a +Uses a provider-neutral completion backend to call a language model as an evaluator. The model receives a structured prompt describing the objective, quality dimensions, and hard constraints, then returns a JSON score object. @@ -14,6 +14,19 @@ import math from typing import Any, Callable +from optimize_anything.llm_backends.base import ( + CompletionBackend, + CompletionRequest, + InvalidResponse, + Role, + SamplingOptions, +) +from optimize_anything.llm_backends.litellm_backend import LiteLLMBackend +from optimize_anything.llm_backends.provenance import completion_event +from optimize_anything.llm_backends.schema import score_output_schema, strip_code_fences + +_strip_code_fences = strip_code_fences + JUDGE_SYSTEM_PROMPT = """\ You are a careful, objective evaluator. You will be given a text artifact and asked to score it. You must return ONLY a JSON object — no markdown, no @@ -81,17 +94,22 @@ def llm_judge_evaluator( objective: str, *, - model: str, + model: str | None = None, quality_dimensions: list[dict[str, Any]] | None = None, hard_constraints: list[str] | None = None, timeout: float = 60.0, temperature: float | None = None, api_base: str | None = None, task_model: str | None = None, + backend: CompletionBackend | None = None, + role: Role = "judge", ) -> Callable[[str, Any | None], tuple[float, dict[str, Any]]]: """Create an LLM-as-judge evaluator compatible with gepa's evaluator contract.""" _validate_objective(objective) - _validate_model_string(model) + use_backend_schema = backend is not None + if backend is None: + _validate_model_string(model) + backend = LiteLLMBackend(model=model, api_base=api_base) dims = quality_dimensions or [] constraints = hard_constraints or [] @@ -105,36 +123,36 @@ def evaluate(candidate: str, example: Any | None = None) -> tuple[float, dict[st example=example, ) try: - import litellm - - completion_kwargs: dict[str, Any] = { - "model": model, - "messages": [ - {"role": "system", "content": JUDGE_SYSTEM_PROMPT}, - {"role": "user", "content": prompt}, - ], - "timeout": timeout, - "response_format": {"type": "json_object"}, - } - if temperature is not None: - completion_kwargs["temperature"] = temperature - if api_base: - completion_kwargs["base_url"] = api_base - - response = litellm.completion(**completion_kwargs) - raw_content = response.choices[0].message.content + result = backend.complete(CompletionRequest( + prompt=prompt, + role=role, + model=model, + output_schema=(score_output_schema( + [dim["name"] for dim in dims if isinstance(dim.get("name"), str) and dim["name"]], + include_hard_constraints=bool(dims or constraints), + ) if use_backend_schema else None), + json_mode=not use_backend_schema, + timeout_seconds=timeout, + sampling=(SamplingOptions(temperature=temperature) if temperature is not None else None), + system_prompt=JUDGE_SYSTEM_PROMPT, + )) + raw_content = result.text except Exception as exc: error_side_info: dict[str, Any] = { "error": f"LLM call failed: {type(exc).__name__}: {exc}", "reasoning": "LLM judge call failed; returned fallback score 0.0.", } + if isinstance(exc, InvalidResponse): + error_side_info["raw_response"] = "" for dim in dims: name = dim.get("name") if isinstance(name, str) and name: error_side_info.setdefault(name, 0.0) return 0.0, error_side_info - return _parse_judge_response(raw_content, dims, constraints) + score, side_info = _parse_judge_response(raw_content, dims, constraints) + side_info["llm_provenance"] = completion_event(result) + return score, side_info return evaluate @@ -144,7 +162,7 @@ def _validate_objective(objective: str) -> None: raise ValueError("objective must be a non-empty string") -def _validate_model_string(model: str) -> None: +def _validate_model_string(model: str | None) -> None: if not isinstance(model, str) or not model.strip(): raise ValueError("model must be a non-empty string") @@ -194,7 +212,7 @@ def _parse_judge_response( return 0.0, {"error": "LLM returned empty response"} # Strip markdown code fences (e.g. ```json ... ```) that some providers add - cleaned = _strip_code_fences(raw_content) + cleaned = strip_code_fences(raw_content) try: parsed = json.loads(cleaned) @@ -304,25 +322,15 @@ def _coerce_float(value: Any) -> float | None: """ -def _strip_code_fences(text: str) -> str: - """Strip markdown code fences from LLM response text.""" - cleaned = text.strip() - if cleaned.startswith("```"): - first_newline = cleaned.index("\n") if "\n" in cleaned else len(cleaned) - cleaned = cleaned[first_newline + 1:] - if cleaned.rstrip().endswith("```"): - cleaned = cleaned.rstrip()[:-len("```")].rstrip() - return cleaned - - def analyze_for_dimensions( artifact: str, objective: str, - model: str, + model: str | None = None, *, api_base: str | None = None, timeout: float = 60.0, temperature: float | None = None, + backend: CompletionBackend | None = None, ) -> dict[str, Any]: """Score an artifact then discover quality dimensions for refinement. @@ -330,9 +338,10 @@ def analyze_for_dimensions( specific quality dimensions where improvement is possible. """ _validate_objective(objective) - _validate_model_string(model) - - import litellm + use_backend_schema = backend is not None + if backend is None: + _validate_model_string(model) + backend = LiteLLMBackend(model=model, api_base=api_base) # --- Call 1: Score the artifact with vague objective --- score_prompt = _build_prompt( @@ -341,23 +350,20 @@ def analyze_for_dimensions( quality_dimensions=[], hard_constraints=[], ) - completion_kwargs: dict[str, Any] = { - "model": model, - "messages": [ - {"role": "system", "content": JUDGE_SYSTEM_PROMPT}, - {"role": "user", "content": score_prompt}, - ], - "timeout": timeout, - "response_format": {"type": "json_object"}, - } - if temperature is not None: - completion_kwargs["temperature"] = temperature - if api_base: - completion_kwargs["base_url"] = api_base - try: - response = litellm.completion(**completion_kwargs) - raw_score_content = response.choices[0].message.content + response = backend.complete(CompletionRequest( + prompt=score_prompt, + role="analysis", + model=model, + output_schema=(score_output_schema([], include_hard_constraints=False) + if use_backend_schema else None), + json_mode=not use_backend_schema, + timeout_seconds=timeout, + sampling=(SamplingOptions(temperature=temperature) if temperature is not None else None), + system_prompt=JUDGE_SYSTEM_PROMPT, + )) + score_result = response + raw_score_content = response.text except Exception as exc: raise RuntimeError(f"Scoring LLM call failed: {type(exc).__name__}: {exc}") from exc @@ -371,23 +377,18 @@ def analyze_for_dimensions( objective=objective, artifact=artifact, ) - analyze_kwargs: dict[str, Any] = { - "model": model, - "messages": [ - {"role": "system", "content": ANALYZE_SYSTEM_PROMPT}, - {"role": "user", "content": analyze_prompt}, - ], - "timeout": timeout, - "response_format": {"type": "json_object"}, - } - if temperature is not None: - analyze_kwargs["temperature"] = temperature - if api_base: - analyze_kwargs["base_url"] = api_base - try: - response = litellm.completion(**analyze_kwargs) - raw_dims_content = response.choices[0].message.content + response = backend.complete(CompletionRequest( + prompt=analyze_prompt, + role="analysis", + model=model, + output_schema=(_dimensions_output_schema() if use_backend_schema else None), + json_mode=not use_backend_schema, + timeout_seconds=timeout, + sampling=(SamplingOptions(temperature=temperature) if temperature is not None else None), + system_prompt=ANALYZE_SYSTEM_PROMPT, + )) + raw_dims_content = response.text except Exception as exc: raise RuntimeError( f"Dimension discovery LLM call failed: {type(exc).__name__}: {exc}" @@ -409,22 +410,48 @@ def analyze_for_dimensions( "reasoning": score_info.get("reasoning", ""), "suggested_dimensions": dimensions, "intake_json": intake_json_str, + "llm_provenance": [completion_event(score_result), completion_event(response)], "recommendation": ( f"Use the suggested dimensions with:\n" f" optimize-anything optimize " - f"--judge-model {model} " + f"--judge-model {model or ''} " f"--objective \"{objective}\" " f"--intake-json '{intake_json_str}'" ), } +def _dimensions_output_schema() -> dict[str, Any]: + return { + "type": "object", + "properties": { + "dimensions": { + "type": "array", + "minItems": 1, + "items": { + "type": "object", + "properties": { + "name": {"type": "string", "minLength": 1}, + "weight": {"type": "number"}, + "score": {"type": "number"}, + "description": {"type": "string"}, + }, + "required": ["name", "weight", "score", "description"], + "additionalProperties": False, + }, + }, + }, + "required": ["dimensions"], + "additionalProperties": False, + } + + def _parse_dimensions_response(raw_content: str | None) -> list[dict[str, Any]]: """Parse the dimension discovery LLM response into validated dimension dicts.""" if not raw_content: raise RuntimeError("Dimension discovery returned empty response") - cleaned = _strip_code_fences(raw_content) + cleaned = strip_code_fences(raw_content) try: parsed = json.loads(cleaned) diff --git a/src/optimize_anything/spec_loader.py b/src/optimize_anything/spec_loader.py index a8c8652..ffa812e 100644 --- a/src/optimize_anything/spec_loader.py +++ b/src/optimize_anything/spec_loader.py @@ -56,6 +56,12 @@ def _normalize_spec(raw: dict[str, Any], *, spec_dir: Path) -> dict[str, Any]: "evaluator_cwd": None, "judge_model": None, "proposer_model": None, + "judge_backend": None, + "proposer_backend": None, + "judge_api_fallback": None, + "proposer_api_fallback": None, + "judge_api_fallback_model": None, + "proposer_api_fallback_model": None, "task_model": None, "intake": None, } @@ -141,13 +147,40 @@ def _normalize_model_section(model: dict[str, Any]) -> dict[str, Any]: normalized: dict[str, Any] = {} if "judge" in model: - normalized["judge_model"] = _require_string(model, "judge", "model") + normalized.update(_normalize_model_role(model["judge"], role="judge")) if "proposer" in model: - normalized["proposer_model"] = _require_string(model, "proposer", "model") + normalized.update(_normalize_model_role(model["proposer"], role="proposer")) return normalized +def _normalize_model_role(value: Any, *, role: str) -> dict[str, Any]: + if isinstance(value, str): + return {f"{role}_model": value, f"{role}_backend": "api"} + if not isinstance(value, dict): + raise SpecLoadError(f"model.{role} must be a string or table") + unknown = set(value) - {"backend", "model", "api_fallback", "api_fallback_model"} + if unknown: + raise SpecLoadError(f"model.{role} has unknown keys: {', '.join(sorted(unknown))}") + backend = value.get("backend", "api") + if backend not in ("api", "codex", "claude"): + raise SpecLoadError(f"model.{role}.backend must be api, codex, or claude") + normalized: dict[str, Any] = {f"{role}_backend": backend} + if "model" in value: + normalized[f"{role}_model"] = _require_string(value, "model", f"model.{role}") + if "api_fallback" in value: + normalized[f"{role}_api_fallback"] = _require_bool( + value, "api_fallback", f"model.{role}" + ) + elif backend != "api": + normalized[f"{role}_api_fallback"] = True + if "api_fallback_model" in value: + normalized[f"{role}_api_fallback_model"] = _require_string( + value, "api_fallback_model", f"model.{role}" + ) + return normalized + + def _require_string(section: dict[str, Any], key: str, section_name: str) -> str: value = section[key] if not isinstance(value, str): diff --git a/tests/test_claude_backend.py b/tests/test_claude_backend.py new file mode 100644 index 0000000..790866d --- /dev/null +++ b/tests/test_claude_backend.py @@ -0,0 +1,151 @@ +"""Offline security and contract tests for Claude CLI completion.""" + +import json +import os +import sys +from pathlib import Path + +import pytest + +from optimize_anything.llm_backends.base import ( + AuthenticationError, CompletionRequest, ConfigurationError, InvalidResponse, Timeout, +) +from optimize_anything.llm_backends.claude_backend import ( + ClaudeCliBackend, _ProcessOutput, _run_bounded, _subscription_env, +) + + +class FakeRunner: + def __init__(self, response=None): + self.calls = [] + self.response = response or {"result": "done", "model": "sonnet"} + self.auth = {"loggedIn": True, "authMethod": "claude.ai", "apiProvider": "firstParty"} + + def __call__(self, argv, *, stdin, env, cwd, timeout, max_output): + self.calls.append((list(argv), stdin, env, cwd, timeout, max_output)) + if "--version" in argv: + return _ProcessOutput(0, b"2.1.278 (Claude Code)", b"") + if "--help" in argv: + from optimize_anything.llm_backends.claude_backend import _REQUIRED_FLAGS + return _ProcessOutput(0, " ".join(_REQUIRED_FLAGS).encode(), b"") + if "auth" in argv: + return _ProcessOutput(0, json.dumps(self.auth).encode(), b"") + assert Path(cwd).is_dir() + assert Path(argv[argv.index("--mcp-config") + 1]).read_text() == '{"mcpServers":{}}' + return _ProcessOutput(0, json.dumps(self.response).encode(), b"") + + +def backend(tmp_path, runner): + executable = tmp_path / "claude" + executable.write_text("") + return ClaudeCliBackend( + executable=str(executable), runner=runner, + environ={ + "HOME": "/saved-login", "CLAUDE_CONFIG_DIR": "/saved-config", + "ANTHROPIC_API_KEY": "sentinel-api-key", "CLAUDE_CODE_OAUTH_TOKEN": "sentinel-token", + "CLAUDECODE": "parent-agent", "AWS_ACCESS_KEY_ID": "sentinel-aws", + "CLAUDE_CODE_USE_BEDROCK": "1", "GOOGLE_APPLICATION_CREDENTIALS": "sentinel-google", + }, + ) + + +def test_preflight_requires_claude_subscription_under_scrubbed_env(tmp_path): + runner = FakeRunner() + adapter = backend(tmp_path, runner) + assert adapter.preflight().auth_source == "claude_subscription" + for _, _, env, _, _, _ in runner.calls: + assert env["HOME"] == "/saved-login" + assert env["CLAUDE_CONFIG_DIR"] == "/saved-config" + assert all("sentinel" not in value for value in env.values()) + assert "CLAUDECODE" not in env + runner.auth["authMethod"] = "apiKey" + with pytest.raises(AuthenticationError): + adapter.preflight() + + +def test_schema_and_user_content_stay_off_argv_and_are_locally_validated(tmp_path): + runner = FakeRunner({"structured_output": {"payload": '{"secret-property":"secret-enum"}'}}) + adapter = backend(tmp_path, runner) + schema = { + "type": "object", "properties": { + "secret-property": {"type": "string", "enum": ["secret-enum"]} + }, "required": ["secret-property"], + } + result = adapter.complete(CompletionRequest( + prompt="private-candidate", role="judge", output_schema=schema, + )) + assert result.structured["secret-property"] == "secret-enum" + argv, stdin, env, cwd, _, _ = runner.calls[-1] + joined = " ".join(argv) + assert all(value not in joined for value in ("private-candidate", "secret-property", "secret-enum")) + assert all(value in stdin.decode() for value in ("private-candidate", "secret-property", "secret-enum")) + assert "--safe-mode" in argv and "--bare" not in argv + assert argv[argv.index("--tools") + 1] == "" + assert "--no-session-persistence" in argv + assert not Path(cwd).exists() + + +def test_original_schema_rejects_transport_valid_value(tmp_path): + runner = FakeRunner({"structured_output": {"payload": '{"score":2}'}}) + adapter = backend(tmp_path, runner) + with pytest.raises(InvalidResponse): + adapter.complete(CompletionRequest( + prompt="judge", role="judge", output_schema={ + "type": "object", "properties": {"score": {"type": "integer", "maximum": 1}}, + }, + )) + + +def test_timeout_is_typed_and_private_workspace_is_removed(tmp_path): + class TimedOut(FakeRunner): + def __call__(self, argv, **kwargs): + if "-p" in argv: + self.calls.append((argv, kwargs["stdin"], kwargs["env"], kwargs["cwd"], kwargs["timeout"], kwargs["max_output"])) + raise Timeout("timed out") + return super().__call__(argv, **kwargs) + + runner = TimedOut() + with pytest.raises(Timeout): + backend(tmp_path, runner).complete(CompletionRequest(prompt="secret", role="proposer")) + assert not Path(runner.calls[-1][3]).exists() + + +def test_scrubber_covers_paid_auth_and_keeps_saved_login_location(): + env = _subscription_env({ + "HOME": "/home/user", "CLAUDE_CONFIG_DIR": "/config", + "ANTHROPIC_BASE_URL": "bad", "AZURE_CLIENT_SECRET": "bad", + "CLOUDSDK_AUTH_CREDENTIAL_FILE_OVERRIDE": "bad", "CLAUDE_CODE_API_KEY_HELPER": "bad", + }) + assert env == {"HOME": "/home/user", "CLAUDE_CONFIG_DIR": "/config"} + + +def test_external_schema_ref_is_rejected_before_completion(tmp_path): + runner = FakeRunner() + with pytest.raises(ConfigurationError): + backend(tmp_path, runner).complete(CompletionRequest( + prompt="private", role="judge", output_schema={"$ref": "https://example.com/schema"}, + )) + assert not any("-p" in call[0] for call in runner.calls) + + +def test_external_dynamic_schema_ref_is_rejected_before_completion(tmp_path): + runner = FakeRunner() + with pytest.raises(ConfigurationError): + backend(tmp_path, runner).complete(CompletionRequest( + prompt="private", role="judge", + output_schema={"$dynamicRef": "http://127.0.0.1/internal"}, + )) + assert not any("-p" in call[0] for call in runner.calls) + + +def test_real_process_runner_bounds_output_and_terminates_on_timeout(tmp_path): + with pytest.raises(InvalidResponse): + _run_bounded( + [sys.executable, "-c", "print('x' * 10000)"], stdin=b"", + env=os.environ, cwd=str(tmp_path), timeout=2, max_output=100, + ) + with pytest.raises(Timeout): + _run_bounded( + [sys.executable, "-c", "import time; time.sleep(2)"], stdin=b"", + env=os.environ, cwd=str(tmp_path), timeout=0.02, max_output=100, + ) diff --git a/tests/test_codex_backend.py b/tests/test_codex_backend.py new file mode 100644 index 0000000..a98206a --- /dev/null +++ b/tests/test_codex_backend.py @@ -0,0 +1,151 @@ +"""Offline account, isolation, schema, and cleanup tests for Codex SDK.""" + +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from optimize_anything.llm_backends.base import ( + AuthenticationError, BackendUnavailable, CompletionRequest, InvalidResponse, Timeout, +) +from optimize_anything.llm_backends.codex_backend import CodexSdkBackend + + +class FakeHandle: + def __init__(self, sdk): + self.sdk = sdk + self.interrupted = False + + def run(self): + if self.sdk.timeout: + import time + time.sleep(0.1) + return SimpleNamespace( + status="completed", error=None, final_response=self.sdk.response, + usage=SimpleNamespace(total=SimpleNamespace( + input_tokens=2, output_tokens=3, total_tokens=5, + )), + ) + + def interrupt(self): + self.interrupted = True + + +class FakeThread: + def __init__(self, sdk): + self.sdk = sdk + + def turn(self, prompt, **kwargs): + self.sdk.turn_args.append((prompt, kwargs)) + self.sdk.handle = FakeHandle(self.sdk) + return self.sdk.handle + + +class FakeClient: + def __init__(self, sdk, config): + self.sdk = sdk + self.config = config + + def __enter__(self): + return self + + def __exit__(self, *_): + self.sdk.closed += 1 + + def account(self): + return SimpleNamespace(account=SimpleNamespace(root=SimpleNamespace(type=self.sdk.auth_type))) + + def thread_start(self, **kwargs): + self.sdk.thread_args.append(kwargs) + self.sdk.workspaces.append(kwargs["cwd"]) + assert list(Path(kwargs["cwd"]).iterdir()) == [] + return FakeThread(self.sdk) + + +class FakeSdk: + __version__ = "0.156.0" + Sandbox = SimpleNamespace(read_only="read-only") + ApprovalMode = SimpleNamespace(deny_all="deny-all") + + def __init__(self): + self.auth_type = "chatgpt" + self.response = "done" + self.timeout = False + self.thread_args = [] + self.turn_args = [] + self.workspaces = [] + self.client_configs = [] + self.closed = 0 + self.handle = None + + def CodexConfig(self, **kwargs): + self.client_configs.append(kwargs) + return kwargs + + def Codex(self, *, config): + return FakeClient(self, config) + + +def adapter(sdk, tmp_path): + auth = tmp_path / "auth.json" + auth.touch() + return CodexSdkBackend(sdk_module=sdk, auth_path=auth) + + +def test_chatgpt_account_and_ephemeral_private_turn(tmp_path): + sdk = FakeSdk() + result = adapter(sdk, tmp_path).complete(CompletionRequest(prompt="private-prompt", role="proposer")) + assert result.text == "done" and result.auth_source == "chatgpt" + thread = sdk.thread_args[0] + assert thread["ephemeral"] is True + assert thread["sandbox"] == "read-only" + assert thread["approval_mode"] == "deny-all" + assert sdk.turn_args[0][0] == "private-prompt" + assert "private-prompt" not in repr(sdk.client_configs) + assert "features.shell_tool=false" in sdk.client_configs[0]["config_overrides"] + private_home = sdk.client_configs[0]["env"]["CODEX_HOME"] + assert not Path(private_home).exists() + assert all(not Path(workspace).exists() for workspace in sdk.workspaces) + assert sdk.closed == 1 + + +def test_api_key_auth_is_rejected_before_thread_dispatch(tmp_path): + sdk = FakeSdk() + sdk.auth_type = "apiKey" + with pytest.raises(AuthenticationError): + adapter(sdk, tmp_path).complete(CompletionRequest(prompt="private", role="judge")) + assert sdk.thread_args == [] + + +def test_version_mismatch_fails_closed(tmp_path): + sdk = FakeSdk() + sdk.__version__ = "0.157.0" + with pytest.raises(BackendUnavailable): + adapter(sdk, tmp_path).preflight() + assert sdk.client_configs == [] + + +def test_structured_output_is_validated_locally(tmp_path): + sdk = FakeSdk() + sdk.response = json.dumps({"score": 2}) + request = CompletionRequest(prompt="judge", role="judge", output_schema={ + "type": "object", "properties": {"score": {"type": "integer", "maximum": 1}}, + }) + with pytest.raises(InvalidResponse): + adapter(sdk, tmp_path).complete(request) + assert sdk.turn_args[0][1]["output_schema"] == { + "type": "object", "properties": {"score": {"type": "integer", "maximum": 1}}, + } + + +def test_timeout_interrupts_and_cleans_up(tmp_path): + sdk = FakeSdk() + sdk.timeout = True + with pytest.raises(Timeout): + adapter(sdk, tmp_path).complete(CompletionRequest( + prompt="private", role="proposer", timeout_seconds=0.01, + )) + assert sdk.handle.interrupted is True + assert sdk.closed == 1 + assert all(not Path(workspace).exists() for workspace in sdk.workspaces) diff --git a/tests/test_evaluator_generator.py b/tests/test_evaluator_generator.py index 5ca8c31..1073205 100644 --- a/tests/test_evaluator_generator.py +++ b/tests/test_evaluator_generator.py @@ -1,13 +1,17 @@ """Tests for evaluator generator.""" +import builtins +import io import json import subprocess import sys +from types import SimpleNamespace from pathlib import Path from typing import Any import pytest from optimize_anything.evaluator_generator import generate_evaluator_script +from optimize_anything import evaluator_runtime @pytest.mark.parametrize("evaluator_type", ["judge", "composite"]) @@ -91,6 +95,60 @@ def test_generated_judge_uses_provider_sampling_defaults() -> None: assert "temperature=" not in script +def test_generated_wrapper_runs_json_lines_through_installed_runtime(monkeypatch): + script = generate_evaluator_script( + seed="hello", objective="score quality", evaluator_type="judge", + dataset=True, backend="codex", + intake={"hard_constraints": ["No invented facts"]}, + ) + prompts = [] + + class Backend: + def complete(self, request): + prompts.append(request.prompt) + return SimpleNamespace(structured={"score": 0.7, "reasoning": "Useful"}) + + monkeypatch.setattr(evaluator_runtime, "_resolve_backend", lambda config, *, role: Backend()) + output = io.StringIO() + monkeypatch.setattr(sys, "stdin", io.StringIO( + json.dumps({"candidate": "Answer", "example": {"expected": "Fact"}}) + "\n" + )) + monkeypatch.setattr(sys, "stdout", output) + namespace = {"__name__": "generated_evaluator"} + exec(compile(script, "", "exec"), namespace) + + assert namespace["EVALUATOR_METADATA"]["min_runtime_contract_version"] == 1 + assert namespace["main"]() == 0 + assert json.loads(output.getvalue())["score"] == 0.7 + assert "No invented facts" in prompts[0] + assert '"expected": "Fact"' in prompts[0] + + +def test_generated_wrapper_reports_missing_installed_runtime(monkeypatch): + script = generate_evaluator_script( + seed="hello", objective="score quality", backend="codex", + ) + namespace = {"__name__": "generated_evaluator"} + exec(compile(script, "", "exec"), namespace) + original_import = builtins.__import__ + + def missing_runtime(name, *args, **kwargs): + if name == "optimize_anything.evaluator_runtime": + raise ImportError("runtime unavailable") + return original_import(name, *args, **kwargs) + + monkeypatch.setattr(builtins, "__import__", missing_runtime) + monkeypatch.setattr(sys, "stdin", io.StringIO('{"candidate": "one"}\n')) + output = io.StringIO() + monkeypatch.setattr(sys, "stdout", output) + + assert namespace["main"]() == 0 + result = json.loads(output.getvalue()) + assert result["score"] == 0.0 + assert result["error"] == "runtime_unavailable" + assert "install or upgrade" in result["reasoning"].lower() + + class TestGenerateEvaluatorScript: def test_command_evaluator_is_bash(self): script = generate_evaluator_script(seed="hello", objective="improve clarity", evaluator_type="command") @@ -116,12 +174,15 @@ def test_http_evaluator_has_server(self): def test_default_is_judge(self): script = generate_evaluator_script(seed="x", objective="y") assert script.startswith("#!/usr/bin/env python3") - assert "litellm" in script + assert "from litellm import completion" in script - def test_judge_evaluator_contains_litellm_and_objective(self): + def test_judge_evaluator_contains_runtime_config_and_objective(self): objective = "assess clarity and usefulness" - script = generate_evaluator_script(seed="hello", objective=objective, evaluator_type="judge") - assert "from litellm import completion" in script + script = generate_evaluator_script( + seed="hello", objective=objective, evaluator_type="judge", backend="codex", + ) + assert "litellm" not in script + assert "runtime_contract_version" in script assert objective in script def test_default_judge_uses_current_evaluator_model(self) -> None: @@ -140,10 +201,14 @@ def test_default_composite_uses_current_evaluator_model(self) -> None: ) assert "MODEL = 'openai/gpt-5.6-luna'" in script - def test_judge_evaluator_handles_missing_api_key_gracefully(self): - script = generate_evaluator_script(seed="hello", objective="test", evaluator_type="judge") - assert "Missing API key" in script - assert "missing_api_key" in script + def test_judge_evaluator_embeds_backend_and_fallback_configuration(self): + script = generate_evaluator_script( + seed="hello", objective="test", evaluator_type="judge", + backend="codex", api_fallback=False, max_concurrency=2, + ) + assert "'backend': 'codex'" in script + assert "'api_fallback': False" in script + assert "'max_concurrency': 2" in script def test_objective_with_quotes_is_safe_in_command_script(self, tmp_path: Path): objective = 'Improve "install docs"\nfor O\'Reilly users' @@ -248,12 +313,13 @@ def test_intake_quality_dimensions_are_embedded(self): ) assert "QUALITY_DIMENSIONS = [('accuracy', 0.7), ('clarity', 0.3)]" in script - def test_composite_evaluator_has_constraints_and_judge(self): - script = generate_evaluator_script(seed="hello", objective="test", evaluator_type="composite") - assert "_constraint_non_empty" in script - assert "hard_constraint_failures" in script - assert "_run_judge" in script - assert "litellm" in script + def test_subscription_composite_evaluator_has_constraints_and_judge(self): + script = generate_evaluator_script( + seed="hello", objective="test", evaluator_type="composite", backend="codex", + ) + assert "'evaluator_type': 'composite'" in script + assert "run_generated_evaluator" in script + assert "litellm" not in script def test_dataset_flag_adds_example_extraction_for_all_types(self): for ev_type in ["judge", "command", "http", "composite"]: @@ -262,6 +328,10 @@ def test_dataset_flag_adds_example_extraction_for_all_types(self): objective="test", evaluator_type=ev_type, dataset=True, + backend="codex" if ev_type in {"judge", "composite"} else "api", ) - assert "example" in script - assert "data.get(\"example\")" in script + if ev_type in {"judge", "composite"}: + assert "'dataset': True" in script + else: + assert "example" in script + assert "data.get(\"example\")" in script diff --git a/tests/test_evaluator_runtime.py b/tests/test_evaluator_runtime.py new file mode 100644 index 0000000..4867750 --- /dev/null +++ b/tests/test_evaluator_runtime.py @@ -0,0 +1,138 @@ +"""Contract tests for the installed generated-evaluator runtime.""" + +import io +import json +from types import SimpleNamespace + +from optimize_anything import evaluator_runtime + + +def _config(**overrides): + config = { + "min_runtime_contract_version": 1, + "evaluator_type": "judge", + "objective": "Assess accuracy", + "template_family": "general_text", + "rubric_summary": "Accurate answers", + "quality_dimensions": [("accuracy", 0.7), ("clarity", 0.3)], + "hard_constraints": ["Do not invent citations"], + "dataset": True, + "backend": "codex", + "model": "gpt-5.6-luna", + "api_base": None, + "api_fallback": False, + "api_fallback_model": None, + "max_concurrency": 1, + } + config.update(overrides) + return config + + +def test_runtime_scores_json_lines_and_forwards_role_config_and_examples(): + requests = [] + resolved = [] + + class Backend: + def complete(self, request): + requests.append(request) + return SimpleNamespace(structured={"score": 0.8, "reasoning": "Good", "accuracy": 0.9, "clarity": 0.6}) + + def resolver(config, *, role): + resolved.append((config, role)) + return Backend() + + output = io.StringIO() + source = io.StringIO('\n'.join([ + json.dumps({"candidate": "First", "example": {"expected": "A"}}), + json.dumps({"candidate": "Second", "example": {"expected": "B"}}), + ]) + '\n') + evaluator_runtime.run_generated_evaluator( + _config(), backend_resolver=resolver, input_stream=source, output_stream=output + ) + + results = [json.loads(line) for line in output.getvalue().splitlines()] + assert len(results) == 2 + assert all(result["score"] == 0.8 for result in results) + assert results[0]["accuracy"] == 0.9 + assert resolved[0][1] == "judge" + assert resolved[0][0]["backend"] == "codex" + assert '"expected": "A"' in requests[0].prompt + assert "Do not invent citations" in requests[0].prompt + assert "First" in requests[0].prompt + assert "Second" in requests[1].prompt + + +def test_composite_constraints_short_circuit_without_backend_call(): + output = io.StringIO() + + def resolver(config, *, role): + raise AssertionError("constraint failure must not dispatch") + + evaluator_runtime.run_generated_evaluator( + _config(evaluator_type="composite"), + backend_resolver=resolver, + input_stream=io.StringIO(json.dumps({"candidate": "TODO"}) + "\n"), + output_stream=output, + ) + result = json.loads(output.getvalue()) + assert result["score"] == 0.0 + assert result["hard_constraints_satisfied"] is False + assert result["hard_constraint_failures"] + + +def test_preflight_probe_does_not_call_backend(): + output = io.StringIO() + + def resolver(config, *, role): + raise AssertionError("preflight must remain local") + + evaluator_runtime.run_generated_evaluator( + _config(), + backend_resolver=resolver, + input_stream=io.StringIO(json.dumps({"candidate": "__optimize_anything_preflight__"}) + "\n"), + output_stream=output, + ) + + assert json.loads(output.getvalue()) == { + "score": 1.0, + "reasoning": "Evaluator runtime is ready.", + } + + +def test_runtime_reports_incompatible_contract_and_invalid_json_as_score_lines(): + output = io.StringIO() + evaluator_runtime.run_generated_evaluator( + _config(min_runtime_contract_version=999), + input_stream=io.StringIO('{}\n'), + output_stream=output, + ) + incompatible = json.loads(output.getvalue()) + assert incompatible["score"] == 0.0 + assert incompatible["error"] == "incompatible_runtime" + assert "upgrade" in incompatible["reasoning"].lower() + + output = io.StringIO() + evaluator_runtime.run_generated_evaluator( + _config(), input_stream=io.StringIO('not json\n'), output_stream=output, + backend_resolver=lambda config, *, role: None, + ) + invalid = json.loads(output.getvalue()) + assert invalid["score"] == 0.0 + assert invalid["error"] == "invalid_input" + + +def test_dimension_name_cannot_replace_canonical_score(): + class Backend: + def complete(self, request): + return SimpleNamespace(structured={"score": 0.8, "reasoning": "Good"}) + + output = io.StringIO() + evaluator_runtime.run_generated_evaluator( + _config(quality_dimensions=[("score", 1.0)], backend="codex"), + backend_resolver=lambda config, *, role: Backend(), + input_stream=io.StringIO('{"candidate": "answer"}\n'), + output_stream=output, + ) + result = json.loads(output.getvalue()) + assert result["score"] == 0.8 + assert result["dimension_scores"]["score"] == 0.8 diff --git a/tests/test_llm_backend_contract.py b/tests/test_llm_backend_contract.py new file mode 100644 index 0000000..43d6fcd --- /dev/null +++ b/tests/test_llm_backend_contract.py @@ -0,0 +1,107 @@ +"""Contract tests for provider-neutral completions.""" + +from dataclasses import FrozenInstanceError +from types import SimpleNamespace + +import pytest + +from optimize_anything.llm_backends import ( + BackendCapabilities, + BackendSpec, + CompletionRequest, + ConfigurationError, + InvalidResponse, + LiteLLMBackend, + SamplingOptions, + cache_fingerprint, +) + + +def test_contract_values_are_immutable() -> None: + request = CompletionRequest( + prompt="private prompt", role="judge", output_schema={"type": "object", "required": ["score"]} + ) + spec = BackendSpec(backend="api", model="openai/test") + capabilities = BackendCapabilities() + with pytest.raises(FrozenInstanceError): + request.prompt = "changed" # type: ignore[misc] + with pytest.raises(TypeError): + request.output_schema["type"] = "string" # type: ignore[index] + with pytest.raises(FrozenInstanceError): + spec.model = "changed" # type: ignore[misc] + with pytest.raises(FrozenInstanceError): + capabilities.structured_output = False # type: ignore[misc] + + +def test_litellm_text_preserves_existing_kwargs_and_provenance() -> None: + calls = [] + + def completion(**kwargs): + calls.append(kwargs) + return SimpleNamespace( + choices=[SimpleNamespace(message=SimpleNamespace(content="hello"))], + model="openai/actual", + usage=SimpleNamespace(prompt_tokens=2, completion_tokens=3, total_tokens=5), + ) + + backend = LiteLLMBackend(model="openai/test", api_base="https://api.example", completion=completion) + result = backend.complete( + CompletionRequest(prompt="hi", system_prompt="system", role="analysis", timeout_seconds=8, + sampling=SamplingOptions(temperature=0.3)) + ) + assert calls == [{"model": "openai/test", "messages": [ + {"role": "system", "content": "system"}, {"role": "user", "content": "hi"}], + "timeout": 8, "temperature": 0.3, "base_url": "https://api.example"}] + assert (result.text, result.auth_class, result.auth_source) == ("hello", "api", "openai_api") + assert (result.requested_model, result.actual_model) == ("openai/test", "openai/actual") + assert result.usage.total_tokens == 5 + with pytest.raises(FrozenInstanceError): + result.text = "changed" # type: ignore[misc] + + +def test_litellm_schema_is_validated_locally() -> None: + def completion(**_kwargs): + return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content='{"score": "bad"}'))]) + + backend = LiteLLMBackend(model="openai/test", completion=completion) + request = CompletionRequest(prompt="score", role="judge", output_schema={ + "type": "object", "required": ["score"], + "properties": {"score": {"type": "number"}}, + }) + with pytest.raises(InvalidResponse): + backend.complete(request) + + +def test_litellm_rejects_unsupported_schema_before_dispatch() -> None: + calls = [] + backend = LiteLLMBackend(model="openai/test", completion=lambda **kw: calls.append(kw)) + with pytest.raises(ConfigurationError): + backend.complete(CompletionRequest(prompt="x", role="judge", output_schema={"$ref": "remote"})) + assert calls == [] + + +def test_provider_error_does_not_expose_provider_message() -> None: + class AuthenticationException(Exception): + pass + + def completion(**_kwargs): + raise AuthenticationException("sk-secret private prompt") + + backend = LiteLLMBackend(model="openai/test", completion=completion) + from optimize_anything.llm_backends import AuthenticationError + with pytest.raises(AuthenticationError) as caught: + backend.complete(CompletionRequest(prompt="private prompt", role="judge")) + assert "sk-secret" not in str(caught.value) + assert caught.value.__cause__ is None + + +def test_cache_fingerprint_uses_actual_route_without_exposing_input() -> None: + from optimize_anything.llm_backends import CompletionResult + + request = CompletionRequest(prompt="private prompt", role="judge") + subscription = CompletionResult("ok", None, "codex", "codex", None, "gpt", "subscription", "chatgpt") + fallback = CompletionResult("ok", None, "codex", "api", None, "openai/gpt", "api", "openai_api") + first = cache_fingerprint(request, subscription, input_identity="private prompt") + second = cache_fingerprint(request, fallback, input_identity="private prompt") + assert first != second + assert "private prompt" not in first + second diff --git a/tests/test_llm_coordination.py b/tests/test_llm_coordination.py new file mode 100644 index 0000000..d5dfb3c --- /dev/null +++ b/tests/test_llm_coordination.py @@ -0,0 +1,64 @@ +"""Run-scoped cross-process slots, circuits, and provenance.""" + +import multiprocessing +import os + +from optimize_anything.llm_backends import RunCoordinator + + +def _hold_slot(path: str, provider: str, ready, release) -> None: + coordinator = RunCoordinator.attach(path) + with coordinator.slot(provider): + ready.set() + release.wait(5) + + +def test_provider_capacity_one_across_processes() -> None: + with RunCoordinator.create() as coordinator: + context = multiprocessing.get_context("spawn") + ready = context.Event() + release = context.Event() + child = context.Process(target=_hold_slot, args=(str(coordinator.path), "codex", ready, release)) + child.start() + try: + assert ready.wait(5) + assert coordinator.try_acquire_slot("codex") is None + with coordinator.slot("claude"): + pass + finally: + release.set() + child.join(5) + assert child.exitcode == 0 + with coordinator.slot("codex"): + pass + + +def test_role_circuits_and_child_provenance_are_shared() -> None: + with RunCoordinator.create() as coordinator: + child = RunCoordinator.attach(str(coordinator.path)) + assert child.open_circuit("codex", "judge", "quota_exceeded") is True + assert coordinator.circuit_reason("codex", "judge") == "quota_exceeded" + assert coordinator.circuit_reason("codex", "proposer") is None + assert coordinator.open_circuit("codex", "judge", "rate_limit") is False + child.record_event({"role": "judge", "requested_backend": "codex", "actual_backend": "api", + "actual_model": "openai/fallback", "auth_class": "api", + "prompt": "must not persist", "api_key": "must not persist"}) + events = coordinator.events() + assert len(events) == 1 + assert events[0]["role"] == "judge" + state = (coordinator.path / "events.jsonl").read_text() + assert "must not persist" not in state + assert oct(os.stat(coordinator.path).st_mode & 0o777) == "0o700" + + +def test_provider_override_allows_exactly_two_slots() -> None: + with RunCoordinator.create({"codex": 2}) as coordinator: + first = coordinator.try_acquire_slot("codex") + second = coordinator.try_acquire_slot("codex") + try: + assert first is not None + assert second is not None + assert coordinator.try_acquire_slot("codex") is None + finally: + first.release() + second.release() diff --git a/tests/test_llm_factory.py b/tests/test_llm_factory.py new file mode 100644 index 0000000..9a123ba --- /dev/null +++ b/tests/test_llm_factory.py @@ -0,0 +1,55 @@ +from __future__ import annotations + +from optimize_anything.llm_backends.base import CompletionRequest, CompletionResult +from optimize_anything.llm_backends.factory import BackendLanguageModel, resolve_backend_spec + + +class _Backend: + def __init__(self) -> None: + self.request: CompletionRequest | None = None + + def complete(self, request: CompletionRequest) -> CompletionResult: + self.request = request + return CompletionResult( + text="improved", + structured=None, + requested_backend="codex", + actual_backend="codex", + requested_model=request.model, + actual_model=request.model, + auth_class="subscription", + auth_source="chatgpt", + role=request.role, + ) + + +def test_resolve_subscription_fallback_is_same_vendor_and_explicit(): + spec = resolve_backend_spec( + backend="codex", + model=None, + openai_api_fallback_model="openai/gpt-5.6-sol", + ) + + assert spec.api_fallback is True + assert spec.api_fallback_model == "openai/gpt-5.6-sol" + + +def test_no_api_fallback_overrides_resolvable_model(): + spec = resolve_backend_spec( + backend="claude", + model="anthropic/claude-sonnet-5", + no_api_fallback=True, + ) + + assert spec.api_fallback is False + + +def test_backend_language_model_adapts_gepa_prompt_lists(): + backend = _Backend() + language_model = BackendLanguageModel(backend, model="gpt-5.6-sol") + + assert language_model([{"role": "user", "content": "improve"}]) == "improved" + assert backend.request is not None + assert backend.request.role == "proposer" + assert backend.request.model == "gpt-5.6-sol" + assert '\"content\": \"improve\"' in backend.request.prompt diff --git a/tests/test_llm_fallback.py b/tests/test_llm_fallback.py new file mode 100644 index 0000000..3e3b4c5 --- /dev/null +++ b/tests/test_llm_fallback.py @@ -0,0 +1,133 @@ +"""Conservative subscription-to-API fallback behavior.""" + +from dataclasses import replace +from contextlib import contextmanager + +import pytest + +from optimize_anything.llm_backends import ( + AuthenticationError, + BackendUnavailable, + CompletionRequest, + CompletionResult, + FallbackBackend, + InvalidResponse, + RunCoordinator, + Timeout, +) + + +class FakeBackend: + def __init__(self, backend: str, result_or_error): + self.backend = backend + self.result_or_error = result_or_error + self.calls = 0 + + def preflight(self): + return None + + def complete(self, request): + self.calls += 1 + if isinstance(self.result_or_error, Exception): + raise self.result_or_error + return replace(self.result_or_error, requested_model=request.model) + + +def _result(backend, model, auth_class, auth_source): + return CompletionResult(text="ok", structured=None, requested_backend=backend, + actual_backend=backend, requested_model=model, actual_model=model, + auth_class=auth_class, auth_source=auth_source) + + +def test_fallback_is_same_vendor_sticky_per_role_and_warns_before_api(capsys): + primary = FakeBackend("codex", BackendUnavailable("unavailable")) + api = FakeBackend("api", _result("api", "openai/fallback", "api", "openai_api")) + wrapper = FallbackBackend(primary=primary, fallback=api, source_backend="codex", + fallback_model="openai/fallback", fallback_ready=lambda: True) + result = wrapper.complete(CompletionRequest(prompt="secret", role="judge")) + assert result.requested_backend == "codex" + assert result.actual_backend == "api" + assert result.fallback.reason == "backend_unavailable" + assert "API billing may apply" in capsys.readouterr().err + wrapper.complete(CompletionRequest(prompt="secret", role="judge")) + assert primary.calls == 1 + wrapper.complete(CompletionRequest(prompt="secret", role="score")) + assert primary.calls == 2 + + +@pytest.mark.parametrize("error", [Timeout("late"), InvalidResponse("bad")]) +def test_no_fallback_for_ambiguous_or_invalid_result(error): + primary = FakeBackend("claude", error) + api = FakeBackend("api", _result("api", "anthropic/fallback", "api", "anthropic_api")) + wrapper = FallbackBackend(primary=primary, fallback=api, source_backend="claude", + fallback_model="anthropic/fallback", fallback_ready=lambda: True) + with pytest.raises(type(error)): + wrapper.complete(CompletionRequest(prompt="x", role="judge")) + assert api.calls == 0 + + +def test_no_fallback_to_other_vendor_or_without_readiness(): + primary = FakeBackend("codex", AuthenticationError("logged out")) + api = FakeBackend("api", _result("api", "anthropic/fallback", "api", "anthropic_api")) + wrapper = FallbackBackend(primary=primary, fallback=api, source_backend="codex", + fallback_model="anthropic/fallback", fallback_ready=lambda: True) + with pytest.raises(AuthenticationError): + wrapper.complete(CompletionRequest(prompt="x", role="judge")) + assert api.calls == 0 + + +def test_preflight_failure_routes_first_request_directly_to_api(capsys): + primary = FakeBackend("claude", _result("claude", "sonnet", "subscription", "claude_subscription")) + primary.preflight = lambda: (_ for _ in ()).throw(AuthenticationError("logged out")) + api = FakeBackend("api", _result("api", "anthropic/fallback", "api", "anthropic_api")) + wrapper = FallbackBackend(primary=primary, fallback=api, source_backend="claude", + fallback_model="anthropic/fallback", fallback_ready=lambda: True) + assert wrapper.preflight().ready + wrapper.complete(CompletionRequest(prompt="x", role="judge")) + assert primary.calls == 0 + assert api.calls == 1 + assert "API billing may apply" in capsys.readouterr().err + + +def test_fallback_provenance_is_shared_with_child_coordinator(): + with RunCoordinator.create() as coordinator: + child = RunCoordinator.attach(str(coordinator.path)) + primary = FakeBackend("codex", BackendUnavailable("unavailable")) + api = FakeBackend("api", _result("api", "openai/fallback", "api", "openai_api")) + wrapper = FallbackBackend(primary=primary, fallback=api, source_backend="codex", + fallback_model="openai/fallback", fallback_ready=lambda: True, + coordinator=child) + wrapper.complete(CompletionRequest(prompt="private prompt", role="judge")) + event = coordinator.events()[0] + assert (event["role"], event["requested_backend"], event["actual_backend"]) == ( + "judge", "codex", "api") + assert "private prompt" not in str(event) + + +def test_queued_call_rechecks_circuit_after_acquiring_slot(): + class Coordinator: + def circuit_reason(self, provider, role): + return "rate_limit" if self.acquired else None + + @contextmanager + def slot(self, provider, *, timeout_seconds): + self.acquired = True + yield + + def record_event(self, event): + pass + + acquired = False + + primary = FakeBackend("codex", _result("codex", "gpt", "subscription", "chatgpt")) + api = FakeBackend("api", _result("api", "openai/fallback", "api", "openai_api")) + wrapper = FallbackBackend( + primary=primary, fallback=api, source_backend="codex", + fallback_model="openai/fallback", fallback_ready=lambda: True, + coordinator=Coordinator(), + ) + + result = wrapper.complete(CompletionRequest(prompt="x", role="judge")) + + assert result.actual_backend == "api" + assert primary.calls == 0 diff --git a/tests/test_spec_loader.py b/tests/test_spec_loader.py index 123938f..e1a01ce 100644 --- a/tests/test_spec_loader.py +++ b/tests/test_spec_loader.py @@ -40,6 +40,31 @@ def test_model_section_parsed(self, tmp_path: Path): assert result["proposer_model"] == "anthropic/claude-sonnet-4-6" assert result["judge_model"] == "openai/gpt-4o-mini" + def test_structured_subscription_model_roles(self, tmp_path: Path): + spec_file = tmp_path / "opt.toml" + spec_file.write_text( + """ +[model.proposer] +backend = "codex" +api_fallback = false + +[model.judge] +backend = "claude" +model = "sonnet" +api_fallback_model = "anthropic/claude-sonnet-5" +""" + ) + + result = load_spec(spec_file) + + assert result["proposer_backend"] == "codex" + assert result["proposer_model"] is None + assert result["proposer_api_fallback"] is False + assert result["judge_backend"] == "claude" + assert result["judge_model"] == "sonnet" + assert result["judge_api_fallback"] is True + assert result["judge_api_fallback_model"] == "anthropic/claude-sonnet-5" + def test_task_model_parsed_from_optimization_section(self, tmp_path: Path): spec_file = tmp_path / "opt.toml" diff --git a/tests/test_subscription_live.py b/tests/test_subscription_live.py new file mode 100644 index 0000000..7c21aa2 --- /dev/null +++ b/tests/test_subscription_live.py @@ -0,0 +1,116 @@ +"""Opt-in live gates for locally authenticated subscription backends.""" + +from __future__ import annotations + +import os +import io +import json +import sys +from pathlib import Path + +import pytest + +from optimize_anything.llm_backends.base import CompletionRequest, thaw_json + + +pytestmark = [ + pytest.mark.integration, + pytest.mark.skipif( + os.environ.get("OPTIMIZE_ANYTHING_RUN_SUBSCRIPTION_LIVE") != "1", + reason="set OPTIMIZE_ANYTHING_RUN_SUBSCRIPTION_LIVE=1 to use local subscription quota", + ), +] + +_SCHEMA = { + "type": "object", + "properties": {"ok": {"type": "boolean"}}, + "required": ["ok"], + "additionalProperties": False, +} + + +@pytest.mark.parametrize("provider", ["codex", "claude"]) +def test_saved_subscription_structured_completion(provider, monkeypatch): + monkeypatch.setenv("OPENAI_API_KEY", "deliberately-invalid-live-gate") + monkeypatch.setenv("ANTHROPIC_API_KEY", "deliberately-invalid-live-gate") + if provider == "codex": + from optimize_anything.llm_backends.codex_backend import CodexSdkBackend + + backend = CodexSdkBackend() + expected_source = "chatgpt" + else: + from optimize_anything.llm_backends.claude_backend import ClaudeCliBackend + + backend = ClaudeCliBackend() + expected_source = "claude_subscription" + + status = backend.preflight() + result = backend.complete( + CompletionRequest( + prompt="Return JSON with ok set to true.", + role="validation", + output_schema=_SCHEMA, + timeout_seconds=120, + ) + ) + + assert status.auth_class == "subscription" + assert result.auth_class == "subscription" + assert result.auth_source == expected_source + assert thaw_json(result.structured) == {"ok": True} + + +@pytest.mark.parametrize("provider", ["codex", "claude"]) +def test_saved_subscription_seedless_budget_one(provider, monkeypatch, capsys): + from optimize_anything.cli import main + + monkeypatch.setenv("OPENAI_API_KEY", "deliberately-invalid-live-gate") + monkeypatch.setenv("ANTHROPIC_API_KEY", "deliberately-invalid-live-gate") + evaluator = Path(__file__).parents[1] / "examples/evaluators/echo_score.sh" + + return_code = main([ + "optimize", + "--no-seed", + "--objective", + "Write a concise friendly greeting.", + "--budget", + "1", + "--proposer-backend", + provider, + "--no-api-fallback", + "--evaluator-command", + "bash", + str(evaluator), + ]) + + captured = capsys.readouterr() + assert return_code == 0 + assert f'"actual_backend": "{provider}"' in captured.out + assert '"auth_class": "subscription"' in captured.out + + +@pytest.mark.parametrize("provider", ["codex", "claude"]) +def test_generated_evaluator_uses_saved_subscription(provider, monkeypatch): + from optimize_anything.evaluator_generator import generate_evaluator_script + + monkeypatch.setenv("OPENAI_API_KEY", "deliberately-invalid-live-gate") + monkeypatch.setenv("ANTHROPIC_API_KEY", "deliberately-invalid-live-gate") + script = generate_evaluator_script( + seed="hello", + objective="Score clarity.", + evaluator_type="judge", + backend=provider, + model=None, + api_fallback=False, + ) + output = io.StringIO() + monkeypatch.setattr(sys, "stdin", io.StringIO('{"candidate":"Hello there."}\n')) + monkeypatch.setattr(sys, "stdout", output) + namespace = {"__name__": "generated_evaluator"} + exec(compile(script, "", "exec"), namespace) + + assert namespace["main"]() == 0 + result = json.loads(output.getvalue()) + assert 0.0 <= result["score"] <= 1.0 + assert result["llm_provenance"]["actual_backend"] == provider + assert result["llm_provenance"]["auth_class"] == "subscription" diff --git a/uv.lock b/uv.lock index 4121c0e..a9b97e1 100644 --- a/uv.lock +++ b/uv.lock @@ -454,7 +454,7 @@ name = "exceptiongroup" version = "1.3.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "typing-extensions", marker = "python_full_version < '3.13'" }, + { name = "typing-extensions" }, ] sdist = { url = "https://files.pythonhosted.org/packages/50/79/66800aadf48771f6b62f7eb014e352e5d06856655206165d775e675a02c9/exceptiongroup-1.3.1.tar.gz", hash = "sha256:8b412432c6055b0b7d14c310000ae93352ed6754f70fa8f7c34141f91c4e3219", size = 30371, upload-time = "2025-11-21T23:01:54.787Z" } wheels = [ @@ -1363,6 +1363,35 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/cc/56/0a89092a453bb2c676d66abee44f863e742b2110d4dbb1dbcca3f7e5fc33/openai-2.21.0-py3-none-any.whl", hash = "sha256:0bc1c775e5b1536c294eded39ee08f8407656537ccc71b1004104fe1602e267c", size = 1103065, upload-time = "2026-02-14T00:11:59.603Z" }, ] +[[package]] +name = "openai-codex" +version = "0.156.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "openai-codex-cli-bin" }, + { name = "packaging" }, + { name = "pydantic" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/fe/a2/c1d378fef1f1a04b57af6af968bde66dc68849ce1ab91c6beb03782430b7/openai_codex-0.156.0.tar.gz", hash = "sha256:2d766358edce206325a8ee6ccd5f54250b6743f72ac58ba53c3e0706e2f7caee", size = 87934, upload-time = "2026-09-22T20:03:30.398Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2c/65/7e1bde4685f3cf3811f3a6b7f56148908104d4030a96c68506a68f423cc0/openai_codex-0.156.0-py3-none-any.whl", hash = "sha256:2fbd38e12e79091d2aa1f2ae92d97457289daa6ab9f7c2231deba337e703fe3f", size = 96057, upload-time = "2026-09-22T20:03:28.814Z" }, +] + +[[package]] +name = "openai-codex-cli-bin" +version = "0.156.0" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c6/5a/6ab804125153352a6d7d39518d893c6252cf1682013b0b3e78abc08ac1bc/openai_codex_cli_bin-0.156.0-py3-none-macosx_10_9_x86_64.whl", hash = "sha256:af36bbb750a808351020c29519d16e421625ed29c800a087d182cba4b87ee616", size = 130007766, upload-time = "2026-09-22T20:01:25.593Z" }, + { url = "https://files.pythonhosted.org/packages/ea/42/734e6fd84cb149a33b8982da52b3e6a820c4e3e7694ba6b176ce3993759f/openai_codex_cli_bin-0.156.0-py3-none-macosx_11_0_arm64.whl", hash = "sha256:b0fe3f27224ea22454f78d92d3022560f3126c25e8b063c53a85cba9ca5fd2f8", size = 119159163, upload-time = "2026-09-22T20:01:34.139Z" }, + { url = "https://files.pythonhosted.org/packages/30/49/e93dbdf994988b0defd5f4124d9aba3895ecc31fc4597adbdc4819b22a48/openai_codex_cli_bin-0.156.0-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:7b7b0c436e8af55aad9a619849ab5d78e92e8b6d2649109a1255350e0869c429", size = 127264576, upload-time = "2026-09-22T20:01:42.59Z" }, + { url = "https://files.pythonhosted.org/packages/73/a9/da942602042b9942727ce22691e2bc6237888adf938fddc246a35ed30f8d/openai_codex_cli_bin-0.156.0-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:f299e5b5ef595108333f86ea9f1b9fa03beeaf25266d50e29ef21c7b6f50ded0", size = 136206270, upload-time = "2026-09-22T20:01:53.22Z" }, + { url = "https://files.pythonhosted.org/packages/04/59/afb7a7114b70a827985c5fa13f917ec8dcce5eb56550b207a5970226002a/openai_codex_cli_bin-0.156.0-py3-none-musllinux_1_1_aarch64.whl", hash = "sha256:1363e1972a80e1b51a89ef37d1a36edce21492dcd569d4427b34b73a96ae2489", size = 136890375, upload-time = "2026-09-22T20:02:07.27Z" }, + { url = "https://files.pythonhosted.org/packages/6e/42/4a63ef1c381614f99540c2ac6f25c373f2fdbacd7e77ff4d4cb7bdee77e9/openai_codex_cli_bin-0.156.0-py3-none-musllinux_1_1_x86_64.whl", hash = "sha256:331910a4d991d1e8819705cf9deb4504bd8aa8bd38e1dcd777289742c090af80", size = 145971890, upload-time = "2026-09-22T20:02:17.877Z" }, + { url = "https://files.pythonhosted.org/packages/38/d9/6fae394b013049f475bbb34a87579ec6c950b7646adac007691c6695aa9b/openai_codex_cli_bin-0.156.0-py3-none-win_amd64.whl", hash = "sha256:f2e0dad06bbb212b8a8a02c09a402d1e4074d6e082a43c43ba45941e55056a84", size = 145787414, upload-time = "2026-09-22T20:02:28.31Z" }, + { url = "https://files.pythonhosted.org/packages/37/92/7c04977dd1a45a446c70f88d26a1c64cca6c433056a6989c04f25a9aa015/openai_codex_cli_bin-0.156.0-py3-none-win_arm64.whl", hash = "sha256:81a70445465572e7bd5fd0b7743f50fe2b2e0768b090c265c23fd869978eae7e", size = 134742526, upload-time = "2026-09-22T20:02:39.276Z" }, +] + [[package]] name = "optimize-anything" version = "0.5.1" @@ -1371,9 +1400,15 @@ dependencies = [ { name = "cloudpickle" }, { name = "gepa" }, { name = "httpx" }, + { name = "jsonschema" }, { name = "litellm" }, ] +[package.optional-dependencies] +codex = [ + { name = "openai-codex" }, +] + [package.dev-dependencies] dev = [ { name = "mypy" }, @@ -1387,8 +1422,11 @@ requires-dist = [ { name = "cloudpickle" }, { name = "gepa", specifier = ">=0.1.4,<0.2.0" }, { name = "httpx" }, + { name = "jsonschema", specifier = ">=4.26,<5" }, { name = "litellm", specifier = "!=1.82.7,!=1.82.8,>=1.83.0,<1.92" }, + { name = "openai-codex", marker = "extra == 'codex'", specifier = ">=0.156.0,<0.157.0" }, ] +provides-extras = ["codex"] [package.metadata.requires-dev] dev = [ @@ -1400,11 +1438,11 @@ dev = [ [[package]] name = "packaging" -version = "26.0" +version = "26.3" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/65/ee/299d360cdc32edc7d2cf530f3accf79c4fca01e96ffc950d8a52213bd8e4/packaging-26.0.tar.gz", hash = "sha256:00243ae351a257117b6a241061796684b084ed1c516a08c48a3f7e147a9d80b4", size = 143416, upload-time = "2026-01-21T20:50:39.064Z" } +sdist = { url = "https://files.pythonhosted.org/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", hash = "sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79", size = 313412, upload-time = "2026-08-04T18:15:28.737Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b7/b9/c538f279a4e237a006a2c98387d081e9eb060d203d8ed34467cc0f0b9b53/packaging-26.0-py3-none-any.whl", hash = "sha256:b36f1fef9334a5588b4166f8bcd26a14e521f2b55e6b9de3aaa80d3ff7a37529", size = 74366, upload-time = "2026-01-21T20:50:37.788Z" }, + { url = "https://files.pythonhosted.org/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", hash = "sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c", size = 129956, upload-time = "2026-08-04T18:15:27.159Z" }, ] [[package]]