diff --git a/.github/workflows/opencode-review-dispatch.yml b/.github/workflows/opencode-review-dispatch.yml index dd65d90e1..0df7a17cc 100644 --- a/.github/workflows/opencode-review-dispatch.yml +++ b/.github/workflows/opencode-review-dispatch.yml @@ -4058,8 +4058,8 @@ jobs: "apiKey": "{env:OPENAI_API_KEY}" }, "models": { - "gpt-5.6-luna": { - "name": "OpenAI GPT-5.6 Luna (direct)", + "gpt-5.4": { + "name": "OpenAI GPT-5.4 (direct)", "tool_call": true, "reasoning": true, "options": { @@ -4471,17 +4471,17 @@ jobs: # or used for product/model improvement, so private repositories # include neither NIM nor anonymous free candidates and start at the # existing keyed fallback list: OpenCode Zen GPT-5.6 Terra, DeepSeek - # V3, the direct GPT-5.6 Luna slot, and pinned PAID + # V3, the direct GPT-5.4 slot, and pinned PAID # OpenRouter coder models (free-tier candidates hit the shared # free-models-per-day cap and hung for the full candidate timeout, # so the OpenRouter slots use cheap paid models billed against the # org's OpenRouter credits), then the full-size GPT-4.1 long-context # endpoint and provider-specific GPT/o3 fallbacks. - # The direct-OpenAI slot runs GPT-5.6 Luna: the newest family's - # cost-efficient tier, cheaper than the legacy gpt-5 it replaced - # ($1/$6 vs $1.25/$10 per 1M tokens) so the org OpenAI budget - # stretches further between top-ups. - OPENCODE_MODEL_CANDIDATES: "${{ needs.validate-pr-metadata.outputs.is_private == 'false' && 'nvidia-nim/nvidia/llama-3.3-nemotron-super-49b-v1.5 nvidia-nim/nvidia/llama-3.1-nemotron-ultra-253b-v1 nvidia-nim/nvidia/nemotron-3-super-120b-a12b nvidia-nim/nvidia/nemotron-3-ultra-550b-a55b nvidia-nim/meta/llama-3.3-70b-instruct nvidia-nim/deepseek-ai/deepseek-v4-pro nvidia-nim/mistralai/codestral-22b-instruct-v0.1 opencode-free/nemotron-3-ultra-free opencode-free/deepseek-v4-flash-free opencode-free/north-mini-code-free opencode-free/laguna-s-2.1-free opencode-free/ling-3.0-flash-free opencode-free/big-pickle opencode-free/mimo-v2.5-free opencode-free/hy3-free opencode-free/minimax-m3-free opencode-free/glm-5-free opencode-free/kimi-k2.5-free opencode-free/qwen3.6-plus-free ' || '' }}opencode/gpt-5.6-terra github-models/deepseek/deepseek-v3-0324 openai/gpt-5.6-luna openrouter/deepseek/deepseek-v3.2 openrouter/qwen/qwen3-coder github-models/openai/gpt-4.1 github-models/openai/gpt-5 github-models/openai/gpt-5-chat github-models/openai/o3 github-models/deepseek/deepseek-r1-0528 github-models/deepseek/deepseek-r1" + # The direct-OpenAI slot runs GPT-5.4: gpt-5.6-luna returns 404 on + # the OpenAI API (see a724582), so the pool keeps the newest VALID + # direct-OpenAI model instead of burning a candidate on a certain + # failure. + OPENCODE_MODEL_CANDIDATES: "${{ needs.validate-pr-metadata.outputs.is_private == 'false' && 'nvidia-nim/nvidia/llama-3.3-nemotron-super-49b-v1.5 nvidia-nim/nvidia/llama-3.1-nemotron-ultra-253b-v1 nvidia-nim/nvidia/nemotron-3-super-120b-a12b nvidia-nim/nvidia/nemotron-3-ultra-550b-a55b nvidia-nim/meta/llama-3.3-70b-instruct nvidia-nim/deepseek-ai/deepseek-v4-pro nvidia-nim/mistralai/codestral-22b-instruct-v0.1 opencode-free/nemotron-3-ultra-free opencode-free/deepseek-v4-flash-free opencode-free/north-mini-code-free opencode-free/laguna-s-2.1-free opencode-free/ling-3.0-flash-free opencode-free/big-pickle opencode-free/mimo-v2.5-free opencode-free/hy3-free opencode-free/minimax-m3-free opencode-free/glm-5-free opencode-free/kimi-k2.5-free opencode-free/qwen3.6-plus-free ' || '' }}opencode/gpt-5.6-terra github-models/deepseek/deepseek-v3-0324 openai/gpt-5.4 openrouter/deepseek/deepseek-v3.2 openrouter/qwen/qwen3-coder github-models/openai/gpt-4.1 github-models/openai/gpt-5 github-models/openai/gpt-5-chat github-models/openai/o3 github-models/deepseek/deepseek-r1-0528 github-models/deepseek/deepseek-r1" # One attempt per model, then fall through to the next model. Retrying # the SAME model 5x let a rate-limited/hung leader consume the whole # step, so the pool never reached a healthy fallback model. diff --git a/.github/workflows/strix.yml b/.github/workflows/strix.yml index f89119070..5a1ef688d 100644 --- a/.github/workflows/strix.yml +++ b/.github/workflows/strix.yml @@ -863,6 +863,17 @@ jobs: export "STRIX_PROCESS_${budget_suffix}_SECONDS=$process_budget_seconds" export "STRIX_TOTAL_${budget_suffix}_SECONDS=5700" + # Recognized signals that the LLM backend was unavailable / starved. + # Defined before the gate loop so the bounded retry decision below + # can classify outcomes without duplicating the patterns later. + backend_unavailable_signal='RateLimitError|Too many requests\. For more on scraping GitHub|exceeded your current quota|insufficient_quota|billing details|"status"[[:space:]]*:[[:space:]]*"RESOURCE_EXHAUSTED"|tokens_limit_reached|Request body too large|Max size:[[:space:]]*[0-9]+[[:space:]]+tokens|Error code:[[:space:]]*413|LLM CONNECTION FAILED|Could not establish connection to the language model|LLM warm-up failed|Configured model and fallback models were unavailable|Configured Vertex model and fallback models were unavailable|emitted provider infrastructure or failure-signal output|before provider infrastructure failure|litellm(\.exceptions)?\.NotFoundError[^[:cntrl:]]*Nvidia_nimException[^[:cntrl:]]*Error code:[[:space:]]*404|Error during penetration test: loginAsGuest failed after [0-9]+ attempts: curl exit 7: curl: \(7\) Failed to connect to 127\.0\.0\.1 port 48080' + model_behavior_error_signal='(^|[^A-Za-z0-9_])(agents|pydantic_ai|strix)(\.[A-Za-z_][A-Za-z0-9_]*)*\.ModelBehaviorError([^A-Za-z0-9_]|$)' + # Any evidence that a vulnerability was actually reported. Its presence + # forces a hard failure so real findings are NEVER downgraded. Keep the + # severity branch anchored away from identifiers so environment lines + # such as STRIX_FAIL_ON_MIN_SEVERITY do not look like findings. + reported_vulnerability_signal='Vulnerabilities[[:space:]]+[1-9]|(^|[^A-Za-z0-9_])severity[[:space:]]*:' + # Capture the gate exit code plus its console output. The gate returns # exit 1 both for genuine blocking vulnerabilities AND for # LLM-backend-unavailable outcomes (GitHub Models "Too many requests" @@ -871,11 +882,60 @@ jobs: # could not complete a scan. Provider failure is typed infrastructure # evidence, but remains non-passing because no authoritative complete # vulnerability result exists. + # + # A typed provider outage with no reported vulnerability finding is + # retried with bounded linear backoff inside this step so transient + # provider failures do not fail the required check on the first + # attempt. Genuine findings, configuration failures, and unexpected + # exit codes never retry; the deadline keeps every path inside the + # deterministic 120-minute job budget, and all-terminal outcomes + # remain fail-closed. strix_run_log="$RUNNER_TEMP/strix_gate_console.log" strix_rc=0 + strix_gate_attempt=1 + strix_gate_deadline=$(( SECONDS + 6000 )) set +e - bash "$TRUSTED_STRIX_GATE" 2>&1 | tee "$strix_run_log" - strix_rc="${PIPESTATUS[0]}" + while : ; do + : > "$strix_run_log" + bash "$TRUSTED_STRIX_GATE" 2>&1 | tee "$strix_run_log" + strix_rc="${PIPESTATUS[0]}" + if [ "$strix_rc" -eq 0 ]; then + break + fi + # Only exit-code 1 scan failures can be infrastructure outcomes. + if [ "$strix_rc" -ne 1 ]; then + break + fi + # Scope this attempt's retry decision to the log tail after the + # last pipeline-continuation marker, exactly like the terminal + # classification below: an already-exempted finding before the + # marker must not mask a retryable outage after it. + strix_retry_scope_log="$strix_run_log" + if grep -Fq 'allowing pipeline continuation' "$strix_run_log"; then + strix_retry_scope_log="$RUNNER_TEMP/strix_gate_console_tail.log" + awk '/allowing pipeline continuation/{buf=""; next} {buf=buf $0 "\n"} END{printf "%s", buf}' \ + "$strix_run_log" > "$strix_retry_scope_log" + fi + # A reported vulnerability is authoritative evidence: never retry + # and never risk downgrading it. + if grep -Eiq "$reported_vulnerability_signal" "$strix_retry_scope_log"; then + break + fi + # Retry only recognized provider-outage / model-behavior classes. + if ! grep -Eiq "$backend_unavailable_signal" "$strix_retry_scope_log" \ + && ! grep -Eq "$model_behavior_error_signal" "$strix_retry_scope_log"; then + break + fi + remaining_seconds=$(( strix_gate_deadline - SECONDS )) + if [ "$strix_gate_attempt" -ge 3 ] || [ "$remaining_seconds" -lt 600 ]; then + echo "Provider-unavailable Strix attempt ${strix_gate_attempt} reached the bounded retry limit or the remaining job time budget (${remaining_seconds}s) is too small to retry; failing closed." >&2 + break + fi + backoff_seconds=$(( ${STRIX_GATE_RETRY_BACKOFF_SECONDS:-90} * strix_gate_attempt )) + echo "Strix provider outage on attempt ${strix_gate_attempt}; retrying after ${backoff_seconds}s backoff." >&2 + sleep "$backoff_seconds" + strix_gate_attempt=$(( strix_gate_attempt + 1 )) + done set -e if [ "$strix_rc" -eq 0 ]; then @@ -889,15 +949,6 @@ jobs: exit "$strix_rc" fi - # Recognized signals that the LLM backend was unavailable / starved. - backend_unavailable_signal='RateLimitError|Too many requests\. For more on scraping GitHub|exceeded your current quota|insufficient_quota|billing details|"status"[[:space:]]*:[[:space:]]*"RESOURCE_EXHAUSTED"|tokens_limit_reached|Request body too large|Max size:[[:space:]]*[0-9]+[[:space:]]+tokens|Error code:[[:space:]]*413|LLM CONNECTION FAILED|Could not establish connection to the language model|LLM warm-up failed|Configured model and fallback models were unavailable|Configured Vertex model and fallback models were unavailable|emitted provider infrastructure or failure-signal output|before provider infrastructure failure|litellm(\.exceptions)?\.NotFoundError[^[:cntrl:]]*Nvidia_nimException[^[:cntrl:]]*Error code:[[:space:]]*404|Error during penetration test: loginAsGuest failed after [0-9]+ attempts: curl exit 7: curl: \(7\) Failed to connect to 127\.0\.0\.1 port 48080' - model_behavior_error_signal='(^|[^A-Za-z0-9_])(agents|pydantic_ai|strix)(\.[A-Za-z_][A-Za-z0-9_]*)*\.ModelBehaviorError([^A-Za-z0-9_]|$)' - # Any evidence that a vulnerability was actually reported. Its presence - # forces a hard failure so real findings are NEVER downgraded. Keep the - # severity branch anchored away from identifiers so environment lines - # such as STRIX_FAIL_ON_MIN_SEVERITY do not look like findings. - reported_vulnerability_signal='Vulnerabilities[[:space:]]+[1-9]|(^|[^A-Za-z0-9_])severity[[:space:]]*:' - # An earlier out-of-scope/below-threshold finding may already have # been exempted by the trusted gate. Classify a later provider # outage from the tail after the last continuation marker, but keep diff --git a/tests/test_opencode_agent_contract.py b/tests/test_opencode_agent_contract.py index 8ee6e86fc..2360fdb26 100644 --- a/tests/test_opencode_agent_contract.py +++ b/tests/test_opencode_agent_contract.py @@ -186,7 +186,7 @@ def test_opencode_model_pool_sets_high_effort_for_capable_candidates(): ["opencode-free", "qwen3.6-plus-free"], ["opencode", "gpt-5.6-terra"], ["github-models", "deepseek/deepseek-v3-0324"], - ["openai", "gpt-5.6-luna"], + ["openai", "gpt-5.4"], ["openrouter", "deepseek/deepseek-v3.2"], ["openrouter", "qwen/qwen3-coder"], ["github-models", "openai/gpt-4.1"], @@ -197,7 +197,7 @@ def test_opencode_model_pool_sets_high_effort_for_capable_candidates(): ["github-models", "deepseek/deepseek-r1"], ] assert zen_models == ["gpt-5.6-terra"] - assert direct_openai_models == ["gpt-5.6-luna"] + assert direct_openai_models == ["gpt-5.4"] assert openrouter_models == [ "deepseek/deepseek-v3.2", "qwen/qwen3-coder", @@ -1740,7 +1740,7 @@ def test_workflow_provisions_sandbox_tool_and_reviewer_agent(): assert ( "opencode/gpt-5.6-terra " "github-models/deepseek/deepseek-v3-0324 " - "openai/gpt-5.6-luna " + "openai/gpt-5.4 " "openrouter/deepseek/deepseek-v3.2 " "openrouter/qwen/qwen3-coder " "github-models/openai/gpt-4.1 " @@ -1887,7 +1887,7 @@ def test_workflow_provisions_sandbox_tool_and_reviewer_agent(): ) assert ( "github-models/deepseek/deepseek-v3-0324 " - "openai/gpt-5.6-luna " + "openai/gpt-5.4 " "openrouter/deepseek/deepseek-v3.2 " "openrouter/qwen/qwen3-coder " "github-models/openai/gpt-4.1 " diff --git a/tests/test_pr_review_autofix_nvidia_nim_contract.py b/tests/test_pr_review_autofix_nvidia_nim_contract.py index 799b9e9fb..a5d25379a 100644 --- a/tests/test_pr_review_autofix_nvidia_nim_contract.py +++ b/tests/test_pr_review_autofix_nvidia_nim_contract.py @@ -19,7 +19,7 @@ DOCTORING_RECORD = Path("docs/doctoring/hourly-nvidia-nim-autofix.md") CHANGELOG = Path("CHANGELOG.md") REVIEW_DISPATCH_WORKFLOW = Path(".github/workflows/opencode-review-dispatch.yml") -REVIEW_DISPATCH_BLOB_SHA = "dd65d90e10e5040562b501ade1a40f89572f0984" +REVIEW_DISPATCH_BLOB_SHA = "0df7a17cc72a79585cec169c8299e0646f93ab02" def _workflow_text(path: Path) -> str: diff --git a/tests/test_required_workflow_queue_contract.py b/tests/test_required_workflow_queue_contract.py index e58f5e6c0..1d79f1daa 100644 --- a/tests/test_required_workflow_queue_contract.py +++ b/tests/test_required_workflow_queue_contract.py @@ -505,7 +505,7 @@ def test_nvidia_nim_defaults_preserve_existing_fallbacks_without_secret( assert strix.returncode == 0, strix.stderr assert { "provider_mode=openai_direct", - "strix_model=gpt-5.6-luna", + "strix_model=gpt-5.4", } <= set(strix_output.read_text().splitlines()) assert ( "STRIX_MODEL: ${{ steps.gate.outputs.strix_model }}" diff --git a/tests/test_strix_backend_unavailable_after_exempted_finding.py b/tests/test_strix_backend_unavailable_after_exempted_finding.py index 3355a8448..433a5a250 100644 --- a/tests/test_strix_backend_unavailable_after_exempted_finding.py +++ b/tests/test_strix_backend_unavailable_after_exempted_finding.py @@ -21,6 +21,7 @@ from __future__ import annotations import re +import shlex import subprocess import tempfile import unittest @@ -63,9 +64,12 @@ def _extract_neutralization_block(workflow: str) -> str: stale logic. """ - start_marker = ( - " # Recognized signals that the LLM backend was unavailable" - ) + # The classification block starts at the neutralization-scope assignment + # and runs to the terminal failure exit. The gate-execution retry loop and + # the raw signal definitions live outside this region; the signal values + # are injected by _run_gate_tail so the extracted decision logic stays the + # single tested authority. + start_marker = ' strix_neutralization_scope_log="$strix_run_log"' terminal_failure_marker = ( ' echo "Strix reported security findings or failed for a ' 'non-backend reason; failing the required check' @@ -77,6 +81,23 @@ def _extract_neutralization_block(workflow: str) -> str: return workflow[start:end] +def _extract_signal_definitions(workflow: str) -> str: + """Return the canonical backend-outage / finding-signal definitions. + + Bounded by the same unique anchors used in production so the injected + patterns cannot drift from the ones the gate itself classifies with. + """ + + start_marker = ( + " # Recognized signals that the LLM backend was unavailable" + ) + end_marker = "reported_vulnerability_signal=" + start = workflow.index(start_marker) + end = workflow.index(end_marker, start) + end = workflow.index("\n", end) + 1 + return workflow[start:end] + + def _run_gate_tail(log_text: str) -> int: """Execute the extracted block against a synthetic log; return its exit code. @@ -85,6 +106,7 @@ def _run_gate_tail(log_text: str) -> int: """ workflow = STRIX_WORKFLOW.read_text(encoding="utf-8") + signals = _extract_signal_definitions(workflow) block = _extract_neutralization_block(workflow) with tempfile.TemporaryDirectory(prefix="strix-tail-scope-") as temp_dir: strix_run_log = Path(temp_dir) / "strix_gate_console.log" @@ -94,6 +116,7 @@ def _run_gate_tail(log_text: str) -> int: "set -uo pipefail", 'strix_run_log="$1"', "strix_rc=1", + signals, block, ) ) @@ -113,6 +136,69 @@ def _run_gate_tail(log_text: str) -> int: return completed.returncode + +def _extract_retry_loop_region(workflow: str) -> str: + """Return the bounded provider-outage retry region, verbatim from the yml. + + Spans the signal definitions through the post-loop success exit so the + retry decision, its tail-scoping, and its terminal success path are all + exercised against a scripted fake gate. + """ + + start_marker = ( + " # Recognized signals that the LLM backend was unavailable" + ) + end_marker = ( + " # Preserve configuration failures (exit 2) and any unexpected exit" + ) + start = workflow.index(start_marker) + end = workflow.index(end_marker, start) + return workflow[start:end] + + +def _run_gate_retry(gate_script: str) -> tuple[int, int]: + """Run the extracted retry loop against a scripted gate; return (rc, calls). + + The fake gate appends one line to a call-counter file on every invocation + so tests can prove exactly how many attempts the loop spent. + """ + + workflow = STRIX_WORKFLOW.read_text(encoding="utf-8") + signals = _extract_signal_definitions(workflow) + region = _extract_retry_loop_region(workflow) + with tempfile.TemporaryDirectory(prefix="strix-retry-scope-") as temp_dir: + counter = Path(temp_dir) / "gate_calls" + counter.write_text("0\n", encoding="utf-8") + gate_path = Path(temp_dir) / "fake_gate.sh" + gate_path.write_text( + gate_script.replace("__COUNTER__", str(counter)), + encoding="utf-8", + ) + gate_path.chmod(0o755) + script = "\n".join( + ( + "set -uo pipefail", + f"export TRUSTED_STRIX_GATE={shlex.quote(str(gate_path))}", + "export RUNNER_TEMP=" + shlex.quote(temp_dir), + "export STRIX_GATE_RETRY_BACKOFF_SECONDS=1", + signals, + region, + 'exit "$strix_rc"', + ) + ) + completed = subprocess.run( + ["bash", "-c", script], + check=False, + capture_output=True, + text=True, + env={"RUNNER_TEMP": temp_dir, "PATH": "/usr/bin:/bin"}, + ) + calls = int(counter.read_text().strip()) + if completed.stdout.endswith(f"RC={completed.returncode}"): + pass + return completed.returncode, calls + + class StrixBackendUnavailableAfterExemptedFindingTests(unittest.TestCase): """Protect the PR #392-shaped scenario without weakening the real gate.""" @@ -153,6 +239,42 @@ def test_bare_backend_outage_with_no_finding_is_non_passing( self.assertEqual(_run_gate_tail(GITHUB_MODELS_BROWNOUT), 1) + def test_exempted_finding_then_outage_recovers_on_second_attempt(self) -> None: + """An exempt finding before continuation must not block outage retry.""" + + gate = r"""#!/usr/bin/env bash +calls=$(( $(cat __COUNTER__) + 1 )) +echo "$calls" > __COUNTER__ +if [ "$calls" -le 1 ]; then + printf '%s\n' \ + "Strix findings are limited to unchanged files in this pull request; allowing pipeline continuation." \ + "LLM CONNECTION FAILED" \ + "Configured model and fallback models were unavailable." + exit 1 +fi +echo "scan complete" +exit 0 +""" + returncode, calls = _run_gate_retry(gate) + self.assertEqual(returncode, 0) + self.assertEqual(calls, 2) + + def test_real_finding_after_continuation_never_retries(self) -> None: + """A tail-scoped real finding is authoritative: zero retries, fail closed.""" + + gate = r"""#!/usr/bin/env bash +calls=$(( $(cat __COUNTER__) + 1 )) +echo "$calls" > __COUNTER__ +printf '%s\n' \ + "Strix findings are limited to unchanged files in this pull request; allowing pipeline continuation." \ + "LLM CONNECTION FAILED" \ + "Vulnerability Report" "Severity: CRITICAL" "Vulnerabilities 1" +exit 1 +""" + returncode, calls = _run_gate_retry(gate) + self.assertEqual(returncode, 1) + self.assertEqual(calls, 1) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_strix_nvidia_nim_not_found_fallback.py b/tests/test_strix_nvidia_nim_not_found_fallback.py index 990269725..17f0e9a30 100644 --- a/tests/test_strix_nvidia_nim_not_found_fallback.py +++ b/tests/test_strix_nvidia_nim_not_found_fallback.py @@ -192,7 +192,7 @@ def test_workflow_uses_available_free_first_nvidia_plan(self) -> None: workflow = STRIX_WORKFLOW.read_text(encoding="utf-8") default_expression = ( "steps.target_visibility.outputs.is_private == 'false' && " - f"'{DEFAULT_NVIDIA_MODEL}' || 'gpt-5.6-luna'" + f"'{DEFAULT_NVIDIA_MODEL}' || 'gpt-5.4'" ) self.assertIn(default_expression, workflow) self.assertIn( @@ -202,7 +202,7 @@ def test_workflow_uses_available_free_first_nvidia_plan(self) -> None: ) self.assertIn( "steps.gate.outputs.provider_mode == 'nvidia_nim' && " - f"'{FREE_NVIDIA_FALLBACK} openai-direct/gpt-5.6-luna'", + f"'{FREE_NVIDIA_FALLBACK} openai-direct/gpt-5.4'", workflow, )