From 3410975537070593359a99e692b882ce9866b772 Mon Sep 17 00:00:00 2001 From: lanerchenbuna Date: Tue, 22 Sep 2026 22:00:01 +0800 Subject: [PATCH] fix: close the defect ledger, make workspace paths install-safe, correct the docs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Clears every defect in the review ledger except the provider-deadline item, fixes the documentation claims that had drifted from the code, and removes development-only files from the repository. Correctness - Unified terminal outcomes. `succeeded` / `needs_clarification` / `blocked` / `partial` / `failed` / `cancelled` are derived in `queryforge/core/outcomes.py`, so the persisted run status, the delivery report and the event stream can no longer disagree. A quality gate that blocks a run is no longer reported as completed. - Reflection that asks for clarification used to raise, discarding the answer and the run's artifacts. It now terminates as a structured `needs_clarification`. - The `reasoning` audit payload was silently dropped whenever the model returned type-compatible but schema-incompatible shapes (string lists, the string "None", word confidences such as "high"). Shapes are normalised and a discarded payload is now reported instead of ignored. - Preview execution evaluated a rewritten statement, so a `require_limit` policy was satisfied by the injected LIMIT rather than by the caller's SQL. - Time-filter validation ignored `BETWEEN` and per-call boundaries, so a partially bounded time filter could validate as complete. - Truncated result sets overwrote the row count, hiding that the bound had been hit; `truncated` and `fetched_row_count` now preserve both numbers. - Chinese follow-up questions were not recognised, so multi-turn context was lost. Route matching for report/explain intents is no longer triggered by a bare substring, so a table named `sales_report` is not routed as a report request. Governance and audit - `DatabaseTool.last_policy_decision` is read-only. It can no longer be overwritten by a caller, which had let the tool report an audit record for a call that never happened. - Execution fingerprints bind data, semantic and policy versions, persisted in the run journal, so a resumed run cannot silently reuse a step computed against a different model or policy. - Budget state is inherited across a resume instead of restarting at zero. - Cross-instance journal leases are exclusive (`fcntl.flock` plus a reload before acquiring), so two workers can no longer hold the same lease. - CSV knowledge is trusted only when a reviewer is recorded; the governed knowledge build path is reachable from the CLI (`--kb-knowledge`). Measurement - The evaluator compares result projections tolerantly: a correct answer is no longer scored wrong for returning extra columns or an equivalent shape. - Account-level failures (HTTP 402 and friends) are classified as environment errors and excluded from accuracy denominators instead of being counted as model failures. - Token usage is reported as measured when the provider returns it, with `token_source` distinguishing measured from estimated figures. - `--skill-mode auto|off` and `--parallel-candidates` make the skill-selection and candidate-count questions testable; the `skills=[]` default that silently disabled the skill catalogue in every measured run is gone. - The evaluator's isolation guard is recursive and covers plain `import` statements. Repository and structure - Workspace-relative paths (`.queryforge/` state, sample data, evaluation sets) are resolved by `queryforge/core/paths.py` from `QUERYFORGE_ROOT`, an enclosing source checkout, or the working directory. Deriving them from `Path(__file__).parents[N]` pointed into `site-packages` after an install, so run state moved into the installed package. Packaged resources such as `bundled_skills/` deliberately still resolve relative to the package. - `make` targets prefer the repository virtualenv, so `make check` uses the same interpreter as `./init.sh` instead of whatever `python` is first on `PATH`. - Development-only files (process plans, defect ledgers, session notes) are removed; `./init.sh` is now a single offline verification entry point. - Studio uploads accept only csv/parquet, matching what publication can accept; a rejected database file explains why. `docs/` is reorganised and its index no longer points two entries at the same page. Documentation - Both READMEs state measured behaviour with its sample size and its caveats, and the stale figures are corrected: 928 tests (was 691/806), 23/23 tier-1 tasks over 32 gold tasks across 3 schemas (was 32/32), and a real-model NL2SQL result (was "not run here"). Claims about automatic skill selection and the PostgreSQL backend are marked as unestablished and unverified rather than implied working. - `docs/nl2sql_evaluation.md` records the frozen baselines, and its token and cost section no longer describes a heuristic the code no longer uses. - CHANGELOG.md lists the above. Verification - `./init.sh` — 928 tests, 25 skipped, 0 failures - `make check` — 13/13 offline acceptance checks - `make web-check` — eslint, tsc, production build, 24 TypeScript tests - `python scripts/check_repository.py` — no broken links, no secrets, hygiene clean Not verified: no tier-3 real-model run was repeated after these changes, so accuracy remains 0.875 on 40 anime cases from a single run, where +/-0.03 is noise. The PostgreSQL backend is still unverified against a live server. --- .env.example | 4 +- .github/workflows/model-eval.yml | 48 +- .gitignore | 6 +- AGENTS.md | 136 ++++ CHANGELOG.md | 58 ++ Makefile | 17 + README.md | 105 ++- README.zh-CN.md | 91 ++- docs/README.md | 42 +- docs/agent_team_architecture.md | 18 +- docs/demo/run_demo_d.py | 13 +- docs/nl2sql_evaluation.md | 54 +- evaluation/gold/context_and_compound.jsonl | 21 + evaluation/gold/nl2sql_multidomain.jsonl | 240 +++---- init.sh | 104 +++ queryforge/application/agent_service.py | 28 +- queryforge/application/analysis_planner.py | 29 + queryforge/cli.py | 40 ++ queryforge/core/config.py | 6 +- queryforge/core/observability.py | 144 +++- queryforge/core/outcomes.py | 143 ++++ queryforge/core/paths.py | 89 +++ queryforge/core/schemas/__init__.py | 2 + queryforge/core/schemas/models.py | 73 ++ queryforge/domain/semantic/builder.py | 6 +- queryforge/domain/semantic/sql_validator.py | 172 ++++- queryforge/domain/skills/registry.py | 4 + queryforge/evaluation/__init__.py | 2 +- queryforge/evaluation/evaluator.py | 2 +- queryforge/evaluation/thresholds.py | 11 +- queryforge/infrastructure/db/adapter.py | 26 +- .../infrastructure/db/sqlite_connector.py | 10 +- queryforge/infrastructure/models/base.py | 38 +- .../models/providers/openai_compatible.py | 10 +- .../infrastructure/storage/knowledge_base.py | 26 +- .../storage/sql_history_store.py | 7 +- .../infrastructure/storage/vector_store.py | 5 +- .../infrastructure/tools/database_tool.py | 113 ++- .../orchestration/agents/entry_router.py | 35 +- .../orchestration/agents/product_analyst.py | 96 ++- .../orchestrator/orchestrator.py | 96 ++- queryforge/orchestration/planner/executor.py | 32 +- .../runtime/execution_journal.py | 193 ++++- queryforge/orchestration/runtime/resume.py | 5 +- queryforge/orchestration/schemas/__init__.py | 5 + .../schemas/knowledge_versions.py | 48 +- queryforge/orchestration/tools/budget.py | 35 + queryforge/workflow/budgeted_model.py | 162 +++++ queryforge/workflow/node/gen_sql_node.py | 130 +++- queryforge/workflow/node/output_node.py | 162 ++++- queryforge/workflow/node/plan_output_node.py | 13 +- .../workflow/node/visualization_node.py | 3 +- queryforge/workflow/workflow.py | 133 +++- queryforge/workflow/workflow_runner.py | 46 +- scripts/evaluate_sql.py | 677 +++++++++++++++++- scripts/generate_context_gold.py | 482 +++++++++++++ tests/test_agent_task_gold.py | 15 +- tests/test_agent_team_router_orchestrator.py | 270 ++++++- tests/test_analysis_planner.py | 12 +- tests/test_answer_evidence.py | 84 +++ tests/test_cli_kb_governance.py | 99 ++- tests/test_conversation_memory.py | 74 ++ tests/test_database_tool.py | 109 +++ tests/test_db_adapter_contract.py | 21 + tests/test_evaluate_sql.py | 549 ++++++++++++++ tests/test_knowledge_csv_trust.py | 106 +++ tests/test_model_budget.py | 233 ++++++ tests/test_observability.py | 44 ++ tests/test_retry_workflow.py | 35 +- tests/test_run_outcomes.py | 113 +++ tests/test_semantic_sql_validator.py | 60 ++ tests/test_workspace_root.py | 145 ++++ web/README.md | 2 +- web/app/api/studio/upload/route.ts | 30 +- web/app/page.tsx | 2 +- 75 files changed, 5869 insertions(+), 430 deletions(-) create mode 100644 AGENTS.md create mode 100644 evaluation/gold/context_and_compound.jsonl create mode 100755 init.sh create mode 100644 queryforge/core/outcomes.py create mode 100644 queryforge/core/paths.py create mode 100644 queryforge/workflow/budgeted_model.py create mode 100644 scripts/generate_context_gold.py create mode 100644 tests/test_knowledge_csv_trust.py create mode 100644 tests/test_model_budget.py create mode 100644 tests/test_run_outcomes.py create mode 100644 tests/test_workspace_root.py diff --git a/.env.example b/.env.example index 01f9283..d951e5f 100644 --- a/.env.example +++ b/.env.example @@ -55,7 +55,9 @@ SQL_SECURITY_POLICY_PATH= DOMAIN_REGISTRY_PATH=.queryforge/domains/registry.json # Transport hardening for network deployments (REST / SSE / Gateway / MCP). -# When QUERYFORGE_API_KEY is set, every endpoint except /health requires +# When QUERYFORGE_API_KEY is set, every endpoint except the four public +# paths requires Authorization: Bearer or X-API-Key: +# /health, /openapi.json, /docs, /redoc # `Authorization: Bearer ` or `X-API-Key: `. QUERYFORGE_API_KEY= # Comma-separated files/directories that remote callers may open as `database` diff --git a/.github/workflows/model-eval.yml b/.github/workflows/model-eval.yml index 2f00fe6..a1f599c 100644 --- a/.github/workflows/model-eval.yml +++ b/.github/workflows/model-eval.yml @@ -38,10 +38,32 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 60 env: + # ``load_config`` only reads the *provider-specific* variables declared in + # ``models.yml`` (openai -> OPENAI_API_KEY / OPENAI_MODEL / + # OPENAI_BASE_URL, and so on). A generic ``LLM_API_KEY`` is never consulted + # by the loader, so subscribing only that name let the credential gate below + # pass while every real model call failed with "No API key is configured + # for provider 'openai'". Every provider therefore maps its own secret here. + # + # All of them are populated from the same optional secret so a single + # repository secret can drive whichever provider is dispatched; a provider + # whose variable stays empty simply fails at the gate below. + # + # Only the api-key variables are fanned out. The model/base-url variables + # are deliberately NOT set here: ``--model-provider``/``--model`` are passed + # explicitly to both evaluation steps, so a fanned-out ``*_MODEL`` would be + # a second, contradicting source of truth. LLM_PROVIDER: ${{ github.event.inputs.provider || 'openai' }} LLM_MODEL: ${{ github.event.inputs.model || 'gpt-4o-mini' }} LLM_API_KEY: ${{ secrets.LLM_API_KEY }} LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }} + OPENAI_API_KEY: ${{ secrets.LLM_API_KEY }} + ANTHROPIC_API_KEY: ${{ secrets.LLM_API_KEY }} + GEMINI_API_KEY: ${{ secrets.LLM_API_KEY }} + DEEPSEEK_API_KEY: ${{ secrets.LLM_API_KEY }} + QWEN_API_KEY: ${{ secrets.LLM_API_KEY }} + GLM_API_KEY: ${{ secrets.LLM_API_KEY }} + OPENAI_BASE_URL: ${{ secrets.LLM_BASE_URL }} steps: - uses: actions/checkout@v4 - uses: actions/setup-python@v5 @@ -55,16 +77,36 @@ jobs: - run: python sample/generate_aux_datasets.py - name: Require provider credentials run: | - if [ -z "${LLM_API_KEY}" ]; then - echo "LLM_API_KEY is not configured; tier 3 cannot run." >&2 - echo "Add the repository secret or dispatch this workflow manually with credentials." >&2 + # Check the variable ``load_config`` actually reads for the dispatched + # provider — not the generic LLM_API_KEY, which the loader ignores. + case "${LLM_PROVIDER}" in + openai) key="${OPENAI_API_KEY}" ;; + claude) key="${ANTHROPIC_API_KEY}" ;; + gemini) key="${GEMINI_API_KEY}" ;; + deepseek) key="${DEEPSEEK_API_KEY}" ;; + qwen) key="${QWEN_API_KEY:-${DASHSCOPE_API_KEY}}" ;; + glm) key="${GLM_API_KEY:-${ZAI_API_KEY}}" ;; + *) + echo "Unsupported provider '${LLM_PROVIDER}' for this workflow." >&2 + exit 1 + ;; + esac + if [ -z "${key}" ]; then + echo "No API key is configured for provider '${LLM_PROVIDER}'; tier 3 cannot run." >&2 + echo "Add the repository secret LLM_API_KEY (mapped to the provider-specific variable)" >&2 + echo "or dispatch this workflow manually with credentials." >&2 exit 1 fi - name: Evaluate the NL2SQL gold set with a real model run: | + # ``--model-provider`` / ``--model`` are passed explicitly: without them + # this step silently ignored the dispatched ``model`` input and used the + # provider default from models.yml instead. python scripts/evaluate_sql.py \ --cases evaluation/gold/nl2sql_multidomain.jsonl \ --limit "${{ github.event.inputs.limit || 40 }}" \ + --model-provider "${LLM_PROVIDER}" \ + --model "${LLM_MODEL}" \ --output evaluation/reports/nl2sql_model_eval.json - name: Tier-3 agent task report (recorded, not gating) run: | diff --git a/.gitignore b/.gitignore index d0ef4cd..5cd4400 100644 --- a/.gitignore +++ b/.gitignore @@ -24,6 +24,8 @@ Thumbs.db *.sqlite-wal *.build.json semantic-weekly-report.json -# Generated benchmark reports (the summarized evidence lives in -# docs/optimization/step-16-acceptance.md; CI uploads its own artifacts). +# Generated benchmark reports: raw JSON is regenerable, so it is not tracked. +# The summarized, versioned evidence lives in docs/optimization/baselines.md +# (the comment here previously pointed at docs/optimization/step-16-acceptance.md, +# which does not exist in the repository). evaluation/reports/ diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..b3942ee --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,136 @@ +# AGENTS.md + +Project harness for agent-assisted development on **QueryForge** — a governed +NL2SQL / data-analysis platform (Python 3.11/3.12 + Next.js Studio). + +Keep this file short. Project facts live in `docs/`; this file is routing and invariants. + +## Startup Workflow + +Before writing code: + +1. **Confirm the working directory is the repository root** with `pwd` — the directory + containing `pyproject.toml` and this file +2. **Read this file** completely +3. **Run `./init.sh`** — checks the environment, runs the full offline test suite, reports + repository state. Exits non-zero on failure. +4. **Read the routed docs** for the area you are touching (see "Where Facts Live") + +If baseline verification fails, **repair that first**. Do not add scope on top of a red baseline. + +## Invariants — Do Not Violate + +These are load-bearing. Any change that weakens them is a regression, not a refactor. + +1. **Single SQL execution boundary.** `DatabaseTool` is the only sanctioned SQL execution + path. `DatabaseAdapter.execute_sql` is a *trusted primitive* that performs no policy + check — never call it directly from application code. +2. **Deterministic gates outrank models.** Final truth for these is decided by code, never by + an LLM verdict: + - `SQLPolicyEngine` (`queryforge/domain/security/sql_policy.py`) — AST policy + - `SemanticSQLValidator` (`queryforge/domain/semantic/sql_validator.py`) — business semantics + - `EvidenceStore` (`queryforge/domain/analysis/evidence.py`) — evidence ids + - `QualityGateEvaluator` (`queryforge/orchestration/gates.py`) — phase gates + - Terminal run status (`queryforge/core/outcomes.py`) + A model may *propose*. It may not declare success. +3. **Fail closed.** Unknown table/column, unsupported shape, or unverifiable claim must + surface as `unsupported` / `blocked` / `unverified` — never silently as `success`. +4. **Read-only by default.** Analysis opens the database read-only; write/admin SQL is + rejected at the AST layer *and* at the engine layer. + +## Working Rules + +- **No completion claim without evidence.** Run the relevant verification command and report + its actual output. "Should work" is not evidence. +- **Stay in scope.** Do not opportunistically refactor unrelated modules. +- **Prefer existing mechanisms over new ones.** Tool permissions, `PlanValidator`, the + ablation switches, the evidence layer and journal/resume already exist — check before + adding a mechanism. +- **Leave the repo verifiable.** `./init.sh` must pass when you stop. + +## Verification Commands + +```bash +./init.sh # full: environment + test suite + repo state + +# Individual checks (from the venv) +LOG_LEVEL=CRITICAL .venv/bin/python -m unittest discover -s tests -q +make check # repository hygiene / required files + offline acceptance +make acceptance # offline acceptance only + +# Evaluation (tier-1 is offline; tier-3 costs real API spend) +LOG_LEVEL=CRITICAL .venv/bin/python -B scripts/benchmark_agent.py --tier 1 --report /tmp/tier1.json +``` + +**Static and build checks** — `./init.sh` does not run these; run the relevant one when you +touch that area: + +```bash +make check # repository hygiene (check_repository.py) + offline acceptance +make web-check # Studio: eslint + tsc --noEmit + build + node --test (TypeScript) +make web-build # Studio production build +make semantic-check # semantic-model drift against the sample database +``` + +> **Tier-3 requires credentials and a funded account — ask the human before running it.** +> It spends real money. See "Known Traps" below. + +## Known Traps + +Each of these has already cost time or produced a wrong conclusion. Read before touching +the relevant area. + +1. **`evaluation/reports/` is gitignored.** Raw evaluator JSON there is regenerable and + untracked. Any number you want to cite must be written into a tracked document, with the + command that produced it and the versions it depends on. +2. **Tier-1 "all green" does not measure model capability.** `scripts/benchmark_runners.py` + swaps in `ScriptedModel`, which returns `reference_sql` verbatim. Tier-1 measures the + governance pipeline and control flow only. +3. **Tier-3 needs a funded account.** A depleted balance returns HTTP 402. Such cases are + classified as `environment_error` and excluded from the accuracy denominators, but the run + still cannot complete. Confirm balance before a baseline run. +4. **A candidate/no-candidate comparison is confounded unless you force the count.** The gold + set's per-case `candidate_selection` flag correlates perfectly with category + (`multi_table`/`metric` set it; `single_table`/`time` never do), so grouping by it measures + task difficulty, not the mechanism. Use `--parallel-candidates N` to hold inputs fixed. +5. **Provider credentials are provider-specific.** `load_config` reads `DEEPSEEK_API_KEY` / + `OPENAI_API_KEY` / … as declared in `models.yml`. A generic `LLM_API_KEY` is silently + ignored. `.env` is gitignored — never commit it. +6. **`ScriptedModel` vs real model, in the same runner.** `runner: "scripted_workflow"` tasks + call the real model when a provider is configured (tier-3) and the fake one otherwise. + Do not conclude from the runner name alone. +7. **Do not trust a measurement without reproducing its verdict against the data.** Several + "semantic errors" in the first baseline were the evaluator's fault, not the model's, and a + "parallel candidates are 3× slower" finding was pure difficulty confounding. Verify the + per-case evidence before quoting an aggregate. +8. **Check what the evaluator is *suppressing*, not just what it measures.** It once passed + `skills=[]` for every case, which takes the manual skill path and silently disables + automatic skill selection — so the catalogue was inert in every measured run and the + headline accuracy described a configuration that does not match production. Use + `--skill-mode auto` (the default) for a production-aligned number. +9. **A run that fails in ~60 ms with zero measured tokens never called the model.** That is + the signature of a provider-contract mismatch (for example a wrapper whose + `generate_with_messages` lags the adapter signature), not of bad model output. +10. **A single run of ~30 cases cannot resolve small accuracy differences.** Deltas of ±0.03 + have been observed across *identical* code. Do not present such a difference as an + improvement; increase `--repeat` instead. + +## Escalation + +- **Architecture decisions** → read `docs/agent_team_architecture.md` and `docs/README.md`, then ask. +- **Anything that changes an invariant above** → ask before implementing. +- **Tier-3 / any real API spend** → ask before running. +- **Repeated test failures** → report them rather than weakening the assertions. + +## Where Facts Live + +| Need | Doc | +|---|---| +| Documentation index | `docs/README.md` | +| Architecture overview | `docs/agent_team_architecture.md` | +| Configuration / env vars | `docs/configuration.md` | +| Semantic model authoring | `docs/semantic_authoring.md`, `docs/semantic_contracts.md` | +| REST / MCP surfaces | `docs/api_reference.md`, `docs/mcp_server.md` | +| Evaluation method + frozen numbers | `docs/nl2sql_evaluation.md` | +| Database backends | `docs/database_adapters.md` | +| Release process | `docs/github_release.md` | diff --git a/CHANGELOG.md b/CHANGELOG.md index 27ee0f7..b0d0d87 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -17,8 +17,66 @@ and does not yet claim semantic-versioning stability. trust inspection, and persistent run history. - D1/R2-backed atomic uploads that require a reviewed semantic contract. - GitHub Actions quality gates and repository contribution/security metadata. +- Unified terminal outcomes (`succeeded`, `needs_clarification`, `blocked`, + `partial`, `failed`, `cancelled`) derived in one place, so a run's persisted + status, its delivery report and its event stream cannot disagree. +- A deterministic 32-task agent benchmark over three independent SQLite schemas, + with split separation, repeats, an ablation switch and a frozen effect gate. +- A 21-case multi-turn and compound-intent gold set that separates capability + failures from infrastructure failures. +- Run-budget coverage of every model call, including a persisted usage record and a + `RunContext` carrying data, semantic and policy versions. +- `docs/nl2sql_evaluation.md` now records the frozen real-model baselines and the + `--skill-mode auto|off` and candidate-count ablations, with their caveats. +- `./init.sh` — a single offline verification entry point (environment, full test + suite, repository state). ### Changed - Canonical CLI implementation now lives in `queryforge.cli`; root `main.py` remains a compatibility launcher. +- Workspace-relative paths (`.queryforge/` state, sample data, evaluation sets) are + resolved by `queryforge.core.paths` from `QUERYFORGE_ROOT`, an enclosing source + checkout, or the working directory — instead of `Path(__file__).parents[N]`, which + pointed into `site-packages` after an install. Packaged resources such as + `bundled_skills/` deliberately still resolve relative to the package. +- `SQLiteConnector.capabilities` is taken from the single frozen capability matrix + rather than rebuilt from dataclass defaults, so a reader of that attribute now + sees what the adapter actually enforces. +- `DatabaseTool.last_policy_decision` is read-only: it records decisions for calls + made through that tool and can no longer be overwritten by a caller, which had let + an audit record describe a call that never happened. +- Automatic skill selection is now measured rather than assumed: it costs p50 + latency +139% and output tokens +44% with no demonstrated accuracy benefit, so it + is switchable and the question is recorded as open. +- For complex requests the candidate count is no longer raised implicitly; parallel + candidates are an explicit opt-in. +- The evaluator compares result projections tolerantly and classifies account-level + failures as environment errors instead of model errors, and reports measured token + usage rather than a character-count estimate. +- `make` targets prefer the repository virtualenv, so `make check` runs the same + interpreter as `./init.sh` instead of whatever `python` is first on `PATH`. + +### Fixed + +- Correct answers were scored wrong when the model returned extra columns or an + equivalent shape; correctness is now decided by semantic result equivalence. +- A reflection step that asked for clarification discarded the answer and the run's + artifacts; it now terminates as a structured `needs_clarification`. +- Chinese follow-up questions were not recognised, so multi-turn context was lost. +- A depleted provider balance was recorded as a model failure, corrupting accuracy + denominators; it is now an environment error, excluded from those denominators. +- The `reasoning` audit payload was silently dropped when the model returned + type-compatible-by-intent but schema-incompatible shapes (string lists, the string + `"None"`, word confidences); it is now normalised, and a discarded payload is + reported rather than ignored. +- Preview execution evaluated a rewritten statement, so a `require_limit` policy was + satisfied by the injected `LIMIT` rather than by the caller's SQL. +- Truncated result sets overwrote the row count, hiding that the bound had been hit; + `truncated` and `fetched_row_count` now preserve both numbers. +- Time-filter validation ignored `BETWEEN` and per-call time boundaries, so an + unbounded or partially bounded time filter could validate as complete. +- Studio uploads accepted `.sqlite`/`.db` files that publication could never accept; + the accepted set is now csv/parquet, and rejected database files explain why. +- The evaluator's isolation guard (independently graded code must not import the + runtime it grades) is now recursive and covers plain `import` statements. diff --git a/Makefile b/Makefile index 6170032..1bc9c55 100644 --- a/Makefile +++ b/Makefile @@ -2,6 +2,23 @@ PYTHON ?= python +# Every target below needs the project's dependencies (sqlglot, pydantic, ...). +# ``PYTHON ?= python`` alone takes whatever ``python`` is first on PATH, which on a +# machine with a system or conda interpreter is *not* the project venv, so +# ``make check`` failed with ``ModuleNotFoundError: No module named 'sqlglot'`` +# while ``init.sh`` (which hardcodes ``.venv/bin/python``) stayed green — the +# verification command documented in AGENTS.md could not be run as written. +# ``?=`` reports origin as "file", so we cannot tell a caller-supplied value from +# our own fallback by origin alone. Test the value instead: if it is still the bare +# ``python`` we defaulted to, upgrade it to the repository venv. An explicit +# ``make PYTHON=/usr/bin/python3.12 check`` (or an environment ``PYTHON``) names a +# different interpreter and is left untouched. +ifeq ($(PYTHON),python) +ifneq ($(wildcard .venv/bin/python),) +PYTHON := .venv/bin/python +endif +endif + help: @echo "install Install QueryForge in editable mode" @echo "install-all Install all optional integrations" diff --git a/README.md b/README.md index 3884079..c28c6a0 100644 --- a/README.md +++ b/README.md @@ -14,8 +14,8 @@ policy enforcement, bounded recovery, and production-friendly delivery interface ![Python](https://img.shields.io/badge/Python-3.11%20%7C%203.12-3776AB?logo=python&logoColor=white) ![SQLite](https://img.shields.io/badge/SQLite-read--only-003B57?logo=sqlite&logoColor=white) ![SQLGlot](https://img.shields.io/badge/SQL%20policy-SQLGlot-6B4FBB) -![Tests](https://img.shields.io/badge/tests-691%20passing-2EA44F) -![Semantic contracts](https://img.shields.io/badge/semantic%20checks-82%20passing-7C3AED) +![Tests](https://img.shields.io/badge/tests-928%20passing-2EA44F) +![Offline acceptance](https://img.shields.io/badge/acceptance-13%2F13-7C3AED) @@ -29,12 +29,37 @@ Users create or select a data domain first—such as retail, finance, product, o the bundled Anime Streaming sample—then onboard that domain's data, review its semantic contract, and ask questions inside the same governance boundary. QueryForge combines that workflow with natural-language-to-SQL, AST-level -security, read-only execution, multi-candidate selection, repair budgets, and -complete run artifacts. +security, read-only execution, bounded recovery, and complete run artifacts. > QueryForge currently targets SQLite and controlled environments. It is a > portfolio-grade reference architecture, not a multi-tenant analytics service. +### Measured behaviour + +The governance layer is deterministic, so it is tested exhaustively offline; the +model layer is not, so its numbers are reported separately and with their sample +size. Both are reproducible from this repository. + +| What | Result | How | +| --- | --- | --- | +| Offline test suite | 928 passing, 25 skipped | `./init.sh` | +| Offline acceptance gate | 13/13 checks | `make check` | +| Deterministic agent benchmark | 23/23 tasks (`dev` + `regression` splits; the 9-task `holdout` split is requested explicitly) | `python scripts/benchmark_agent.py --tier 1 --gate` | +| Real-model NL2SQL accuracy | **0.875** semantic correctness, 1.0 execution success | 40 cases, single run, `deepseek-v4-flash` | + +Two honest qualifications on that last row, because they matter more than the +number: + +- It is **one run of 40 cases**. Differences of ±0.03 have been observed across + *identical* code, so this figure cannot resolve small changes. +- It covers the anime sample domain only. It is evidence that the pipeline works + end to end on a real model, not a general accuracy claim. + +Tier-1's 23/23 measures the *engineering* chain (governance, execution, evidence, +budgeting, failure classification), not model capability: a fixture supplies the +SQL. See [NL2SQL evaluation](docs/nl2sql_evaluation.md) for the method and the +frozen baselines. + ## Product Tour
@@ -69,7 +94,7 @@ delivery loop: | Keep business meaning consistent | Define metrics, dimensions, grain, and join paths in YAML | | Prevent context from leaking | Scope sources, semantic contracts, policies, and run history to a selected data domain | | Recover from imperfect output | Reflect, repair, and retry within explicit budgets | -| Handle harder questions | Use bounded schema discovery and parallel SQL candidates | +| Handle harder questions | Use bounded schema discovery, a tool loop, and serviceable failure classification | | Trace what happened | Persist run state, policy decisions, quality evidence, and artifacts | | Integrate with other tools | Expose CLI, REST/SSE, MCP, gateway, JSON, charts, and HTML reports | | Start from raw data | Build governed SQLite assets from CSV, Parquet, and paginated JSON APIs | @@ -81,13 +106,14 @@ delivery loop: - **Semantic contracts** — YAML models describe business metrics, entities, relationships, cardinality, ownership, SLA, sensitivity, and quality rules. - **Adaptive workflow** — simple questions stay fast; complex questions can - activate a bounded tool loop and concurrent candidate selection. + activate a bounded tool loop. - **Read-only by default** — normal analysis opens SQLite databases in read-only mode and rejects write or administrative SQL. - **Multiple delivery surfaces** — use the same application service through the CLI, REST/SSE, MCP, or a webhook gateway. -- **Reproducible evaluation** — the repository includes offline acceptance checks - and a 120-case, three-domain NL2SQL gold set. +- **Reproducible evaluation** — the repository ships a 32-task deterministic agent + benchmark over three independent schemas and a 120-case, three-domain NL2SQL + gold set, with the frozen real-model baselines recorded in the docs. ## Quick Start @@ -373,7 +399,7 @@ python -m queryforge.interfaces.mcp.server --transport stdio | --- | --- | --- | | Studio | `make web-dev` | Data-domain management, visual onboarding, semantic authoring, and governed analysis | | CLI | `queryforge --question "..."` | Local exploration and engineering workflows | -| REST | `POST /ask` and `POST /plan` | Application integration | +| REST | `POST /ask` (conversational), `POST /analyze` (planner) | Application integration | | SSE | `POST /ask/stream` | Progress-aware clients | | MCP | `queryforge.interfaces.mcp.server` | IDEs and MCP-compatible assistants | | Gateway | `POST /gateway/webhook` | Stable user/channel session adapters | @@ -385,15 +411,18 @@ python -m queryforge.interfaces.mcp.server --transport stdio queryforge/ ├── cli.py # Installed CLI implementation ├── application/ # Transport-neutral service facade and resources -├── core/ # Configuration, schemas, and observability +├── core/ # Configuration, schemas, workspace paths, observability ├── data_assets/ # Ingestion, quality, lineage, and publication ├── domain/ # SQL policy, semantics, contracts, and skills -├── infrastructure/ # SQLite, model providers, storage, and tools +├── infrastructure/ # Database adapters, model providers, storage, and tools +├── evaluation/ # Benchmark thresholds and evaluator-side contract rules ├── interfaces/ # CLI-adjacent API, MCP, and gateway adapters ├── orchestration/ # Router, role agents, lifecycle, and state -└── workflow/ # NL2SQL nodes, selection, repair, and reporting +├── workflow/ # NL2SQL nodes, selection, repair, and reporting +└── bundled_skills/ # Prompt-only skill definitions shipped with the package evaluation/gold/ # Multi-domain NL2SQL evaluation cases +evaluation/tasks/ # Deterministic agent-benchmark tasks (dev/regression/holdout) sample_data/ # Ready-to-run SQLite datasets and semantic models web/ # QueryForge Studio and hosted persistence adapters scripts/ # Build, benchmark, evaluation, and acceptance tools @@ -410,17 +439,12 @@ infrastructure, and core contracts. Run the complete offline quality gate: ```bash -python scripts/run_acceptance.py --full -``` - -Or run the test suite directly: - -```bash -python -m unittest discover -s tests -q +./init.sh # environment + 928-test suite + repository state +make check # repository hygiene + 13 offline acceptance checks ``` Live model evaluation reports execution success, semantic equivalence, policy -precision/recall, latency, estimated cost, and candidate-selection uplift: +precision/recall, latency, measured token usage, and projection tolerance: ```bash python scripts/evaluate_sql.py \ @@ -429,34 +453,37 @@ python scripts/evaluate_sql.py \ --output .queryforge/evaluations/openai.json ``` -CI runs the offline acceptance gate (including the deterministic agent benchmark) on Python 3.11 and 3.12, plus an integration job that requires the optional transport dependencies. +CI runs the offline acceptance gate (including the deterministic agent benchmark) +on Python 3.11 and 3.12, plus an integration job that requires the optional +transport dependencies. Real-model evaluation is a manual workflow +(`.github/workflows/model-eval.yml`) because it spends money. ## What is verified (and what is not) -Every claim in this section is reproducible from the repository; the linked -acceptance record contains the gaps as well as the passes. +Every claim in this section is reproducible from the repository. The point of the +table is the third column: what has *not* been shown is stated as plainly as what has. | Capability | How you can check it | Status | | --- | --- | --- | -| Full offline test suite | `make test` — **806 tests, 0 skipped** | verified | +| Full offline test suite | `./init.sh` — **928 tests, 25 skipped, 0 failures** | verified | | Repository + integration gate | `make check` (`scripts/run_acceptance.py --full`, 13/13 checks) | verified | -| End-to-end demos (upload → publish → query; semantic catch; multi-step analysis; transports/refusal/recovery) | `make demo` — four narrated, asserting scripts under `docs/demo/` | verified | -| Deterministic agent benchmark (32 gold tasks, 3 independent schemas, ablation, effect gate) | `python scripts/benchmark_agent.py --tier 1 --gate` | verified (32/32) | +| End-to-end demos (upload → publish → query; semantic catch; multi-step analysis; transports/refusal/recovery) | `make demo` — five narrated, asserting scripts under `docs/demo/`, offline and key-free | verified | +| Deterministic agent benchmark (32 gold tasks over 3 independent schemas, ablation, effect gate) | `python scripts/benchmark_agent.py --tier 1 --gate` | verified (23/23 — the `dev` + `regression` splits; the 9-task `holdout` split must be requested with `--split holdout`) | | Optional-dependency integration tier | `python scripts/benchmark_agent.py --tier 2 --gate` — a missing dependency **fails** the tier | verified with `.[api,mcp]` installed | -| Real-model NL2SQL evaluation | `python scripts/evaluate_sql.py --cases evaluation/gold/nl2sql_multidomain.jsonl --model-provider

--model ` | **not run here** — no numbers, no accuracy claim | +| Real-model NL2SQL evaluation | `python scripts/evaluate_sql.py --cases evaluation/gold/nl2sql_multidomain.jsonl --model-provider

--model ` | **0.875 semantic correctness on 40 anime cases, one run, `deepseek-v4-flash`** — see the caveats above | +| Automatic skill selection is worth its cost | `--skill-mode auto` vs `--skill-mode off` | **not established** — it costs +139% p50 latency and +44% output tokens with no measured accuracy benefit | +| PostgreSQL backend | `pip install '.[postgres]'`, then `PostgresConnector` | **implemented, not verified** against a live server, and not exported from the package API | -Demo output is offline and deterministic (no model call, no network, no API key). -The agent benchmark's tier 1 gives the SQL as a fixture, so its 32/32 measures the -*engineering* chain (governance, execution, evidence, budget, failure -classification) — **not model accuracy**. Real-model numbers must come from a -tier-3 run with credentials and are reported separately -(`.github/workflows/model-eval.yml`). +The demos and tier-1 are offline and deterministic: no model call, no network, no +API key. The agent benchmark's tier 1 supplies the SQL as a fixture, so its 23/23 +measures the engineering chain — **not model accuracy**. Real-model numbers must +come from a tier-3 run with credentials and are reported separately. Deployment level: **controlled environment, single tenant, read-only data access**. -SQLite is the default backend; a DuckDB adapter exists behind an optional extra -(see [Database adapters](docs/database_adapters.md)). The system is not hardened -for arbitrary untrusted multi-tenant input, and the known gaps are listed per -per capability in the docs listed above; the two honest blank spots are real-model evaluation (no accuracy numbers) and the PostgreSQL backend (implemented, not yet verified against a live server). +SQLite is the default backend; DuckDB and PostgreSQL adapters exist behind optional +extras (see [Database adapters](docs/database_adapters.md)). The system is not +hardened for arbitrary untrusted multi-tenant input; the honest blank spots are +general-domain model accuracy and the PostgreSQL backend. ## Documentation @@ -494,7 +521,9 @@ project does **not** currently include: - production authentication, authorization, tenant isolation, or rate limiting; - durable distributed workflow recovery or token-level cancellation; -- PostgreSQL, MySQL, warehouse, lakehouse, or streaming-system adapters; +- MySQL, warehouse, lakehouse, or streaming-system adapters (a PostgreSQL + connector exists but is neither exported from the package API nor verified + against a live server); - provider-normalized billing or a trained-model lifecycle. Keep REST and MCP transports inside a controlled environment. Do not commit diff --git a/README.zh-CN.md b/README.zh-CN.md index 83d8a97..fd7b970 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -14,8 +14,8 @@ ![Python](https://img.shields.io/badge/Python-3.11%20%7C%203.12-3776AB?logo=python&logoColor=white) ![SQLite](https://img.shields.io/badge/SQLite-只读执行-003B57?logo=sqlite&logoColor=white) ![SQLGlot](https://img.shields.io/badge/SQL%20治理-SQLGlot-6B4FBB) -![Tests](https://img.shields.io/badge/tests-691%20passing-2EA44F) -![Semantic contracts](https://img.shields.io/badge/semantic%20checks-82%20passing-7C3AED) +![Tests](https://img.shields.io/badge/tests-928%20passing-2EA44F) +![Offline acceptance](https://img.shields.io/badge/acceptance-13%2F13-7C3AED)

@@ -26,11 +26,33 @@ QueryForge 是一个本地优先、以数据域为第一入口的 AI 数据分 用户先创建或选择数据域(如零售、金融、产品分析,或仓库内置的 Anime Streaming 示例域),再在该域内上传数据、评审语义契约和发起分析。项目把这套域级工作流与 -NL2SQL、AST 级安全策略、只读执行、多候选选择、有界修复和完整运行产物串成闭环。 +NL2SQL、AST 级安全策略、只读执行、有界修复和完整运行产物串成闭环。 > QueryForge 当前专注 SQLite 和受控环境,是面向作品展示与架构验证的参考项目, > 不是可直接公网部署的多租户分析服务。 +### 实测表现 + +治理层是确定性的,因此可以离线穷尽测试;模型层不是,所以它的数字单独汇报,并且必须 +同时给出样本量。以下两项都可从本仓库复现。 + +| 项目 | 结果 | 复现方式 | +| --- | --- | --- | +| 离线测试套件 | 928 通过,25 跳过 | `./init.sh` | +| 离线验收门禁 | 13/13 项 | `make check` | +| 确定性 Agent 基准 | 23/23 任务(`dev` + `regression` 分片;9 条 `holdout` 需显式请求) | `python scripts/benchmark_agent.py --tier 1 --gate` | +| 真实模型 NL2SQL 准确率 | **语义正确率 0.875**,执行成功率 1.0 | 40 个用例,单次运行,`deepseek-v4-flash` | + +最后一行有两点必须说清楚,它们比数字本身更重要: + +- 这是 **40 个用例的单次运行**。在**完全相同的代码**上曾观察到 ±0.03 的波动,因此这个 + 数字无法分辨小幅变化。 +- 它只覆盖 anime 示例域。它能证明整条链路在真实模型上跑得通,**不构成通用准确率承诺**。 + +Tier-1 的 23/23 衡量的是*工程链路*(治理、执行、证据、预算、失败分类),**不是**模型能力: +SQL 由测试夹具直接提供。方法与冻结基线见 +[NL2SQL 评测](docs/nl2sql_evaluation.md)。 + ## 产品界面
@@ -64,7 +86,7 @@ NL2SQL、AST 级安全策略、只读执行、多候选选择、有界修复和 | 如何保证业务口径一致 | 用 YAML 定义指标、维度、粒度和 Join Path | | 如何防止上下文串域 | 数据源、语义契约、策略和运行历史全部绑定当前数据域 | | 模型输出不完美怎么办 | 在明确预算内反思、修复和重试 | -| 复杂问题如何处理 | 启用有界 Schema 探索和并发 SQL 候选 | +| 复杂问题如何处理 | 启用有界 Schema 探索、Tool Loop 与可用的失败归因 | | 如何追踪运行过程 | 保存状态、策略决策、质量证据和交付产物 | | 如何接入其他应用 | 提供 CLI、REST/SSE、MCP、Gateway、图表和 HTML 报告 | | 原始数据如何进入分析 | 从 CSV、Parquet 和分页 JSON API 构建受治理 SQLite 数据资产 | @@ -74,10 +96,11 @@ NL2SQL、AST 级安全策略、只读执行、多候选选择、有界修复和 - **纵深防御**:SQL 在执行前接受治理,并在数据库执行边界再次校验。 - **语义契约**:YAML 模型描述业务指标、实体、关系、基数、owner、SLA、 敏感级别和质量规则。 -- **自适应工作流**:简单问题保持轻量;复杂问题可启用 Tool Loop 和并发候选选择。 +- **自适应工作流**:简单问题保持轻量;复杂问题可启用有界 Tool Loop。 - **默认只读**:普通分析以只读方式打开 SQLite,并拒绝写操作和管理类 SQL。 - **统一多端交付**:同一应用服务可通过 CLI、REST/SSE、MCP 和 Webhook Gateway 使用。 -- **可复现评测**:仓库内置离线验收流程,以及覆盖三个业务域的 120 条 NL2SQL 金标集。 +- **可复现评测**:仓库内置覆盖三套独立 Schema 的 32 条确定性 Agent 基准任务、覆盖三个 + 业务域的 120 条 NL2SQL 金标集,并在文档中记录冻结的真实模型基线。 ## 快速开始 @@ -352,7 +375,7 @@ python -m queryforge.interfaces.mcp.server --transport stdio | --- | --- | --- | | Studio | `make web-dev` | 数据域管理、可视化接入、语义编写与受治理分析 | | CLI | `queryforge --question "..."` | 本地探索和工程工作流 | -| REST | `POST /ask`、`POST /plan` | 应用集成 | +| REST | `POST /ask`(会话路径)、`POST /analyze`(规划器路径) | 应用集成 | | SSE | `POST /ask/stream` | 需要进度事件的客户端 | | MCP | `queryforge.interfaces.mcp.server` | IDE 和 MCP 兼容助手 | | Gateway | `POST /gateway/webhook` | 稳定的用户/渠道会话适配 | @@ -364,15 +387,18 @@ python -m queryforge.interfaces.mcp.server --transport stdio queryforge/ ├── cli.py # 安装后的 CLI 实现 ├── application/ # 与传输协议无关的服务门面和资源 -├── core/ # 配置、共享 Schema 和可观测性 +├── core/ # 配置、共享 Schema、工作区路径解析和可观测性 ├── data_assets/ # 数据接入、质量、血缘和发布 ├── domain/ # SQL 策略、语义层、契约和 Skills -├── infrastructure/ # SQLite、模型 Provider、存储和工具 +├── infrastructure/ # 数据库适配器、模型 Provider、存储和工具 +├── evaluation/ # 基准阈值与评测侧契约规则 ├── interfaces/ # API、MCP 和 Gateway 适配器 ├── orchestration/ # Router、角色 Agent、生命周期和状态 -└── workflow/ # NL2SQL 节点、候选选择、修复和报告 +├── workflow/ # NL2SQL 节点、候选选择、修复和报告 +└── bundled_skills/ # 随包分发的 prompt-only Skill 定义 evaluation/gold/ # 多业务域 NL2SQL 评测集 +evaluation/tasks/ # 确定性 Agent 基准任务(dev/regression/holdout) sample_data/ # 可直接运行的 SQLite 数据集和语义模型 web/ # QueryForge Studio 与托管持久化适配器 scripts/ # 构建、基准、评测和验收工具 @@ -388,17 +414,12 @@ docs/ # 架构与功能文档 执行完整离线质量门禁: ```bash -python scripts/run_acceptance.py --full -``` - -也可以直接运行测试: - -```bash -python -m unittest discover -s tests -q +./init.sh # 环境检查 + 928 个测试 + 仓库状态 +make check # 仓库卫生检查 + 13 项离线验收 ``` -在线模型评测覆盖执行成功率、语义等价率、策略 precision/recall、延迟、估算成本和 -候选选择提升: +在线模型评测覆盖执行成功率、语义等价率、策略 precision/recall、延迟、实测 token 用量和 +投影容忍度: ```bash python scripts/evaluate_sql.py \ @@ -407,28 +428,33 @@ python scripts/evaluate_sql.py \ --output .queryforge/evaluations/qwen.json ``` -CI 会在 Python 3.11 和 3.12 上执行离线验收。 +CI 会在 Python 3.11 和 3.12 上执行离线验收(含确定性 Agent Benchmark),另有一个需要 +可选传输依赖的集成任务。真实模型评测是手动触发的 workflow +(`.github/workflows/model-eval.yml`),因为它会产生实际费用。 ## 已验证的能力(以及未验证的部分) -本节每条声明都能从仓库复现;对应验收记录里同时写着通过与缺口。 +本节每条声明都能从仓库复现。这张表的价值在第三列:**没被验证的东西**和已验证的东西 +一样写清楚。 | 能力 | 怎么验证 | 状态 | | --- | --- | --- | -| 完整离线测试套件 | `make test` —— **806 个测试,0 skip** | 已验证 | +| 完整离线测试套件 | `./init.sh` —— **928 个测试,25 skip,0 失败** | 已验证 | | 仓库 + 集成门禁 | `make check`(`scripts/run_acceptance.py --full`,13/13 项通过) | 已验证 | -| 端到端 Demo(上传→发布→查询;语义校验抓错;多步分析;跨传输/拒绝/恢复) | `make demo` —— `docs/demo/` 下四个带断言的叙事脚本 | 已验证 | -| 确定性 Agent Benchmark(32 个金标任务、3 个独立 schema、消融、效果门禁) | `python scripts/benchmark_agent.py --tier 1 --gate` | 已验证(32/32) | +| 端到端 Demo(上传→发布→查询;语义校验抓错;多步分析;跨传输/拒绝/恢复) | `make demo` —— `docs/demo/` 下五个带断言的叙事脚本,离线且免 key | 已验证 | +| 确定性 Agent Benchmark(32 个金标任务、3 个独立 schema、消融、效果门禁) | `python scripts/benchmark_agent.py --tier 1 --gate` | 已验证(23/23 —— 即 `dev` + `regression` 分片;9 条 `holdout` 需用 `--split holdout` 显式请求) | | 可选依赖集成层 | `python scripts/benchmark_agent.py --tier 2 --gate` —— 依赖缺失**判定失败**而非跳过 | 已装 `.[api,mcp]` 后通过 | -| 真实模型 NL2SQL 评测 | `python scripts/evaluate_sql.py --cases evaluation/gold/nl2sql_multidomain.jsonl --model-provider

--model ` | **本机未跑**——没有数字,因此不声称准确率 | +| 真实模型 NL2SQL 评测 | `python scripts/evaluate_sql.py --cases evaluation/gold/nl2sql_multidomain.jsonl --model-provider

--model ` | **40 个 anime 用例、单次运行、`deepseek-v4-flash` 下语义正确率 0.875** —— 限制见文首说明 | +| 自动技能选择是否值得其开销 | `--skill-mode auto` 对比 `--skill-mode off` | **未确立** —— 它带来 p50 延迟 +139%、输出 token +44%,而准确率收益未被测出 | +| PostgreSQL 后端 | `pip install '.[postgres]'` 后使用 `PostgresConnector` | **已实现、未验证**:未在真实服务器上验证,且未从包 API 导出 | -Demo 全部离线、确定性(无模型调用、无网络、无需 API key)。Agent Benchmark 的 tier 1 由金标提供 SQL, -所以 32/32 衡量的是**工程链路**(治理、执行、证据、预算、失败分类),**不是模型准确率**。 -真实模型数字必须来自带凭证的 tier 3 运行,并单独报告(`.github/workflows/model-eval.yml`)。 +Demo 与 tier 1 全部离线、确定性(无模型调用、无网络、无需 API key)。Agent Benchmark 的 +tier 1 由金标提供 SQL,所以 23/23 衡量的是**工程链路**(治理、执行、证据、预算、失败分类), +**不是模型准确率**。真实模型数字必须来自带凭证的 tier 3 运行,并单独报告。 -部署等级:**受控环境、单租户、只读数据访问**。默认后端为 SQLite,另有可选的 DuckDB 适配器 -(见 [数据库适配器](docs/database_adapters.md))。系统未针对任意不可信的多租户输入做加固, -逐条记在上方对应能力的文档中;两处诚实的空白是:真实模型评测(无准确率数字)与 PostgreSQL 后端(已实现、未在真实服务器上验证)。 +部署等级:**受控环境、单租户、只读数据访问**。默认后端为 SQLite,另有可选的 DuckDB 与 +PostgreSQL 适配器(见 [数据库适配器](docs/database_adapters.md))。系统未针对任意不可信的 +多租户输入做加固;两处诚实的空白是通用领域的模型准确率与 PostgreSQL 后端。 ## 项目文档 @@ -464,7 +490,8 @@ QueryForge 的安全保证适用于其配置后的 SQLite 执行边界。项目 - 生产级认证授权、多租户隔离和限流; - 持久化分布式工作流恢复或 token 级取消; -- PostgreSQL、MySQL、数仓、湖仓和流处理系统适配器; +- MySQL、数仓、湖仓和流处理系统适配器(PostgreSQL 连接器已存在,但既未从包 API 导出, + 也未在真实服务器上验证); - 跨 Provider 统一计费,或训练模型的完整生命周期。 请将 REST 和 MCP 接口部署在受控环境中,不要提交 Provider 密钥、运行状态, diff --git a/docs/README.md b/docs/README.md index 7944a64..96f554e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -1,25 +1,35 @@ # QueryForge Documentation +Start with whatever matches what you are trying to do. + ## Start Here -- [Configuration](configuration.md) -- [Semantic layer authoring](semantic_authoring.md) -- [Data asset ingestion](data_assets.md) -- [QueryForge Studio](studio.md) -- [REST API](api_reference.md) -- [MCP server](mcp_server.md) +- [Configuration](configuration.md) — environment variables, providers, paths +- [QueryForge Studio](studio.md) — the web UI: connect a dataset, review semantics, analyse +- [REST API](api_reference.md) — HTTP endpoints +- [MCP server](mcp_server.md) — expose QueryForge to an MCP client as tools + +## Semantic Layer + +- [Semantic model authoring](semantic_authoring.md) — describe metrics, dimensions and joins +- [Semantic contracts](semantic_contracts.md) — the rules a semantic model must satisfy +- [Subject tree](subject_tree.md) — domain organisation + +## Architecture and Governance + +- [Agent team architecture](agent_team_architecture.md) — entry routing, orchestration, workflow +- [Database adapters](database_adapters.md) — SQLite default; DuckDB and PostgreSQL optional + +## Data + +- [Data asset ingestion](data_assets.md) — bringing your own datasets in +- [Report artifacts](report_artifact.md) — what a run produces and where it is written -## Architecture +## Evaluation -- [Agent team architecture](agent_team_architecture.md) -- [Subject tree](subject_tree.md) -- [Semantic contracts](semantic_contracts.md) +- [NL2SQL evaluation](nl2sql_evaluation.md) — gold sets, metrics, tiers, how to reproduce a run +- [End-to-end acceptance demos](demo/README.md) — runnable scenarios, no API key required -## Delivery and Evaluation +## Contributing - [GitHub release checklist](github_release.md) -- [Report artifacts](report_artifact.md) -- [NL2SQL evaluation](nl2sql_evaluation.md) -- [Agent benchmark, ablation and effect gates](database_adapters.md) -- [End-to-end acceptance demos](demo/README.md) -- [Database adapters (SQLite default, DuckDB optional)](database_adapters.md) diff --git a/docs/agent_team_architecture.md b/docs/agent_team_architecture.md index 5a24785..4444a97 100644 --- a/docs/agent_team_architecture.md +++ b/docs/agent_team_architecture.md @@ -22,9 +22,21 @@ analysis -> candidate -> execution -> completion -> delivery DatabaseTool + SQLGlot AST policy ``` -Every public transport calls `AgentService`. The router classifies the request; the -orchestrator manages stages, artifacts, state, and delivery; `WorkflowRunner` remains -the only SQL generation and execution kernel. +Every public transport reaches one of **two** application services: + +- `AgentService` for `/ask`, `/ask/stream`, `/plan`, MCP and the Gateway webhook. The + router classifies the request; the orchestrator manages stages, artifacts, state and + delivery; `WorkflowRunner` generates and executes the single governed query. +- `AnalysisPlannerService` for `/analyze` and `queryforge --analyze`. It builds a + `AnalysisPlan` from governed metrics, validates it with `PlanValidator`, executes it + through `AnalysisExecutor` on the same tool registry, and composes an evidence-backed + answer. + +`WorkflowRunner` is therefore the SQL kernel for the conversational path, **not** the +only execution kernel: the planner path reaches `DatabaseTool` through +`ToolRegistry`. Both paths share `DatabaseTool` and the SQLGlot AST policy as their +execution boundary and their safety invariants; the orchestration above them differs. +Unifying that upper layer is tracked work, not a current property. ## Five Stages diff --git a/docs/demo/run_demo_d.py b/docs/demo/run_demo_d.py index c1753e7..50d102e 100644 --- a/docs/demo/run_demo_d.py +++ b/docs/demo/run_demo_d.py @@ -324,7 +324,18 @@ def _resume_after_crash(root: Path) -> dict: "terminal": resumed.get("terminal_outcome"), "reused": sorted(resumed.get("reused_steps") or []), "recomputed": sorted(resumed.get("recomputed_steps") or []), - "tool_calls": int((resumed.get("budgets") or {}).get("usage", {}).get("max_tool_calls") or 0), + # Net of what this attempt inherited: budget usage is cumulative across + # attempts of one run (E-04), so the raw figure counts the original + # attempt's calls too. The demo's claim is about re-querying on resume, + # which is the delta. + "tool_calls": max( + 0, + int((resumed.get("budgets") or {}).get("usage", {}).get("max_tool_calls") or 0) + - int((resumed.get("inherited_usage") or {}).get("max_tool_calls") or 0), + ), + "inherited_tool_calls": int( + (resumed.get("inherited_usage") or {}).get("max_tool_calls") or 0 + ), "first_terminal": first.get("terminal_outcome"), } diff --git a/docs/nl2sql_evaluation.md b/docs/nl2sql_evaluation.md index 30f1063..5ec2b0c 100644 --- a/docs/nl2sql_evaluation.md +++ b/docs/nl2sql_evaluation.md @@ -116,10 +116,56 @@ python scripts/evaluate_sql.py \ - `--min-policy-recall` (default 0.0, disabled): fails when any probe bypasses the policy engine. -Token and cost values are explicitly heuristic (`character_count / 4`) because the -provider adapters do not expose normalized billing usage across all configured model -vendors. Configure per-million prices only for comparable estimates; do not treat -them as invoices. +### Token counts are measured, not estimated + +When the provider returns usage, the report records the real counts and sets +`token_source: "measured"`. Only when a provider exposes no usage at all does the +report fall back to the `character_count / 4` heuristic and set +`token_source: "estimated"`. Read the field before quoting a number: an `estimated` +figure cannot support a cost conclusion. + +Per-million prices are configured for comparable estimates only. Treat cost as an +indicator of relative expense, never as an invoice. + +## Frozen Baselines + +`evaluation/reports/` is gitignored, so raw evaluator output has no versioned carrier. +The numbers below are the tracked record; each row names the report file it came from. +**A baseline is only current if its versions are unchanged** — change the model, the gold +set, the semantic model, the SQL policy, or the comparison rules, and these become +historical. + +Versions: provider `deepseek`, model `deepseek-v4-flash`, gold set +`evaluation/gold/nl2sql_multidomain.jsonl` (first 40 = `anime_content`), semantic model +`sample_data/anime_streaming/semantic_model.yml`. + +| Run (report file) | `semantic_correctness_rate` | `sql_execution_success_rate` | `p50_latency_ms` | In / out tokens | +|---|---|---|---|---| +| `nl2sql_skill_auto.json` — automatic skill selection | 0.84375 | 1.0 | 12260 | 1742 / 633 | +| `nl2sql_model_eval.json` — skills inactive | 0.875 | 1.0 | 5138 | 1446 / 438 | +| `nl2sql_selfverify2.json` — after reasoning-payload and prompt fixes | 0.875 | 1.0 | 11728 | 1734 / 579 | +| `nl2sql_defects_fixed2.json` — after the governance defect pass | 0.875 | 1.0 | 13906 | 1754 / 866 | + +> **These are single runs of 40 cases.** Differences of ±0.03 in +> `semantic_correctness_rate` have been observed across *identical* code, so the three +> 0.875 rows are the same result, not three improvements. Do not present a delta of that +> size as a gain; raise `--repeat` until the interval separates. + +Two findings worth carrying forward, both from controlled comparisons rather than +aggregate impressions: + +- **Automatic skill selection costs p50 +139% and output tokens +44%** (12260 ms vs 5138 ms; + 633 vs 438) with no demonstrated accuracy benefit. `--skill-mode auto|off` exists to test + this; it is unresolved, not settled. +- **A second SQL candidate does not improve accuracy and costs ~20% p50 latency.** Measured + over 20 paired cases with candidate count forced, not grouped by the gold set's + `candidate_selection` flag (which correlates perfectly with task category and therefore + measures difficulty). Parallel candidates remain an explicit opt-in. + +A real-model run spends money. Confirm the account balance first: a depleted balance returns +HTTP 402, which the evaluator classifies as `environment_error` and excludes from the +accuracy denominators. + ## Maintain the Gold Set diff --git a/evaluation/gold/context_and_compound.jsonl b/evaluation/gold/context_and_compound.jsonl new file mode 100644 index 0000000..3e95357 --- /dev/null +++ b/evaluation/gold/context_and_compound.jsonl @@ -0,0 +1,21 @@ +{"id": "ctx_01", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Break that down by playback region.", "expected_sql": "SELECT w.playback_region, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w GROUP BY w.playback_region ORDER BY watch_hours DESC", "follow_up_context": ["Show total watch hours."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "Metric only in the prior turn. playback_region is an allowed dimension."} +{"id": "ctx_02", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "And by anime release year?", "expected_sql": "SELECT a.release_year, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON e.episode_id = w.episode_id JOIN dim_anime a ON a.anime_id = e.anime_id GROUP BY a.release_year ORDER BY a.release_year", "follow_up_context": ["Show total watch hours."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "Elliptical: 'And by ...' carries no metric. release_year is allowed."} +{"id": "ctx_03", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Only APAC.", "expected_sql": "SELECT SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w WHERE w.playback_region = 'APAC'", "follow_up_context": ["Show total watch hours by playback region."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "Neither metric nor aggregation named; the referent is the prior request."} +{"id": "ctx_04", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Only Major studios.", "expected_sql": "SELECT SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON e.episode_id = w.episode_id JOIN dim_anime a ON a.anime_id = e.anime_id JOIN dim_studio s ON s.studio_id = a.studio_id WHERE s.studio_tier = 'Major'", "follow_up_context": ["Show total watch hours by studio."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "Filter-only follow-up on studio.tier, which is allowed for watch_hours."} +{"id": "ctx_05", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Top 5 only.", "expected_sql": "SELECT a.title, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON e.episode_id = w.episode_id JOIN dim_anime a ON a.anime_id = e.anime_id GROUP BY a.title ORDER BY watch_hours DESC LIMIT 5", "follow_up_context": ["Show watch hours by anime title."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "A bare ranking follow-up: neither metric nor entity is restated."} +{"id": "ctx_06", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Why is that so high?", "expected_sql": "SELECT SUM(w.watch_seconds) / 3600.0 AS apac_watch_hours FROM fact_watch_session w WHERE w.playback_region = 'APAC'", "follow_up_context": ["Show total watch hours by playback region.", "APAC is the highest region."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": true, "notes": "'that' resolves only against the previous turns; no metric is named."} +{"id": "ctx_07", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Same thing but for Europe.", "expected_sql": "SELECT SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w WHERE w.playback_region = 'Europe'", "follow_up_context": ["Show total watch hours by playback region. APAC is highest."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "'Same thing' requires the prior metric AND aggregation."} +{"id": "ctx_08", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Compare that with the other release cohort.", "expected_sql": "SELECT a.release_year, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON e.episode_id = w.episode_id JOIN dim_anime a ON a.anime_id = e.anime_id WHERE a.release_year >= 2012 GROUP BY a.release_year ORDER BY a.release_year", "follow_up_context": ["Show watch hours for anime released in 2019."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "Comparison target and metric both come from context. Years present in the sample data are 1998/2005/2012/2019."} +{"id": "ctx_09", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Break that down by product category.", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS net_revenue FROM fact_merch_order_item i JOIN dim_merch_product p ON p.product_id = i.product_id GROUP BY p.product_category ORDER BY net_revenue DESC", "follow_up_context": ["Show total merchandise net revenue."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "Second domain, to avoid tuning to one shape. merch_product.category is allowed."} +{"id": "ctx_10", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Only rewatched sessions.", "expected_sql": "SELECT SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w WHERE w.rewatch_flag = 1", "follow_up_context": ["Show total watch hours."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": ""} +{"id": "ctx_11", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Now average rating instead.", "expected_sql": "SELECT CAST(SUM(r.score) AS REAL) / NULLIF(COUNT(*), 0) AS average_rating FROM fact_rating r", "follow_up_context": ["Show the total number of ratings."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "Metric replacement on the same entity, which the question leaves implicit."} +{"id": "ctx_12", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "context_dependent", "expected_outcome": "query", "question": "Just that one.", "expected_sql": "SELECT a.title, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON e.episode_id = w.episode_id JOIN dim_anime a ON a.anime_id = e.anime_id GROUP BY a.title ORDER BY watch_hours DESC LIMIT 1", "follow_up_context": ["Show watch hours by anime title.", "The top title is the one we want to drill into."], "candidate_selection": false, "requires_context": true, "compound": false, "causal": false, "notes": "Referential restriction with no metric, no dimension and no entity restated."} +{"id": "cmp_01", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "compound", "expected_outcome": "query", "question": "How many anime are there in total, and how many ratings do they have?", "expected_sql": "SELECT (SELECT COUNT(*) FROM dim_anime) AS anime_count, (SELECT COUNT(*) FROM fact_rating) AS rating_count", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": true, "causal": false, "notes": "Two aggregates in one question; answering only the first is incomplete."} +{"id": "cmp_02", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "compound", "expected_outcome": "query", "question": "Show total watch hours and the number of distinct viewers.", "expected_sql": "SELECT SUM(w.watch_seconds) / 3600.0 AS watch_hours, COUNT(DISTINCT w.user_id) AS unique_viewers FROM fact_watch_session w", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": true, "causal": false, "notes": "Volume plus reach."} +{"id": "cmp_03", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "compound", "expected_outcome": "query", "question": "Compare average rating and average review length by anime format, and tell me which format does best.", "expected_sql": "SELECT a.content_format, CAST(SUM(r.score) AS REAL) / NULLIF(COUNT(*), 0) AS average_rating, AVG(r.review_length) AS average_review_length FROM fact_rating r JOIN dim_anime a ON a.anime_id = r.anime_id GROUP BY a.content_format ORDER BY average_rating DESC", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": true, "causal": false, "notes": "The measurement half is the oracle; 'which does best' is a judgement the answer must state rather than silently omit."} +{"id": "cmp_04", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "compound", "expected_outcome": "query", "question": "What is merchandise net revenue by product category, and which category should we invest in next quarter?", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS net_revenue FROM fact_merch_order_item i JOIN dim_merch_product p ON p.product_id = i.product_id GROUP BY p.product_category ORDER BY net_revenue DESC", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": true, "causal": false, "notes": "Measurement plus a recommendation that the data does not settle."} +{"id": "cmp_05", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "compound", "expected_outcome": "query", "question": "List the top 5 anime by watch hours, and give me the session counts too.", "expected_sql": "SELECT a.title, SUM(w.watch_seconds) / 3600.0 AS watch_hours, COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_episode e ON e.episode_id = w.episode_id JOIN dim_anime a ON a.anime_id = e.anime_id GROUP BY a.title ORDER BY watch_hours DESC LIMIT 5", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": true, "causal": false, "notes": "Ranking plus a second measure; dropping the counts is a partial answer."} +{"id": "cmp_06", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "compound", "expected_outcome": "query", "question": "How many users signed up in 2024, and how many of them activated a subscription?", "expected_sql": "SELECT (SELECT COUNT(*) FROM dim_user u WHERE u.signup_date_key >= 20240101 AND u.signup_date_key <= 20241231) AS signups_2024, (SELECT COUNT(DISTINCT s.user_id) FROM fact_subscription s JOIN dim_user u ON u.user_id = s.user_id WHERE u.signup_date_key >= 20240101 AND u.signup_date_key <= 20241231) AS activated_subscribers", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": true, "causal": false, "notes": "Funnel: two dependent counts in one ask."} +{"id": "cau_01", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "causal", "expected_outcome": "query", "question": "Why did watch hours drop last month?", "expected_sql": "SELECT d.year, d.month_number, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_date d ON d.date_key = w.watch_date_key GROUP BY d.year, d.month_number ORDER BY d.year, d.month_number", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": false, "causal": true, "notes": "No cause exists in the data. A defensible answer states the observed trend and either asks what changed or lists candidate explanations as hypotheses with evidence."} +{"id": "cau_02", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "causal", "expected_outcome": "query", "question": "What is driving the growth in merchandise revenue?", "expected_sql": "SELECT d.year, d.month_number, SUM(i.net_amount_usd) AS net_revenue FROM fact_merch_order_item i JOIN fact_merch_order o ON o.order_id = i.order_id JOIN dim_date d ON d.date_key = o.order_date_key GROUP BY d.year, d.month_number ORDER BY d.year, d.month_number", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": false, "causal": true, "notes": "Same shape: measures the trend, must not invent a driver."} +{"id": "cau_03", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "domain": "anime_streaming", "category": "causal", "expected_outcome": "query", "question": "Users who watch more rate higher, right?", "expected_sql": "SELECT CAST(SUM(r.score) AS REAL) / NULLIF(COUNT(*), 0) AS overall_average_rating FROM fact_rating r", "follow_up_context": [], "candidate_selection": false, "requires_context": false, "compound": false, "causal": true, "notes": "Leading question with a false presupposition and no comparison in the data. A defensible answer declines to confirm it and states what would be needed to test it."} diff --git a/evaluation/gold/nl2sql_multidomain.jsonl b/evaluation/gold/nl2sql_multidomain.jsonl index a194e52..23e894c 100644 --- a/evaluation/gold/nl2sql_multidomain.jsonl +++ b/evaluation/gold/nl2sql_multidomain.jsonl @@ -1,120 +1,120 @@ -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 1", "follow_up_context": [], "id": "anime_content_01", "question": "List the first 1 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 2", "follow_up_context": [], "id": "anime_content_02", "question": "List the first 2 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 3", "follow_up_context": [], "id": "anime_content_03", "question": "List the first 3 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 4", "follow_up_context": [], "id": "anime_content_04", "question": "List the first 4 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 5", "follow_up_context": [], "id": "anime_content_05", "question": "List the first 5 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 6", "follow_up_context": [], "id": "anime_content_06", "question": "List the first 6 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 7", "follow_up_context": [], "id": "anime_content_07", "question": "List the first 7 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 8", "follow_up_context": [], "id": "anime_content_08", "question": "List the first 8 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 9", "follow_up_context": [], "id": "anime_content_09", "question": "List the first 9 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 10", "follow_up_context": [], "id": "anime_content_10", "question": "List the first 10 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Major' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_11", "question": "Show anime titles produced by Major studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Growth' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_12", "question": "Show anime titles produced by Growth studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Indie' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_13", "question": "Show anime titles produced by Indie studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Major' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_14", "question": "Show anime titles produced by Major studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Growth' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_15", "question": "Show anime titles produced by Growth studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Indie' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_16", "question": "Show anime titles produced by Indie studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Major' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_17", "question": "Show anime titles produced by Major studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Growth' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_18", "question": "Show anime titles produced by Growth studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2020", "follow_up_context": [], "id": "anime_content_19", "question": "Count anime released in 2020 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2021", "follow_up_context": [], "id": "anime_content_20", "question": "Count anime released in 2021 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2022", "follow_up_context": [], "id": "anime_content_21", "question": "Count anime released in 2022 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2023", "follow_up_context": [], "id": "anime_content_22", "question": "Count anime released in 2023 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2024", "follow_up_context": [], "id": "anime_content_23", "question": "Count anime released in 2024 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2025", "follow_up_context": [], "id": "anime_content_24", "question": "Count anime released in 2025 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, AVG(r.score) AS average_rating FROM fact_rating r JOIN dim_anime a ON r.anime_id = a.anime_id GROUP BY a.content_format", "follow_up_context": [], "id": "anime_content_25", "question": "Calculate average audience rating by anime format.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT s.country, COUNT(DISTINCT a.anime_id) AS anime_count FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id GROUP BY s.country", "follow_up_context": [], "id": "anime_content_26", "question": "Calculate anime count by studio country.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, COUNT(e.episode_id) AS episode_count FROM dim_episode e JOIN dim_anime a ON e.anime_id = a.anime_id GROUP BY a.content_format", "follow_up_context": [], "id": "anime_content_27", "question": "Calculate episode count by anime format.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT source_material, AVG(production_budget_usd) AS average_budget FROM dim_anime GROUP BY source_material", "follow_up_context": [], "id": "anime_content_28", "question": "Calculate average production budget by source material.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT b.anime_id) AS anime_count FROM bridge_anime_genre b JOIN dim_genre g ON b.genre_id = g.genre_id WHERE g.genre_name = 'Action'", "follow_up_context": ["Show anime counts by genre."], "id": "anime_content_29", "question": "Now count anime assigned to the Action genre.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT b.anime_id) AS anime_count FROM bridge_anime_genre b JOIN dim_genre g ON b.genre_id = g.genre_id WHERE g.genre_name = 'Romance'", "follow_up_context": ["Show anime counts by genre."], "id": "anime_content_30", "question": "Now count anime assigned to the Romance genre.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT b.anime_id) AS anime_count FROM bridge_anime_genre b JOIN dim_genre g ON b.genre_id = g.genre_id WHERE g.genre_name = 'Sci-Fi'", "follow_up_context": ["Show anime counts by genre."], "id": "anime_content_31", "question": "Now count anime assigned to the Sci-Fi genre.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT b.anime_id) AS anime_count FROM bridge_anime_genre b JOIN dim_genre g ON b.genre_id = g.genre_id WHERE g.genre_name = 'Slice of Life'", "follow_up_context": ["Show anime counts by genre."], "id": "anime_content_32", "question": "Now count anime assigned to the Slice of Life genre.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_33", "policy_probe_sql": "DELETE FROM dim_anime", "question": "Delete all rows", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_34", "policy_probe_sql": "DROP TABLE dim_anime", "question": "Drop the table", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_35", "policy_probe_sql": "INSERT INTO dim_anime DEFAULT VALUES", "question": "Insert a test row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_36", "policy_probe_sql": "UPDATE dim_anime SET 1 = 1", "question": "Update every row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_37", "policy_probe_sql": "ATTACH DATABASE '/tmp/other.sqlite' AS other", "question": "Attach another database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_38", "policy_probe_sql": "WITH RECURSIVE t(n) AS (SELECT 1 UNION ALL SELECT n + 1 FROM t) SELECT * FROM t", "question": "Run a recursive query", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_39", "policy_probe_sql": "SELECT 1; DELETE FROM dim_anime", "question": "Run two statements", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_40", "policy_probe_sql": "VACUUM", "question": "Vacuum the database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT watch_session_id FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_01", "question": "Show the first five watch-session watch_session_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT user_id FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_02", "question": "Show the first five watch-session user_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT episode_id FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_03", "question": "Show the first five watch-session episode_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT watch_date_key FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_04", "question": "Show the first five watch-session watch_date_key values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT device_type FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_05", "question": "Show the first five watch-session device_type values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_06", "question": "Show the first five watch-session playback_region values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT watch_seconds FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_07", "question": "Show the first five watch-session watch_seconds values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT completion_pct FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_08", "question": "Show the first five watch-session completion_pct values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT completed_flag FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_09", "question": "Show the first five watch-session completed_flag values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT rewatch_flag FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_10", "question": "Show the first five watch-session rewatch_flag values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'Series' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_11", "question": "Show watch hours for Series anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'Movie' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_12", "question": "Show watch hours for Movie anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'OVA' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_13", "question": "Show watch hours for OVA anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'ONA' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_14", "question": "Show watch hours for ONA anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'Series' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_15", "question": "Show watch hours for Series anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'Movie' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_16", "question": "Show watch hours for Movie anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'OVA' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_17", "question": "Show watch hours for OVA anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'ONA' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_18", "question": "Show watch hours for ONA anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 1", "follow_up_context": [], "id": "viewer_engagement_19", "question": "Count watch sessions in calendar month 1.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 2", "follow_up_context": [], "id": "viewer_engagement_20", "question": "Count watch sessions in calendar month 2.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 3", "follow_up_context": [], "id": "viewer_engagement_21", "question": "Count watch sessions in calendar month 3.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 4", "follow_up_context": [], "id": "viewer_engagement_22", "question": "Count watch sessions in calendar month 4.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 5", "follow_up_context": [], "id": "viewer_engagement_23", "question": "Count watch sessions in calendar month 5.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 6", "follow_up_context": [], "id": "viewer_engagement_24", "question": "Count watch sessions in calendar month 6.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT SUM(watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session", "follow_up_context": [], "id": "viewer_engagement_25", "question": "Calculate engagement metric 1.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session", "follow_up_context": [], "id": "viewer_engagement_26", "question": "Calculate engagement metric 2.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT CAST(SUM(completed_flag) AS REAL) / COUNT(*) AS completion_rate FROM fact_watch_session", "follow_up_context": [], "id": "viewer_engagement_27", "question": "Calculate engagement metric 3.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT AVG(score) AS average_rating FROM fact_rating", "follow_up_context": [], "id": "viewer_engagement_28", "question": "Calculate engagement metric 4.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region, COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session WHERE playback_region = 'APAC' GROUP BY playback_region", "follow_up_context": ["Show unique viewers by playback region."], "id": "viewer_engagement_29", "question": "Now show unique viewers in APAC.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region, COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session WHERE playback_region = 'Europe' GROUP BY playback_region", "follow_up_context": ["Show unique viewers by playback region."], "id": "viewer_engagement_30", "question": "Now show unique viewers in Europe.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region, COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session WHERE playback_region = 'North America' GROUP BY playback_region", "follow_up_context": ["Show unique viewers by playback region."], "id": "viewer_engagement_31", "question": "Now show unique viewers in North America.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region, COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session WHERE playback_region = 'Latin America' GROUP BY playback_region", "follow_up_context": ["Show unique viewers by playback region."], "id": "viewer_engagement_32", "question": "Now show unique viewers in Latin America.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_33", "policy_probe_sql": "DELETE FROM fact_watch_session", "question": "Delete all rows", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_34", "policy_probe_sql": "DROP TABLE fact_watch_session", "question": "Drop the table", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_35", "policy_probe_sql": "INSERT INTO fact_watch_session DEFAULT VALUES", "question": "Insert a test row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_36", "policy_probe_sql": "UPDATE fact_watch_session SET 1 = 1", "question": "Update every row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_37", "policy_probe_sql": "ATTACH DATABASE '/tmp/other.sqlite' AS other", "question": "Attach another database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_38", "policy_probe_sql": "WITH RECURSIVE t(n) AS (SELECT 1 UNION ALL SELECT n + 1 FROM t) SELECT * FROM t", "question": "Run a recursive query", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_39", "policy_probe_sql": "SELECT 1; DELETE FROM fact_watch_session", "question": "Run two statements", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_40", "policy_probe_sql": "VACUUM", "question": "Vacuum the database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT subscription_id FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_01", "question": "Show the first five subscription subscription_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT user_id FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_02", "question": "Show the first five subscription user_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT plan_name FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_03", "question": "Show the first five subscription plan_name values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT billing_cycle FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_04", "question": "Show the first five subscription billing_cycle values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT status FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_05", "question": "Show the first five subscription status values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT monthly_price_usd FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_06", "question": "Show the first five subscription monthly_price_usd values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT discount_usd FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_07", "question": "Show the first five subscription discount_usd values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT recognized_revenue_usd FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_08", "question": "Show the first five subscription recognized_revenue_usd values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT start_date_key FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_09", "question": "Show the first five subscription start_date_key values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT end_date_key FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_10", "question": "Show the first five subscription end_date_key values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Fan' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_11", "question": "Show subscription revenue for the Fan plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Premium' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_12", "question": "Show subscription revenue for the Premium plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Family' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_13", "question": "Show subscription revenue for the Family plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Fan' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_14", "question": "Show subscription revenue for the Fan plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Premium' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_15", "question": "Show subscription revenue for the Premium plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Family' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_16", "question": "Show subscription revenue for the Family plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Fan' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_17", "question": "Show subscription revenue for the Fan plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Premium' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_18", "question": "Show subscription revenue for the Premium plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 1", "follow_up_context": [], "id": "platform_monetization_19", "question": "Show ad revenue in calendar month 1.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 2", "follow_up_context": [], "id": "platform_monetization_20", "question": "Show ad revenue in calendar month 2.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 3", "follow_up_context": [], "id": "platform_monetization_21", "question": "Show ad revenue in calendar month 3.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 4", "follow_up_context": [], "id": "platform_monetization_22", "question": "Show ad revenue in calendar month 4.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 5", "follow_up_context": [], "id": "platform_monetization_23", "question": "Show ad revenue in calendar month 5.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 6", "follow_up_context": [], "id": "platform_monetization_24", "question": "Show ad revenue in calendar month 6.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(recognized_revenue_usd) AS subscription_revenue FROM fact_subscription", "follow_up_context": [], "id": "platform_monetization_25", "question": "Calculate monetization metric 1.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(revenue_usd) AS ad_revenue FROM fact_ad_impression", "follow_up_context": [], "id": "platform_monetization_26", "question": "Calculate monetization metric 2.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(net_amount_usd) AS merch_gmv FROM fact_merch_order_item", "follow_up_context": [], "id": "platform_monetization_27", "question": "Calculate monetization metric 3.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT CAST(SUM(clicked_flag) AS REAL) / COUNT(*) AS ad_ctr FROM fact_ad_impression", "follow_up_context": [], "id": "platform_monetization_28", "question": "Calculate monetization metric 4.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS merch_gmv FROM fact_merch_order_item i JOIN dim_merch_product p ON i.product_id = p.product_id WHERE p.product_category = 'Figure' GROUP BY p.product_category", "follow_up_context": ["Show merchandise GMV by product category."], "id": "platform_monetization_29", "question": "Now show merchandise GMV for Figure.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS merch_gmv FROM fact_merch_order_item i JOIN dim_merch_product p ON i.product_id = p.product_id WHERE p.product_category = 'Apparel' GROUP BY p.product_category", "follow_up_context": ["Show merchandise GMV by product category."], "id": "platform_monetization_30", "question": "Now show merchandise GMV for Apparel.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS merch_gmv FROM fact_merch_order_item i JOIN dim_merch_product p ON i.product_id = p.product_id WHERE p.product_category = 'Poster' GROUP BY p.product_category", "follow_up_context": ["Show merchandise GMV by product category."], "id": "platform_monetization_31", "question": "Now show merchandise GMV for Poster.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS merch_gmv FROM fact_merch_order_item i JOIN dim_merch_product p ON i.product_id = p.product_id WHERE p.product_category = 'Blu-ray' GROUP BY p.product_category", "follow_up_context": ["Show merchandise GMV by product category."], "id": "platform_monetization_32", "question": "Now show merchandise GMV for Blu-ray.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_33", "policy_probe_sql": "DELETE FROM fact_subscription", "question": "Delete all rows", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_34", "policy_probe_sql": "DROP TABLE fact_subscription", "question": "Drop the table", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_35", "policy_probe_sql": "INSERT INTO fact_subscription DEFAULT VALUES", "question": "Insert a test row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_36", "policy_probe_sql": "UPDATE fact_subscription SET 1 = 1", "question": "Update every row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_37", "policy_probe_sql": "ATTACH DATABASE '/tmp/other.sqlite' AS other", "question": "Attach another database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_38", "policy_probe_sql": "WITH RECURSIVE t(n) AS (SELECT 1 UNION ALL SELECT n + 1 FROM t) SELECT * FROM t", "question": "Run a recursive query", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_39", "policy_probe_sql": "SELECT 1; DELETE FROM fact_subscription", "question": "Run two statements", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} -{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_40", "policy_probe_sql": "VACUUM", "question": "Vacuum the database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml"} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 1", "follow_up_context": [], "id": "anime_content_01", "question": "List the first 1 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 2", "follow_up_context": [], "id": "anime_content_02", "question": "List the first 2 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 3", "follow_up_context": [], "id": "anime_content_03", "question": "List the first 3 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 4", "follow_up_context": [], "id": "anime_content_04", "question": "List the first 4 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 5", "follow_up_context": [], "id": "anime_content_05", "question": "List the first 5 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 6", "follow_up_context": [], "id": "anime_content_06", "question": "List the first 6 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 7", "follow_up_context": [], "id": "anime_content_07", "question": "List the first 7 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 8", "follow_up_context": [], "id": "anime_content_08", "question": "List the first 8 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 9", "follow_up_context": [], "id": "anime_content_09", "question": "List the first 9 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT title FROM dim_anime ORDER BY title LIMIT 10", "follow_up_context": [], "id": "anime_content_10", "question": "List the first 10 anime titles alphabetically.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Major' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_11", "question": "Show anime titles produced by Major studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Growth' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_12", "question": "Show anime titles produced by Growth studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Indie' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_13", "question": "Show anime titles produced by Indie studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Major' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_14", "question": "Show anime titles produced by Major studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Growth' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_15", "question": "Show anime titles produced by Growth studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Indie' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_16", "question": "Show anime titles produced by Indie studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Major' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_17", "question": "Show anime titles produced by Major studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.title, s.studio_name, s.studio_tier FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id WHERE s.studio_tier = 'Growth' ORDER BY a.title", "follow_up_context": [], "id": "anime_content_18", "question": "Show anime titles produced by Growth studios.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2020", "follow_up_context": [], "id": "anime_content_19", "question": "Count anime released in 2020 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2021", "follow_up_context": [], "id": "anime_content_20", "question": "Count anime released in 2021 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2022", "follow_up_context": [], "id": "anime_content_21", "question": "Count anime released in 2022 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2023", "follow_up_context": [], "id": "anime_content_22", "question": "Count anime released in 2023 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2024", "follow_up_context": [], "id": "anime_content_23", "question": "Count anime released in 2024 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS anime_count FROM dim_anime WHERE release_year >= 2025", "follow_up_context": [], "id": "anime_content_24", "question": "Count anime released in 2025 or later.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, AVG(r.score) AS average_rating FROM fact_rating r JOIN dim_anime a ON r.anime_id = a.anime_id GROUP BY a.content_format", "follow_up_context": [], "id": "anime_content_25", "question": "Calculate average audience rating by anime format.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT s.country, COUNT(DISTINCT a.anime_id) AS anime_count FROM dim_anime a JOIN dim_studio s ON a.studio_id = s.studio_id GROUP BY s.country", "follow_up_context": [], "id": "anime_content_26", "question": "Calculate anime count by studio country.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, COUNT(e.episode_id) AS episode_count FROM dim_episode e JOIN dim_anime a ON e.anime_id = a.anime_id GROUP BY a.content_format", "follow_up_context": [], "id": "anime_content_27", "question": "Calculate episode count by anime format.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT source_material, AVG(production_budget_usd) AS average_budget FROM dim_anime GROUP BY source_material", "follow_up_context": [], "id": "anime_content_28", "question": "Calculate average production budget by source material.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT b.anime_id) AS anime_count FROM bridge_anime_genre b JOIN dim_genre g ON b.genre_id = g.genre_id WHERE g.genre_name = 'Action'", "follow_up_context": ["Show anime counts by genre."], "id": "anime_content_29", "question": "Now count anime assigned to the Action genre.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT b.anime_id) AS anime_count FROM bridge_anime_genre b JOIN dim_genre g ON b.genre_id = g.genre_id WHERE g.genre_name = 'Romance'", "follow_up_context": ["Show anime counts by genre."], "id": "anime_content_30", "question": "Now count anime assigned to the Romance genre.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT b.anime_id) AS anime_count FROM bridge_anime_genre b JOIN dim_genre g ON b.genre_id = g.genre_id WHERE g.genre_name = 'Sci-Fi'", "follow_up_context": ["Show anime counts by genre."], "id": "anime_content_31", "question": "Now count anime assigned to the Sci-Fi genre.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT b.anime_id) AS anime_count FROM bridge_anime_genre b JOIN dim_genre g ON b.genre_id = g.genre_id WHERE g.genre_name = 'Slice of Life'", "follow_up_context": ["Show anime counts by genre."], "id": "anime_content_32", "question": "Now count anime assigned to the Slice of Life genre.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_33", "policy_probe_sql": "DELETE FROM dim_anime", "question": "Delete all rows", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_34", "policy_probe_sql": "DROP TABLE dim_anime", "question": "Drop the table", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_35", "policy_probe_sql": "INSERT INTO dim_anime DEFAULT VALUES", "question": "Insert a test row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_36", "policy_probe_sql": "UPDATE dim_anime SET 1 = 1", "question": "Update every row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_37", "policy_probe_sql": "ATTACH DATABASE '/tmp/other.sqlite' AS other", "question": "Attach another database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_38", "policy_probe_sql": "WITH RECURSIVE t(n) AS (SELECT 1 UNION ALL SELECT n + 1 FROM t) SELECT * FROM t", "question": "Run a recursive query", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_39", "policy_probe_sql": "SELECT 1; DELETE FROM dim_anime", "question": "Run two statements", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "anime_content", "expected_outcome": "policy_rejection", "id": "anime_content_40", "policy_probe_sql": "VACUUM", "question": "Vacuum the database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT watch_session_id FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_01", "question": "Show the first five watch-session watch_session_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT user_id FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_02", "question": "Show the first five watch-session user_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT episode_id FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_03", "question": "Show the first five watch-session episode_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT watch_date_key FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_04", "question": "Show the first five watch-session watch_date_key values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT device_type FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_05", "question": "Show the first five watch-session device_type values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_06", "question": "Show the first five watch-session playback_region values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT watch_seconds FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_07", "question": "Show the first five watch-session watch_seconds values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT completion_pct FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_08", "question": "Show the first five watch-session completion_pct values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT completed_flag FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_09", "question": "Show the first five watch-session completed_flag values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT rewatch_flag FROM fact_watch_session ORDER BY watch_session_id LIMIT 5", "follow_up_context": [], "id": "viewer_engagement_10", "question": "Show the first five watch-session rewatch_flag values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'Series' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_11", "question": "Show watch hours for Series anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'Movie' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_12", "question": "Show watch hours for Movie anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'OVA' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_13", "question": "Show watch hours for OVA anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'ONA' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_14", "question": "Show watch hours for ONA anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'Series' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_15", "question": "Show watch hours for Series anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'Movie' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_16", "question": "Show watch hours for Movie anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'OVA' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_17", "question": "Show watch hours for OVA anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT a.content_format, SUM(w.watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session w JOIN dim_episode e ON w.episode_id = e.episode_id JOIN dim_anime a ON e.anime_id = a.anime_id WHERE a.content_format = 'ONA' GROUP BY a.content_format", "follow_up_context": [], "id": "viewer_engagement_18", "question": "Show watch hours for ONA anime.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 1", "follow_up_context": [], "id": "viewer_engagement_19", "question": "Count watch sessions in calendar month 1.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 2", "follow_up_context": [], "id": "viewer_engagement_20", "question": "Count watch sessions in calendar month 2.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 3", "follow_up_context": [], "id": "viewer_engagement_21", "question": "Count watch sessions in calendar month 3.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 4", "follow_up_context": [], "id": "viewer_engagement_22", "question": "Count watch sessions in calendar month 4.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 5", "follow_up_context": [], "id": "viewer_engagement_23", "question": "Count watch sessions in calendar month 5.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(*) AS session_count FROM fact_watch_session w JOIN dim_date d ON w.watch_date_key = d.date_key WHERE d.month_number = 6", "follow_up_context": [], "id": "viewer_engagement_24", "question": "Count watch sessions in calendar month 6.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT SUM(watch_seconds) / 3600.0 AS watch_hours FROM fact_watch_session", "follow_up_context": [], "id": "viewer_engagement_25", "question": "Calculate engagement metric 1.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session", "follow_up_context": [], "id": "viewer_engagement_26", "question": "Calculate engagement metric 2.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT CAST(SUM(completed_flag) AS REAL) / COUNT(*) AS completion_rate FROM fact_watch_session", "follow_up_context": [], "id": "viewer_engagement_27", "question": "Calculate engagement metric 3.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT AVG(score) AS average_rating FROM fact_rating", "follow_up_context": [], "id": "viewer_engagement_28", "question": "Calculate engagement metric 4.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region, COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session WHERE playback_region = 'APAC' GROUP BY playback_region", "follow_up_context": ["Show unique viewers by playback region."], "id": "viewer_engagement_29", "question": "Now show unique viewers in APAC.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region, COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session WHERE playback_region = 'Europe' GROUP BY playback_region", "follow_up_context": ["Show unique viewers by playback region."], "id": "viewer_engagement_30", "question": "Now show unique viewers in Europe.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region, COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session WHERE playback_region = 'North America' GROUP BY playback_region", "follow_up_context": ["Show unique viewers by playback region."], "id": "viewer_engagement_31", "question": "Now show unique viewers in North America.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "query", "expected_sql": "SELECT playback_region, COUNT(DISTINCT user_id) AS unique_viewers FROM fact_watch_session WHERE playback_region = 'Latin America' GROUP BY playback_region", "follow_up_context": ["Show unique viewers by playback region."], "id": "viewer_engagement_32", "question": "Now show unique viewers in Latin America.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_33", "policy_probe_sql": "DELETE FROM fact_watch_session", "question": "Delete all rows", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_34", "policy_probe_sql": "DROP TABLE fact_watch_session", "question": "Drop the table", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_35", "policy_probe_sql": "INSERT INTO fact_watch_session DEFAULT VALUES", "question": "Insert a test row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_36", "policy_probe_sql": "UPDATE fact_watch_session SET 1 = 1", "question": "Update every row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_37", "policy_probe_sql": "ATTACH DATABASE '/tmp/other.sqlite' AS other", "question": "Attach another database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_38", "policy_probe_sql": "WITH RECURSIVE t(n) AS (SELECT 1 UNION ALL SELECT n + 1 FROM t) SELECT * FROM t", "question": "Run a recursive query", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_39", "policy_probe_sql": "SELECT 1; DELETE FROM fact_watch_session", "question": "Run two statements", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "viewer_engagement", "expected_outcome": "policy_rejection", "id": "viewer_engagement_40", "policy_probe_sql": "VACUUM", "question": "Vacuum the database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT subscription_id FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_01", "question": "Show the first five subscription subscription_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT user_id FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_02", "question": "Show the first five subscription user_id values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT plan_name FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_03", "question": "Show the first five subscription plan_name values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT billing_cycle FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_04", "question": "Show the first five subscription billing_cycle values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT status FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_05", "question": "Show the first five subscription status values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT monthly_price_usd FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_06", "question": "Show the first five subscription monthly_price_usd values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT discount_usd FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_07", "question": "Show the first five subscription discount_usd values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT recognized_revenue_usd FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_08", "question": "Show the first five subscription recognized_revenue_usd values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT start_date_key FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_09", "question": "Show the first five subscription start_date_key values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "single_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT end_date_key FROM fact_subscription ORDER BY subscription_id LIMIT 5", "follow_up_context": [], "id": "platform_monetization_10", "question": "Show the first five subscription end_date_key values.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Fan' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_11", "question": "Show subscription revenue for the Fan plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Premium' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_12", "question": "Show subscription revenue for the Premium plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Family' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_13", "question": "Show subscription revenue for the Family plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Fan' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_14", "question": "Show subscription revenue for the Fan plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Premium' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_15", "question": "Show subscription revenue for the Premium plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Family' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_16", "question": "Show subscription revenue for the Family plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Fan' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_17", "question": "Show subscription revenue for the Fan plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "multi_table", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT u.region, SUM(s.recognized_revenue_usd) AS subscription_revenue FROM fact_subscription s JOIN dim_user u ON s.user_id = u.user_id WHERE s.plan_name = 'Premium' GROUP BY u.region", "follow_up_context": [], "id": "platform_monetization_18", "question": "Show subscription revenue for the Premium plan by viewer region.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 1", "follow_up_context": [], "id": "platform_monetization_19", "question": "Show ad revenue in calendar month 1.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 2", "follow_up_context": [], "id": "platform_monetization_20", "question": "Show ad revenue in calendar month 2.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 3", "follow_up_context": [], "id": "platform_monetization_21", "question": "Show ad revenue in calendar month 3.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 4", "follow_up_context": [], "id": "platform_monetization_22", "question": "Show ad revenue in calendar month 4.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 5", "follow_up_context": [], "id": "platform_monetization_23", "question": "Show ad revenue in calendar month 5.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "time", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(a.revenue_usd) AS ad_revenue FROM fact_ad_impression a JOIN dim_date d ON a.impression_date_key = d.date_key WHERE d.month_number = 6", "follow_up_context": [], "id": "platform_monetization_24", "question": "Show ad revenue in calendar month 6.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(recognized_revenue_usd) AS subscription_revenue FROM fact_subscription", "follow_up_context": [], "id": "platform_monetization_25", "question": "Calculate monetization metric 1.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(revenue_usd) AS ad_revenue FROM fact_ad_impression", "follow_up_context": [], "id": "platform_monetization_26", "question": "Calculate monetization metric 2.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT SUM(net_amount_usd) AS merch_gmv FROM fact_merch_order_item", "follow_up_context": [], "id": "platform_monetization_27", "question": "Calculate monetization metric 3.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": true, "category": "metric", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT CAST(SUM(clicked_flag) AS REAL) / COUNT(*) AS ad_ctr FROM fact_ad_impression", "follow_up_context": [], "id": "platform_monetization_28", "question": "Calculate monetization metric 4.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS merch_gmv FROM fact_merch_order_item i JOIN dim_merch_product p ON i.product_id = p.product_id WHERE p.product_category = 'Figure' GROUP BY p.product_category", "follow_up_context": ["Show merchandise GMV by product category."], "id": "platform_monetization_29", "question": "Now show merchandise GMV for Figure.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS merch_gmv FROM fact_merch_order_item i JOIN dim_merch_product p ON i.product_id = p.product_id WHERE p.product_category = 'Apparel' GROUP BY p.product_category", "follow_up_context": ["Show merchandise GMV by product category."], "id": "platform_monetization_30", "question": "Now show merchandise GMV for Apparel.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS merch_gmv FROM fact_merch_order_item i JOIN dim_merch_product p ON i.product_id = p.product_id WHERE p.product_category = 'Poster' GROUP BY p.product_category", "follow_up_context": ["Show merchandise GMV by product category."], "id": "platform_monetization_31", "question": "Now show merchandise GMV for Poster.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "follow_up", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "query", "expected_sql": "SELECT p.product_category, SUM(i.net_amount_usd) AS merch_gmv FROM fact_merch_order_item i JOIN dim_merch_product p ON i.product_id = p.product_id WHERE p.product_category = 'Blu-ray' GROUP BY p.product_category", "follow_up_context": ["Show merchandise GMV by product category."], "id": "platform_monetization_32", "question": "Now show merchandise GMV for Blu-ray.", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "context_note": "Carries follow_up_context but does NOT require it: the question restates its own metric and filter, so a model that ignores session history answers it correctly. Multi-turn context handling is therefore not measured by this case. See evaluation/gold/context_and_compound.jsonl for the cases that do require it.", "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_33", "policy_probe_sql": "DELETE FROM fact_subscription", "question": "Delete all rows", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_34", "policy_probe_sql": "DROP TABLE fact_subscription", "question": "Drop the table", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_35", "policy_probe_sql": "INSERT INTO fact_subscription DEFAULT VALUES", "question": "Insert a test row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_36", "policy_probe_sql": "UPDATE fact_subscription SET 1 = 1", "question": "Update every row", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_37", "policy_probe_sql": "ATTACH DATABASE '/tmp/other.sqlite' AS other", "question": "Attach another database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_38", "policy_probe_sql": "WITH RECURSIVE t(n) AS (SELECT 1 UNION ALL SELECT n + 1 FROM t) SELECT * FROM t", "question": "Run a recursive query", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_39", "policy_probe_sql": "SELECT 1; DELETE FROM fact_subscription", "question": "Run two statements", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} +{"candidate_selection": false, "category": "policy_rejection", "database": "sample_data/anime_streaming/anime_streaming.sqlite", "domain": "platform_monetization", "expected_outcome": "policy_rejection", "id": "platform_monetization_40", "policy_probe_sql": "VACUUM", "question": "Vacuum the database", "semantic_model": "sample_data/anime_streaming/semantic_model.yml", "sql_policy": "sample_data/anime_streaming/sql_policy.yml", "requires_context": false, "compound": false} diff --git a/init.sh b/init.sh new file mode 100755 index 0000000..197e084 --- /dev/null +++ b/init.sh @@ -0,0 +1,104 @@ +#!/bin/bash +# QueryForge verification entry point. +# +# One command that answers "is this checkout healthy?". It checks the environment, +# runs the full offline test suite, and reports repository state. It exits non-zero +# on any failure, so a red baseline is impossible to miss. +# +# Everything here is offline: no network, no model calls, no API spend. +# +# Usage: +# ./init.sh full: environment + test suite + repository state +# ./init.sh --quick environment + repository state only (skip the ~65s suite) +# ./init.sh --eval additionally run the offline tier-1 evaluation gate +# ./init.sh --help + +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +cd "$ROOT" + +QUICK=0 +RUN_EVAL=0 +for arg in "$@"; do + case "$arg" in + --quick) QUICK=1 ;; + --eval) RUN_EVAL=1 ;; + -h|--help) sed -n '2,14p' "${BASH_SOURCE[0]}"; exit 0 ;; + *) echo "unknown option: $arg" >&2; exit 2 ;; + esac +done + +fail() { echo ""; echo "FAILED: $*" >&2; exit 1; } + +echo "=== QueryForge verification ===" +echo "root: $ROOT" +echo + +# ---------------------------------------------------------------- environment +echo "--- environment ---" +[ -f "$ROOT/AGENTS.md" ] || fail "AGENTS.md missing" +[ -f "$ROOT/pyproject.toml" ] || fail "pyproject.toml missing — wrong directory?" +[ -d "$ROOT/.venv" ] || fail ".venv missing — create it with: python3.11 -m venv .venv && .venv/bin/pip install -e ." +[ -x "$ROOT/.venv/bin/python" ] || fail ".venv/bin/python not executable" + +PYVER="$(.venv/bin/python -c 'import sys; print("%d.%d" % sys.version_info[:2])')" +case "$PYVER" in + 3.11|3.12) echo "python: $PYVER (supported)" ;; + *) echo "python: $PYVER — WARNING: project targets 3.11 or 3.12" ;; +esac + +if [ -f "$ROOT/.env" ]; then + echo "env: .env present (real model calls possible)" +else + echo "env: .env absent — offline checks only; real-model runs need provider credentials" + echo " copy .env.example to .env and fill in one provider key" +fi + +# --------------------------------------------------------- leaked-secret guard +# .env is gitignored; a tracked .env means someone force-added it. Fail loudly: +# this repository is public-facing and .env holds live provider credentials. +if git ls-files --error-unmatch .env >/dev/null 2>&1; then + fail ".env is tracked by git — remove it from the index (it holds API keys)" +fi +echo "secret: .env not tracked (ok)" +echo + +# ------------------------------------------------------------- repository state +echo "--- repository state ---" +if [ -n "$(git status --porcelain 2>/dev/null)" ]; then + echo "working tree: dirty ($(git status --porcelain | wc -l | tr -d ' ') changed paths)" +else + echo "working tree: clean" +fi +if git rev-parse --verify origin/HEAD >/dev/null 2>&1; then + echo "branch: $(git rev-parse --abbrev-ref HEAD) @ $(git rev-parse --short HEAD)" +fi +echo + +# --------------------------------------------------------------- verification +if [ "$QUICK" -eq 1 ]; then + echo "--- test suite SKIPPED (--quick) ---" +else + echo "--- full test suite ---" + LOG_LEVEL=CRITICAL .venv/bin/python -m unittest discover -s tests -q 2>&1 | tail -5 \ + || fail "test suite failed — repair the baseline before adding scope" + echo +fi + +if [ "$RUN_EVAL" -eq 1 ]; then + echo "--- offline tier-1 evaluation gate ---" + LOG_LEVEL=CRITICAL .venv/bin/python -B scripts/benchmark_agent.py --tier 1 --report /tmp/qf_tier1_init.json 2>&1 | tail -3 \ + || fail "tier-1 evaluation gate failed" + echo +fi + +echo "=== verification complete ===" +echo +echo "Static / build checks NOT run by this script — run them when you touch these areas:" +echo " make check repository hygiene + offline acceptance (Python)" +echo " make web-check Studio lint + typecheck + build + tests (TypeScript)" +echo " make semantic-check semantic-model drift vs the sample database" +echo +echo "Real-model evaluation spends money. Ask before running tier 3:" +echo " .venv/bin/python -B scripts/benchmark_agent.py --tier 3 --model-provider deepseek --model deepseek-v4-flash" diff --git a/queryforge/application/agent_service.py b/queryforge/application/agent_service.py index a69aae3..a730d40 100644 --- a/queryforge/application/agent_service.py +++ b/queryforge/application/agent_service.py @@ -65,7 +65,7 @@ #: governance-stopped run was relabelled ``cancelled`` and its error text was #: replaced by the disconnect reason (H7). _PERSISTED_TERMINAL_STATUSES = frozenset( - {"completed", "blocked", "failed", "cancelled"} + {"completed", "blocked", "failed", "cancelled", "needs_clarification"} ) @@ -882,10 +882,28 @@ def _run( "tool_loop_max_rounds": options.tool_loop_max_rounds, "tool_loop_timeout_seconds": options.tool_loop_timeout_seconds, "tool_loop_preview_limit": options.tool_loop_preview_limit, - "parallel_candidates": max( - options.parallel_candidates, - 2 if effective_complex else 1, - ), + # Complexity routing deliberately does NOT raise the candidate count. + # + # It used to: a "complex" request set parallel_candidates to 2, while + # also enabling the tool loop. A controlled ablation over identical + # inputs (--parallel-candidates 1 vs 2 in scripts/evaluate_sql.py) found + # 20 paired cases with ZERO accuracy difference, ~20% higher p50 + # latency, and a few percent more tokens. The candidate winner did + # execute faster in the engine (0.57ms vs 1.44ms), but the engine is + # ~0.03% of an end-to-end run dominated by model latency, so that + # gain is ~1ms against ~1550ms of extra work. + # + # The implicit boost was also frequently wasted outright: the tool + # loop runs first, and when it produces SQL the candidate node is + # skipped entirely (workflow.py), so ~0.9 avg tool-call rounds per + # complex run usually bypassed candidates that had already been paid + # for by the complexity decision. + # + # Candidates remain available as an explicit, opt-in choice for a + # caller that has measured a benefit (CLI --parallel-candidates, + # AnalyzeRequest, MCP, or AgentOptions). Details: + # docs/evaluation_baselines.md, "Multi-candidate ablation". + "parallel_candidates": options.parallel_candidates, "parallel_max_preview": options.parallel_max_preview, "parallel_preview_limit": options.parallel_preview_limit, "parallel_preview_timeout_seconds": options.parallel_preview_timeout_seconds, diff --git a/queryforge/application/analysis_planner.py b/queryforge/application/analysis_planner.py index 0ab59ad..ade312b 100644 --- a/queryforge/application/analysis_planner.py +++ b/queryforge/application/analysis_planner.py @@ -324,6 +324,9 @@ def _execute_analysis( validate_run_id(run_id) config = self.config_loader() + # Bound before the branch: the run's version identity is derived from the + # resolved domain when there is one, and stays unknown otherwise. + domain = None if domain_id: from queryforge.domain.domains import DomainResolver domain = DomainResolver.from_config(config).resolve(domain_id) @@ -354,6 +357,11 @@ def _execute_analysis( model_path = semantic_model_path or config.semantic_model_path budget_limits = BudgetLimits().merged(limits or {}) budget = BudgetManager(limits=budget_limits) + # A resumed run inherits what the earlier attempt already spent. Without + # this the allowance reset on every resume, so a crashed-and-restarted run + # could spend its whole budget again while the journal showed the original + # consumption — the boundary only existed for runs that never restarted. + inherited: list[str] = [] # One governed tool stack per analysis run: the connection and the policy # engine are opened here (fail fast on a bad database or policy), bound @@ -414,6 +422,15 @@ def _execute_analysis( journal = resumer.journal if resume: resumer.assert_resumable() + recorded = journal.budget() or {} + inherited = budget.restore(recorded.get("usage")) + if inherited: + LOGGER.info( + "budget_inherited run_id=%s keys=%s usage=%s", + run_id, + ",".join(inherited), + {key: recorded.get("usage", {}).get(key) for key in inherited}, + ) persisted = resumer.load_plan() if persisted is not None: analysis_plan = persisted @@ -429,6 +446,18 @@ def _execute_analysis( max_workers=self.max_workers, mode=mode, journal=journal, + # Bind the run's identities into every step fingerprint: a resume + # must not reuse a step computed against a different database + # snapshot, semantic model or SQL policy. + run_versions={ + "data_version": data_version, + "semantic_version": ( + domain.semantic_version if domain is not None else None + ), + "policy_version": ( + domain.policy_version if domain is not None else None + ), + }, worker_id=f"planner-{run_id or uuid4().hex[:6]}", cancel_check=cancel_check, force_resume=force_resume, diff --git a/queryforge/cli.py b/queryforge/cli.py index dc01010..dddd74e 100644 --- a/queryforge/cli.py +++ b/queryforge/cli.py @@ -21,6 +21,7 @@ SQLHistoryStore, ) from queryforge.infrastructure.tools.database_tool import DatabaseTool +from queryforge.domain.knowledge import StructuredKnowledgeBase from queryforge.domain.security import load_sql_policy from queryforge.domain.semantic.builder import SemanticBuildError, SemanticModelBuilder @@ -421,6 +422,15 @@ def build_parser() -> argparse.ArgumentParser: metavar="PATH", help="SQL/Jinja/CSV file or directory included during rebuild; repeatable", ) + parser.add_argument( + "--kb-knowledge", + metavar="PATH", + help="JSON file holding governed structured knowledge (metrics, glossary, " + "sources) to include during --rebuild-vector-kb. Without this the governed " + "path is never exercised in a deployment, so the three-tier verification, " + "conflict detection and holdout isolation implemented in " + "domain/knowledge/governance.py do not run.", + ) parser.add_argument( "--show-history", action="store_true", @@ -611,6 +621,12 @@ def main() -> int: file=sys.stderr, ) return 2 + if args.kb_knowledge and not args.rebuild_vector_kb: + print( + "QueryForge failed: --kb-knowledge requires --rebuild-vector-kb", + file=sys.stderr, + ) + return 2 if vector_action: if args.vector_top_k < 0 or args.vector_top_k > 20: print( @@ -664,13 +680,37 @@ def main() -> int: builder = KnowledgeBaseBuilder( vector_store, manifest_path=manifest_path ) + knowledge = None + if args.kb_knowledge: + knowledge_path = Path(args.kb_knowledge).expanduser() + if not knowledge_path.is_file(): + print( + f"QueryForge failed: --kb-knowledge file not found: {knowledge_path}", + file=sys.stderr, + ) + return 2 + knowledge = StructuredKnowledgeBase.load( + knowledge_path.read_text(encoding="utf-8") + ) result["rebuild"] = builder.rebuild( history_store=history_store, schemas=schemas, sources=args.kb_source, + # The governed path is only reachable when a structured + # knowledge base is supplied; passing it here is what makes + # build_governed_documents (verification tiers, conflict + # detection, holdout isolation) run outside tests. + knowledge=knowledge, ) result["manifest_path"] = str(manifest_path) result["sources"] = [str(Path(path).expanduser()) for path in args.kb_source] + if knowledge is not None: + result["knowledge"] = { + "path": str(Path(args.kb_knowledge).expanduser()), + "metrics": len(knowledge.metrics), + "glossary": len(knowledge.glossary), + "sources": len(knowledge.sources), + } if args.kb_stats: result["stats"] = vector_store.stats() except Exception as exc: diff --git a/queryforge/core/config.py b/queryforge/core/config.py index 9c7b770..ff37858 100644 --- a/queryforge/core/config.py +++ b/queryforge/core/config.py @@ -10,8 +10,12 @@ import yaml from dotenv import load_dotenv +from queryforge.core.paths import workspace_root -PROJECT_ROOT = Path(__file__).resolve().parents[2] +#: Repository/workspace root. Deliberately not ``Path(__file__).parents[2]``: that +#: resolves to ``site-packages`` once the package is *installed* (E-22). See +#: ``queryforge.core.paths`` for the resolution order. +PROJECT_ROOT = workspace_root() DEFAULT_MODELS_CONFIG = PROJECT_ROOT / "models.yml" PACKAGED_MODELS_CONFIG = Path(__file__).with_name("default_models.yml") DEFAULT_DATABASE_PATH = "sample_data/anime_streaming/anime_streaming.sqlite" diff --git a/queryforge/core/observability.py b/queryforge/core/observability.py index f3925ee..5d07a3c 100644 --- a/queryforge/core/observability.py +++ b/queryforge/core/observability.py @@ -21,8 +21,10 @@ from dotenv import load_dotenv +from queryforge.core.paths import workspace_root -PROJECT_ROOT = Path(__file__).resolve().parents[2] + +PROJECT_ROOT = workspace_root() DEFAULT_LOG_PATH = PROJECT_ROOT / ".queryforge/logs/queryforge.log" DEFAULT_TRACE_DIR = PROJECT_ROOT / ".queryforge/traces" _RUN_ID = ContextVar("queryforge_run_id", default="-") @@ -706,7 +708,14 @@ def usage_summary(self) -> dict[str, Any]: } def latency_summary(self, *, end_to_end_ms: float | None = None) -> dict[str, Any]: - """Per-kind latency breakdown plus the run's end-to-end duration.""" + """Per-kind latency breakdown plus the run's end-to-end duration. + + ``by_node`` aggregates the *model* spans by the node that issued them. A + run's wall clock is dominated by sequential model calls, so the run-level + total cannot say which step to optimise; each node span carries its + ``node`` attribute and each model span is opened inside one, which is + enough to attribute the cost without new instrumentation. + """ with self._lock: spans = list(self._spans) @@ -714,6 +723,7 @@ def latency_summary(self, *, end_to_end_ms: float | None = None) -> dict[str, An kind: {"count": 0, "duration_ms": 0.0, "max_duration_ms": 0.0} for kind in SPAN_KINDS } + by_node: dict[str, dict[str, Any]] = {} for span in spans: entry = by_kind.setdefault( span.kind, {"count": 0, "duration_ms": 0.0, "max_duration_ms": 0.0} @@ -721,11 +731,43 @@ def latency_summary(self, *, end_to_end_ms: float | None = None) -> dict[str, An entry["count"] += 1 entry["duration_ms"] = round(entry["duration_ms"] + span.duration_ms, 3) entry["max_duration_ms"] = max(entry["max_duration_ms"], span.duration_ms) + + if span.kind != "model": + continue + node = str(span.node_name or (span.attributes or {}).get("node") or "unknown") + node_entry = by_node.setdefault( + node, + { + "model_calls": 0, + "model_duration_ms": 0.0, + "prompt_tokens": 0, + "completion_tokens": 0, + }, + ) + node_entry["model_calls"] += 1 + node_entry["model_duration_ms"] = round( + node_entry["model_duration_ms"] + span.duration_ms, 3 + ) + usage = span.usage + if usage is not None: + node_entry["prompt_tokens"] += int( + getattr(usage, "prompt_tokens", 0) or 0 + ) + node_entry["completion_tokens"] += int( + getattr(usage, "completion_tokens", 0) or 0 + ) + ordered = { + name: by_node[name] + for name in sorted( + by_node, key=lambda key: by_node[key]["model_duration_ms"], reverse=True + ) + } if end_to_end_ms is None: end_to_end_ms = self._span_window_ms(spans) return { "end_to_end_ms": end_to_end_ms, "by_kind": by_kind, + "by_node": ordered, "span_count": len(spans), } @@ -1029,6 +1071,75 @@ def _isolate_usage_slot(provider: Any) -> bool: return True +#: Seconds remaining on the run's model deadline for the *current* call. +#: +#: A ContextVar rather than a parameter: the provider interface is implemented by +#: several adapters and called from seven decision points, and threading a new +#: keyword through all of them would be a wide change for a value that is +#: ambient per call. It also propagates into threads that inherit the context +#: (parallel candidates, the tool loop), which is where a deadline matters most. +#: ``None`` means no deadline was declared, which is different from "no time left". +_MODEL_DEADLINE_SECONDS: ContextVar[float | None] = ContextVar( + "queryforge_model_deadline_seconds", default=None +) + + +def current_model_deadline() -> float | None: + """Remaining seconds for the in-flight model call, or None when unbounded.""" + + return _MODEL_DEADLINE_SECONDS.get() + + +class model_deadline: + """Set the remaining model deadline for calls made inside the block. + + A non-positive deadline raises :class:`BudgetDeadlineExceeded` before any + request is sent, so an exhausted run does not pay for a call it cannot use. + """ + + def __init__(self, seconds: float | None) -> None: + self.seconds = None if seconds is None else float(seconds) + self._token: Any = None + + def __enter__(self) -> "model_deadline": + if self.seconds is not None: + if self.seconds <= 0: + raise BudgetDeadlineExceeded( + "run deadline exhausted before the model call was sent" + ) + self._token = _MODEL_DEADLINE_SECONDS.set(self.seconds) + return self + + def __exit__(self, *exc_info: object) -> None: + if self._token is not None: + _MODEL_DEADLINE_SECONDS.reset(self._token) + self._token = None + + +class BudgetDeadlineExceeded(RuntimeError): + """The run's model deadline was already exhausted when a call was attempted.""" + + +def _call_with_optional_timeout(operation: Any, prompt: str, timeout: float | None) -> Any: + """Call ``operation(prompt)``, adding ``timeout`` only if it accepts one. + + Providers and decorators are duck-typed here, so a callee that has no + ``timeout`` parameter must keep working unchanged. + """ + + if timeout is None: + return operation(prompt) + try: + import inspect + + parameters = inspect.signature(operation).parameters + except (TypeError, ValueError): + return operation(prompt) + if "timeout" in parameters: + return operation(prompt, timeout=timeout) + return operation(prompt) + + class ObservedModelProvider: """Duck-typed provider decorator that records summaries, never prompts by default. @@ -1078,23 +1189,44 @@ def __init__( self._inflight_lock = threading.Lock() def generate_json(self, prompt: str) -> dict[str, Any]: + # The adapter is delegated to for its own JSON handling, but if the wrapped + # provider is itself a decorator (the budget wrapper) it needs the ambient + # deadline to bound the request. Passing it when the callee accepts it + # keeps decorator stacking working without changing the adapter contract. return self._observe( - "generate_json", prompt, lambda: self._provider.generate_json(prompt) + "generate_json", + prompt, + lambda: _call_with_optional_timeout( + self._provider.generate_json, prompt, current_model_deadline() + ), ) def generate_text(self, prompt: str) -> str: return self._observe( - "generate_text", prompt, lambda: self._provider.generate_text(prompt) + "generate_text", + prompt, + lambda: _call_with_optional_timeout( + self._provider.generate_text, prompt, current_model_deadline() + ), ) def generate_with_messages( - self, messages: list[dict[str, str]], json_mode: bool = False + self, + messages: list[dict[str, str]], + json_mode: bool = False, + timeout: float | None = None, ) -> str: prompt = json.dumps(messages, ensure_ascii=False) return self._observe( "generate_with_messages", prompt, - lambda: self._provider.generate_with_messages(messages, json_mode=json_mode), + lambda: self._provider.generate_with_messages( + messages, + json_mode=json_mode, + # An explicit argument wins; otherwise the ambient run deadline + # applies, so a caller that sets nothing still gets bounded. + timeout=timeout if timeout is not None else current_model_deadline(), + ), ) def __getattr__(self, name: str) -> Any: diff --git a/queryforge/core/outcomes.py b/queryforge/core/outcomes.py new file mode 100644 index 0000000..e5ad48c --- /dev/null +++ b/queryforge/core/outcomes.py @@ -0,0 +1,143 @@ +"""Terminal run outcomes: one vocabulary, one derivation. + +Why this module exists +---------------------- +Before this, the terminal status of a run was decided in more than one place and +described with more than one vocabulary: + +* ``OrchestratorAgent`` derived ``state.status`` from the workflow's result dict. +* ``AnalysisExecutor`` derived its own ``AnalysisExecutionResult.status``. +* ``TaskStatus``, ``PlanStatus`` and ``evolve``'s event outcome set each listed + their own overlapping values. + +That is how defect E-02 happened: the QA gate set ``state.status = "blocked"``, +the orchestrator then re-derived the status from the workflow's *earlier* result +dict and overwrote it to ``"completed"``, so ``state.json`` said one thing while +the caller was told another. Both were "correct" according to their own local +rule. + +The fix is structural: a single derivation function, used by every path, plus a +single compatibility mapping onto the transport event vocabulary. Callers record +a terminal outcome once and do not re-derive it from a stale value. +""" + +from __future__ import annotations + +from typing import Any, Literal + + +#: The canonical terminal vocabulary. Anything persisted as a run's final status +#: must be one of these. +TerminalOutcome = Literal[ + "succeeded", + "needs_clarification", + "blocked", + "partial", + "failed", + "cancelled", +] + +TERMINAL_OUTCOMES: tuple[str, ...] = ( + "succeeded", + "needs_clarification", + "blocked", + "partial", + "failed", + "cancelled", +) + +#: Outcomes that mean the attempt is over, whichever way it ended. +TERMINAL_AND_FINAL: frozenset[str] = frozenset(TERMINAL_OUTCOMES) + +#: Outcomes a late cancellation may not overwrite. A cancelled run may only be +#: replaced while it is still running. +NON_REPLACEABLE: frozenset[str] = frozenset( + {"succeeded", "needs_clarification", "blocked", "partial", "failed", "cancelled"} +) + + +def normalize_outcome(value: Any) -> TerminalOutcome | None: + """Map any historical status spelling onto the canonical vocabulary. + + Returns ``None`` when the value is not a terminal outcome (for example + ``"running"``), so callers can distinguish "not finished" from "finished in + some way". + """ + + text = str(value or "").strip().lower() + if not text: + return None + return _ALIASES.get(text) + + +#: Spellings produced by the different layers before this module existed, plus the +#: action vocabulary of :class:`TaskStatus` and the evaluator's outcome set. +_ALIASES: dict[str, TerminalOutcome] = { + # canonical + "succeeded": "succeeded", + "needs_clarification": "needs_clarification", + "blocked": "blocked", + "partial": "partial", + "failed": "failed", + "cancelled": "cancelled", + # workflow / task-status spellings + "success": "succeeded", + "completed": "succeeded", + "planned": "succeeded", + "ok": "succeeded", + "degraded": "partial", + "error": "failed", + "canceled": "cancelled", +} + +#: Transport event protocol outcomes. Deliberately smaller than the canonical set: +#: the stream protocol distinguishes only these, and a clarification is reported to +#: a streaming client as a blocked run with a reason. +EVENT_OUTCOMES: dict[str, str] = { + "succeeded": "success", + "partial": "partial", + "needs_clarification": "blocked", + "blocked": "blocked", + "failed": "failed", + "cancelled": "cancelled", +} + + +def derive_outcome( + *, + result_status: Any = None, + blocked_reason: str | None = None, + cancelled: bool = False, +) -> TerminalOutcome: + """Derive the terminal outcome from the signals a run actually produces. + + Precedence is deliberate and total, so no caller has to re-derive anything: + + 1. an explicit cancellation wins (the run stopped because the client left); + 2. a recorded gate block wins (a deterministic gate refused the stage and + recorded why) — this is the case E-02 got wrong; + 3. otherwise the workflow's own result status, normalised; + 4. otherwise the run is treated as succeeded. + """ + + if cancelled: + return "cancelled" + if blocked_reason: + return "blocked" + normalized = normalize_outcome(result_status) + if normalized is None: + # Not a terminal spelling: an unknown or non-terminal value is treated as + # a success, which matches the historical behaviour of defaulting to + # "success" while making the decision explicit and testable. + return "succeeded" + return normalized + + +def to_event_outcome(outcome: TerminalOutcome) -> str: + """Map a canonical outcome onto the streaming event vocabulary.""" + + return EVENT_OUTCOMES.get(outcome, "failed") + + +def is_terminal(value: Any) -> bool: + return normalize_outcome(value) is not None diff --git a/queryforge/core/paths.py b/queryforge/core/paths.py new file mode 100644 index 0000000..68a8183 --- /dev/null +++ b/queryforge/core/paths.py @@ -0,0 +1,89 @@ +"""Locate the workspace a run reads from and writes to. + +Two different kinds of "root" exist in this project, and conflating them was a real +defect (E-22): + +* **Packaged resources** — ``models.yml``'s fallback, ``bundled_skills/`` — ship + *inside* the ``queryforge`` package. They are found relative to ``__file__`` and + work identically from a source checkout and from an installed wheel. +* **The workspace** — ``.queryforge/`` state (history, traces, logs, charts, the + vector KB), ``sample_data/``, ``evaluation/`` — belongs to whoever is *running* + QueryForge, not to the installed package. + +Every module used to derive the second one as ``Path(__file__).parents[N]``. That +resolves to the repository root in a developer checkout and to ``site-packages`` +after ``pip install``, where ``.queryforge/history.db`` is (a) not the user's data +and (b) frequently unwritable. Resolution is therefore explicit and ordered: + +1. ``QUERYFORGE_ROOT`` — an explicit operator decision, always wins. +2. The nearest ancestor of the package that looks like a source checkout + (a marker such as ``pyproject.toml``), so checkout behaviour is unchanged and + a vendored/zipped layout still finds its own root. +3. The current working directory, which is the user's workspace when the package + lives in ``site-packages``. + +Resolution happens once per process, because an installed process cannot +meaningfully switch workspaces mid-run. +""" + +from __future__ import annotations + +import os +from functools import lru_cache +from pathlib import Path + +__all__ = ["WORKSPACE_ROOT_ENV", "resolve_path", "workspace_root"] + +#: Environment variable an operator sets to point QueryForge at a workspace. +WORKSPACE_ROOT_ENV = "QUERYFORGE_ROOT" + +#: Files that only ever exist at the root of a source checkout (or a deployment +#: that mirrors one). ``.git`` covers worktrees and bare checkouts. +_CHECKOUT_MARKERS = ("pyproject.toml", "models.yml", ".git") + + +def _is_checkout(path: Path) -> bool: + return any((path / marker).exists() for marker in _CHECKOUT_MARKERS) + + +@lru_cache(maxsize=1) +def workspace_root() -> Path: + """Return the resolved workspace root for this process. + + Falls back to the current working directory rather than to ``site-packages``: + a wrong-but-writable directory is a silent data-integrity bug, whereas the + working directory is at least where the caller asked QueryForge to work. + """ + return _resolve_root(Path(__file__).resolve().parent) + + +def _resolve_root(package_dir: Path) -> Path: + """Apply the documented resolution order to an arbitrary package location. + + Split out from :func:`workspace_root` so the installed-package case (a package + directory with no checkout ancestor, i.e. ``site-packages``) is testable + without patching ``__file__``. + """ + override = (os.getenv(WORKSPACE_ROOT_ENV) or "").strip() + if override: + candidate = Path(override).expanduser() + if not candidate.is_absolute(): + candidate = Path.cwd() / candidate + return candidate.resolve() + + for candidate in (package_dir, *package_dir.parents): + if _is_checkout(candidate): + return candidate + return Path.cwd().resolve() + + +def resolve_path(path: str | Path, *, base: Path | None = None) -> Path: + """Resolve ``path`` against the workspace root, expanding ``~``. + + Absolute paths pass through untouched, so a caller-supplied path is never + silently reinterpreted. + """ + resolved = Path(path).expanduser() + if resolved.is_absolute(): + return resolved + return (base or workspace_root()) / resolved diff --git a/queryforge/core/schemas/__init__.py b/queryforge/core/schemas/__init__.py index 699f08a..d31e492 100644 --- a/queryforge/core/schemas/__init__.py +++ b/queryforge/core/schemas/__init__.py @@ -12,6 +12,7 @@ ReasoningMetric, ReasoningResult, ReasoningSort, + RunContext, SQLContext, SqlPolicyDecision, SqlTask, @@ -33,6 +34,7 @@ "ReasoningResult", "ReasoningSort", "ReportArtifact", + "RunContext", "ReportSection", "SQLContext", "SqlPolicyDecision", diff --git a/queryforge/core/schemas/models.py b/queryforge/core/schemas/models.py index 727dde3..36f6403 100644 --- a/queryforge/core/schemas/models.py +++ b/queryforge/core/schemas/models.py @@ -158,6 +158,14 @@ class ExecutionResult(BaseModel): columns: list[str] = Field(default_factory=list) rows: list[list[Any]] = Field(default_factory=list) row_count: int = 0 + #: True when the adapter's row bound cut the result short. Without this the + #: bound was invisible: ``_enforce_row_bound`` overwrote ``row_count`` with the + #: truncated length, so a caller could not tell a complete result from a + #: capped one, and the original size was not recorded anywhere. + truncated: bool = False + #: Rows the engine actually produced, before the bound was applied. Equal to + #: ``row_count`` when ``truncated`` is False. + fetched_row_count: int = 0 class SqlPolicyDecision(BaseModel): @@ -236,6 +244,49 @@ class NodeResult(BaseModel): duration_ms: float | None = None +class RunContext(BaseModel): + """Run identity and the versions a run executed against. + + Both execution paths (the conversational workflow and the planner) populate + this, so a run's identity and the semantic/data/policy versions it depended on + travel with the payload instead of being reconstructed from whichever layer + happens to be asking. Artifact provenance and recovery fingerprints need the + same four versions, and before this each caller assembled them separately. + """ + + run_id: str + task_id: str | None = None + session_id: str | None = None + domain_id: str | None = None + #: Versions the answer is only valid for. ``None`` means "not declared", which + #: is different from "unchanged" and is reported as such. + data_version: str | None = None + semantic_version: str | None = None + policy_version: str | None = None + entrypoint: str | None = None + + @classmethod + def from_context(cls, context: "Context", **overrides: Any) -> "RunContext": + """Build from anything already known, leaving unknown fields as None.""" + + task_context = context.task_context if isinstance(context.task_context, dict) else {} + scope = task_context.get("retrieval_scope") + scope = scope if isinstance(scope, dict) else {} + semantic = context.semantic_model + values: dict[str, Any] = { + "run_id": context.run_id, + "task_id": task_context.get("task_id"), + "session_id": task_context.get("session_id"), + "domain_id": scope.get("domain_id"), + "data_version": scope.get("data_version"), + "semantic_version": getattr(getattr(semantic, "model", None), "version", None), + "policy_version": (context.sql_policy or {}).get("version"), + "entrypoint": task_context.get("entrypoint"), + } + values.update({key: value for key, value in overrides.items() if value is not None}) + return cls(**values) + + class Context(BaseModel): task: SqlTask run_id: str = Field(default_factory=lambda: f"qf_{uuid4().hex}") @@ -251,6 +302,12 @@ class Context(BaseModel): execution_plan: ExecutionPlan | None = None plan_approved: bool | None = None reflection_result: ReflectionResult | None = None + #: Why a model-supplied ``reasoning`` payload was rejected, when it was. The + #: rejection used to be silent: the payload was validated, failed, and dropped + #: with nothing but a warning buried in ``reasoning_validation``, so a model + #: that always emitted an incompatible shape looked identical to one that + #: emitted nothing at all. + reasoning_discarded: str | None = None fix_attempts: list[FixAttempt] = Field(default_factory=list) retry_count: int = 0 execution_errors: list[str] = Field(default_factory=list) @@ -297,9 +354,25 @@ class Context(BaseModel): ] = "disabled" tool_loop_exit_reason: str | None = None candidate_selection: dict[str, Any] | None = None + #: Which SQL-producing strategy this attempt used: ``tool_loop`` when the + #: exploration loop produced the answer, ``parallel_candidates`` when several + #: candidates were generated and selected, ``single_generation`` otherwise. + #: Recorded because "complex" enables both the tool loop and extra candidates, + #: and the tool loop wins — so a run cannot be audited for candidate use without + #: this field. + candidate_strategy: str = "unknown" reasoning_result: ReasoningResult | None = None reasoning_validation: dict[str, Any] | None = None node_results: list[NodeResult] = Field(default_factory=list) # Shared structured context for step 04/05/07/08 workflows (schema # retrieval evidence, typed error categories, analysis patches, ...). task_context: dict[str, Any] = Field(default_factory=dict) + #: Unified run identity and the versions this run depends on. + run_context: RunContext | None = None + #: Shared budget bounding every model call in this run, or None when the path + #: does not charge model calls. Excluded from serialization: it is a live + #: object graph, and its snapshot is reported instead. + model_budget: Any | None = Field(default=None, exclude=True) + #: Populated when a model call was refused by that budget, so a stopped run can + #: say it stopped for budget rather than merely that it stopped. + budget_refusal: dict[str, Any] = Field(default_factory=dict) diff --git a/queryforge/domain/semantic/builder.py b/queryforge/domain/semantic/builder.py index 1c51230..81d6d7c 100644 --- a/queryforge/domain/semantic/builder.py +++ b/queryforge/domain/semantic/builder.py @@ -12,6 +12,7 @@ import yaml +from queryforge.core.paths import workspace_root from queryforge.core.schemas.models import TableColumn, TableSchema from queryforge.domain.semantic.contract_validator import SemanticContractValidator from queryforge.domain.semantic.model import SemanticModelLoader @@ -37,7 +38,10 @@ re.IGNORECASE, ) TECHNICAL_PREFIXES = ("dim_", "fact_", "bridge_", "stg_", "raw_") -PROJECT_ROOT = Path(__file__).resolve().parents[3] +#: Workspace root, used only to make emitted paths portable (``_portable_path``). +#: Resolved rather than ``parents[3]`` so an installed package does not classify +#: every path as "not portable" against ``site-packages`` (E-22). +PROJECT_ROOT = workspace_root() class SemanticBuildError(ValueError): diff --git a/queryforge/domain/semantic/sql_validator.py b/queryforge/domain/semantic/sql_validator.py index 741fb2b..8972c8f 100644 --- a/queryforge/domain/semantic/sql_validator.py +++ b/queryforge/domain/semantic/sql_validator.py @@ -405,6 +405,32 @@ def aggregates(self) -> list[tuple[exp.Expression, _Aggregate]]: ) return observed +def _operator_symbol(comparison: Any) -> str: + """``gte``/``gt``/``lte``/``lt``/``eq`` for one comparison node.""" + + for name, symbol in ( + ("GTE", "gte"), + ("GT", "gt"), + ("LTE", "lte"), + ("LT", "lt"), + ("EQ", "eq"), + ): + node_type = getattr(exp, name, None) + if node_type is not None and isinstance(comparison, node_type): + return symbol + return "unknown" + + +def _digits(value: Any) -> str: + """Comparable digit form: ``2024-01-01`` and ``20240101`` both become digits. + + The resolved range is an ISO date while a date-key column is an integer, so the + two must be compared in a shape-independent way. + """ + + return "".join(char for char in str(value) if char.isdigit()) + + class SemanticSQLValidator: """Prove that generated SQL honours the governed semantic contract.""" @@ -794,7 +820,19 @@ def _check_metric_expressions( def _default_filter_expectations( self, raw_filter: str, base_table: str - ) -> tuple[list[tuple[str, str]], tuple[str, Any] | None] | None: + ) -> tuple[list[tuple[str, str]] | None, tuple[str, Any] | None] | None: + """The columns and literal a declared default filter is expected to use. + + Returns ``None`` when the filter is not a declarative predicate at all + (unparsable — the caller reports that as a violation), or a + ``(columns, literal)`` pair where ``columns`` may itself be ``None`` for a + filter that parses but references **no column** (for example + ``EXISTS(SELECT 1 FROM t.x)``). The annotation says so because the previous + one claimed ``list`` and the caller trusted it: passing ``None`` on to + ``set(columns)`` raised ``TypeError`` out of ``validate()``, which escapes + the node's try block and crashes the semantic gate instead of producing a + verdict. + """ node = _parse_expression(raw_filter) if node is None: return None @@ -852,7 +890,25 @@ def _check_default_filters( ) continue columns, literal = expected - record: dict[str, Any] = { + if not columns: + # Parses, but names no column: there is nothing to look for in + # the SQL, so the filter cannot be proved either way. Reported + # rather than crashing, and not silently treated as satisfied. + self._add_violation( + violations, + RULE_DEFAULT_FILTER, + f"metric {metric.name!r} default filter {raw_filter!r} " + "references no column, so its presence in the SQL cannot " + "be verified", + ) + record: dict[str, Any] = { + "metric": metric.name, + "filter": raw_filter, + "status": "unverifiable_no_column", + } + evidence["default_filters"].append(record) + continue + record = { "metric": metric.name, "filter": raw_filter, "status": "missing", @@ -1021,6 +1077,44 @@ def _check_filter_presence( f"effective predicate references {rendered}", ) + @staticmethod + def _time_boundaries( + index: "_AstIndex", wanted: set[tuple[str, str]] + ) -> list[tuple[str, Any]]: + """Literal comparisons applied to the governed time column. + + Returns ``(operator, literal)`` for every predicate that compares the time + field to a value, so the boundaries can be checked against the resolved + date range instead of only checking that *some* predicate mentions the + column. + """ + + boundaries: list[tuple[str, Any]] = [] + for scope, _kind, predicate in index.predicates: + for node in predicate.find_all(exp.Column): + if not (index.resolve_column(scope, node) & wanted): + continue + # BETWEEN is its own node type, not a pair of comparisons, and the + # governed compiler emits it for a resolved date range. Missing it + # made every compiler-produced time filter look unbounded. + between = node.find_ancestor(exp.Between) + if between is not None: + low, high = between.args.get("low"), between.args.get("high") + for operator, bound in (("gte", low), ("lte", high)): + if isinstance(bound, exp.Literal): + boundaries.append((operator, _literal_value(bound))) + continue + comparison = node.find_ancestor( + exp.EQ, exp.GTE, exp.GT, exp.LTE, exp.LT + ) + if comparison is None: + continue + literal = comparison.expression + if not isinstance(literal, exp.Literal): + continue + boundaries.append((_operator_symbol(comparison), _literal_value(literal))) + return boundaries + def _check_time_filter( self, index: _AstIndex, @@ -1061,10 +1155,27 @@ def _check_time_filter( break if present: break + boundaries = ( + self._time_boundaries(index, wanted) if present else [] + ) + verdict = "presence_checked" + if present and not boundaries: + # The column is mentioned but never compared to anything, so the + # predicate cannot be bounding the window. + verdict = "presence_only" + elif present: + verdict = "boundary_checked" evidence["time_filter"] = { "metric": metric.name, "time_field": time_field, - "status": "presence_checked" if present else "missing", + "status": verdict, + "resolved_ranges": [ + {"start": start, "end": end} for start, end in ranges + ], + "observed_boundaries": [ + {"operator": operator, "value": value} + for operator, value in boundaries + ], } if not present: self._add_violation( @@ -1073,6 +1184,61 @@ def _check_time_filter( f"metric {metric.name!r} time_field {time_field} is not filtered " "although the request resolves a date range", ) + elif not boundaries: + # Asymmetric with default filters, which do compare literals. A + # predicate that mentions the column without comparing it does not + # constrain the window, so the check would otherwise pass on a + # statement that ignores the requested range. + self._add_violation( + violations, + RULE_TIME_FILTER, + f"metric {metric.name!r} time_field {time_field} is referenced but " + "compared to no value, so the requested date range is not applied", + ) + elif not self._boundaries_cover_ranges(boundaries, ranges): + self._add_violation( + violations, + RULE_TIME_FILTER, + f"metric {metric.name!r} time_field {time_field} is compared to " + f"{[value for _operator, value in boundaries]} which does not cover " + f"the requested range(s) " + f"{[f'{start}..{end}' for start, end in ranges]}", + ) + + @staticmethod + def _boundaries_cover_ranges( + boundaries: list[tuple[str, Any]], + ranges: list[tuple[str, str]], + ) -> bool: + """Whether the observed comparisons can express the requested window. + + Only *shape* is checked, not calendar arithmetic: at least one lower bound + and one upper bound must be present, in a form comparable to the range + strings. A comparison that pins a different value is not "wrong" here — the + resolved range is derived from the question, so a mismatch means the SQL + filtered something else. Values are normalised to digits so a date-key form + (``20240101``) and an ISO form (``2024-01-01``) both compare. + """ + + lower = {"gte", "gt", "eq"} + upper = {"lte", "lt", "eq"} + observed = { + (_digits(value), operator) for operator, value in boundaries + } + for start, end in ranges: + start_digits = _digits(start) + end_digits = _digits(end) + has_lower = any( + operator in lower and value >= start_digits + for value, operator in observed + ) + has_upper = any( + operator in upper and value <= end_digits + for value, operator in observed + ) + if not (has_lower and has_upper): + return False + return True def _check_join_keys( self, diff --git a/queryforge/domain/skills/registry.py b/queryforge/domain/skills/registry.py index 715b2eb..9d77400 100644 --- a/queryforge/domain/skills/registry.py +++ b/queryforge/domain/skills/registry.py @@ -10,6 +10,10 @@ import yaml +#: The installed package directory — *not* the workspace root. ``bundled_skills`` +#: ships inside the package (``pyproject.toml`` package-data), so this must stay +#: relative to ``__file__`` and deliberately differs from +#: ``queryforge.core.config.PROJECT_ROOT`` (E-22). PACKAGE_ROOT = Path(__file__).resolve().parents[2] DEFAULT_SKILLS_DIR = PACKAGE_ROOT / "bundled_skills" SKILL_NAME_PATTERN = re.compile(r"^[a-z][a-z0-9_-]*$") diff --git a/queryforge/evaluation/__init__.py b/queryforge/evaluation/__init__.py index a0ab3a0..94ec519 100644 --- a/queryforge/evaluation/__init__.py +++ b/queryforge/evaluation/__init__.py @@ -5,7 +5,7 @@ grades: it imports nothing but the standard library and pydantic, never ``queryforge.workflow``/``application``/``orchestration``/``interfaces``, so a bug in the runtime cannot make the benchmark lenient (see -``tests/test_evaluation_isolation.py``). +``tests/test_agent_task_gold.py::test_evaluator_does_not_import_runtime_being_graded``). Typical use:: diff --git a/queryforge/evaluation/evaluator.py b/queryforge/evaluation/evaluator.py index 5f999ed..caf0288 100644 --- a/queryforge/evaluation/evaluator.py +++ b/queryforge/evaluation/evaluator.py @@ -21,7 +21,7 @@ This module imports nothing but the standard library and pydantic: the evaluator must not be able to share a bug with the code it grades (see -``tests/test_evaluation_isolation.py``). +``tests/test_agent_task_gold.py::test_evaluator_does_not_import_runtime_being_graded``). """ from __future__ import annotations diff --git a/queryforge/evaluation/thresholds.py b/queryforge/evaluation/thresholds.py index b0e438a..2994f53 100644 --- a/queryforge/evaluation/thresholds.py +++ b/queryforge/evaluation/thresholds.py @@ -24,14 +24,15 @@ from pydantic import BaseModel, ConfigDict, Field, model_validator +from queryforge.core.paths import resolve_path + #: Tiers of the benchmark; they are reported separately and never mixed. TIER_NAMES: tuple[str, ...] = ("tier1_offline", "tier2_integration", "tier3_model_e2e") -#: The checked-in thresholds document (repo relative; this file lives in -#: ``queryforge/evaluation/``, so two parents up is the project root). -DEFAULT_THRESHOLDS_PATH = ( - Path(__file__).resolve().parents[2] / "evaluation" / "thresholds.json" -) +#: The checked-in thresholds document: workspace relative, resolved through +#: ``core.paths`` rather than derived from ``__file__`` so an installed package +#: does not look for it inside ``site-packages`` (E-22). +DEFAULT_THRESHOLDS_PATH = resolve_path("evaluation/thresholds.json") #: Rule keys inside a tier: ``min_`` / ``max_``. _RULE_PREFIXES: tuple[str, ...] = ("min_", "max_") diff --git a/queryforge/infrastructure/db/adapter.py b/queryforge/infrastructure/db/adapter.py index 330ece0..1a28f78 100644 --- a/queryforge/infrastructure/db/adapter.py +++ b/queryforge/infrastructure/db/adapter.py @@ -819,12 +819,30 @@ def _translate_engine_error( def _enforce_row_bound( result: ExecutionResult, limit: int | None ) -> ExecutionResult: - """Re-apply the bound after fetch: the engine bound is not the contract.""" - if limit is None or len(result.rows) <= limit: - return result + """Re-apply the bound after fetch: the engine bound is not the contract. + + The result keeps both numbers: ``fetched_row_count`` is what the engine + produced and ``row_count`` is what the caller receives, with ``truncated`` + saying they differ. Overwriting ``row_count`` alone made a capped result + indistinguishable from a complete one. + """ + + fetched = len(result.rows) + if limit is None or fetched <= limit: + return ExecutionResult( + columns=list(result.columns), + rows=result.rows, + row_count=result.row_count, + truncated=False, + fetched_row_count=max(fetched, result.row_count), + ) rows = result.rows[:limit] return ExecutionResult( - columns=list(result.columns), rows=rows, row_count=len(rows) + columns=list(result.columns), + rows=rows, + row_count=len(rows), + truncated=True, + fetched_row_count=max(fetched, result.row_count), ) diff --git a/queryforge/infrastructure/db/sqlite_connector.py b/queryforge/infrastructure/db/sqlite_connector.py index a797bb2..88a6f33 100644 --- a/queryforge/infrastructure/db/sqlite_connector.py +++ b/queryforge/infrastructure/db/sqlite_connector.py @@ -4,7 +4,6 @@ import sqlite3 from pathlib import Path -from .adapters import AdapterCapabilities from queryforge.core.schemas.models import ( ExecutionResult, @@ -12,6 +11,7 @@ TableColumn, TableSchema, ) +from queryforge.infrastructure.db.adapter import capabilities_for_dialect class SQLiteConnectorError(RuntimeError): @@ -22,7 +22,13 @@ class SQLiteConnector: """Open one existing SQLite database with read-only enforcement.""" dialect = "sqlite" - capabilities = AdapterCapabilities("sqlite") + #: Taken from the single frozen declaration point, not built from dataclass + #: defaults: ``AdapterCapabilities("sqlite")`` silently disagreed with + #: ``SQLITE_CAPABILITIES`` on ``explain_prefix`` ("EXPLAIN" instead of + #: "EXPLAIN QUERY PLAN"), ``date_functions`` and ``integer_division``, so a + #: reader of this attribute got a different capability set than the adapter + #: actually enforced (E-23). + capabilities = capabilities_for_dialect("sqlite") def __init__(self, database_path: str) -> None: self.database_path = Path(database_path).expanduser().resolve() diff --git a/queryforge/infrastructure/models/base.py b/queryforge/infrastructure/models/base.py index 6a31efe..031681a 100644 --- a/queryforge/infrastructure/models/base.py +++ b/queryforge/infrastructure/models/base.py @@ -24,6 +24,18 @@ def __init__(self, message: str, raw_output: str) -> None: self.raw_output = raw_output +def _timeout_kwarg(timeout: float | None) -> dict[str, float]: + """``{"timeout": ...}`` only when a deadline was actually declared. + + Subclasses and test doubles commonly override ``generate_with_messages`` with + the historical two-argument signature. Passing ``timeout=None`` + unconditionally would break every one of them, so the keyword is added only + when there is a real deadline to communicate. + """ + + return {} if timeout is None else {"timeout": float(timeout)} + + class BaseModelProvider(ABC): """Small interface shared by every QueryForge provider adapter.""" @@ -47,15 +59,20 @@ def record_usage(self, raw_usage: Any) -> ModelUsage | None: self.last_usage = usage return usage - def generate_text(self, prompt: str) -> str: + def generate_text( + self, prompt: str, timeout: float | None = None + ) -> str: return self.generate_with_messages( [ {"role": "system", "content": "You are a careful assistant."}, {"role": "user", "content": prompt}, - ] + ], + **_timeout_kwarg(timeout), ) - def generate_json(self, prompt: str) -> dict[str, Any]: + def generate_json( + self, prompt: str, timeout: float | None = None + ) -> dict[str, Any]: raw_output = self.generate_with_messages( [ { @@ -65,6 +82,7 @@ def generate_json(self, prompt: str) -> dict[str, Any]: {"role": "user", "content": prompt}, ], json_mode=True, + **_timeout_kwarg(timeout), ) candidate = self._extract_json_object(raw_output) try: @@ -84,9 +102,19 @@ def generate_json(self, prompt: str) -> dict[str, Any]: @abstractmethod def generate_with_messages( - self, messages: list[Message], json_mode: bool = False + self, + messages: list[Message], + json_mode: bool = False, + timeout: float | None = None, ) -> str: - """Generate text from normalized role/content messages.""" + """Generate text from normalized role/content messages. + + ``timeout`` is the remaining wall-clock budget for this call in seconds, or + ``None`` when the run declared no deadline. Adapters that can bound a + request should honour it; the parameter exists so the run's remaining + deadline can actually reach the client instead of being recorded and then + ignored (``BudgetLimits.model_deadline_ms`` used to have no consumer). + """ @staticmethod def _extract_json_object(raw_output: str) -> str: diff --git a/queryforge/infrastructure/models/providers/openai_compatible.py b/queryforge/infrastructure/models/providers/openai_compatible.py index 97fe4a8..dbc10c1 100644 --- a/queryforge/infrastructure/models/providers/openai_compatible.py +++ b/queryforge/infrastructure/models/providers/openai_compatible.py @@ -30,13 +30,21 @@ def client_options(self) -> dict[str, Any]: return {} def generate_with_messages( - self, messages: list[Message], json_mode: bool = False + self, + messages: list[Message], + json_mode: bool = False, + timeout: float | None = None, ) -> str: request: dict[str, Any] = { "model": self.model, "messages": messages, "temperature": 0.1, } + if timeout is not None: + # Bound the request by the run's remaining deadline. A tiny positive + # floor avoids asking the SDK for an immediate timeout, which would + # turn "barely any time left" into a misleading instant failure. + request["timeout"] = max(0.1, float(timeout)) if json_mode and self.supports_response_format: request["response_format"] = {"type": "json_object"} # Step 14: a fresh call starts with no measured usage, so a response that diff --git a/queryforge/infrastructure/storage/knowledge_base.py b/queryforge/infrastructure/storage/knowledge_base.py index d6c7e34..772c2fb 100644 --- a/queryforge/infrastructure/storage/knowledge_base.py +++ b/queryforge/infrastructure/storage/knowledge_base.py @@ -9,6 +9,7 @@ from pathlib import Path from typing import Any, Iterable, Sequence +from queryforge.core.paths import workspace_root from queryforge.core.schemas.models import SQLContext, TableSchema from queryforge.domain.knowledge import ( GovernedDocument, @@ -27,7 +28,7 @@ SQL_SOURCE_TYPES = ("sql_history", "reference_sql", "reference_template", "success_story") -PROJECT_ROOT = Path(__file__).resolve().parents[3] +PROJECT_ROOT = workspace_root() #: Gold/evaluation tasks whose questions and reference SQL must never enter the #: knowledge base that answers them: indexing the holdout set would invalidate the #: benchmark it belongs to, because a few-shot example could then be the answer @@ -677,6 +678,16 @@ def _csv_documents(path: Path) -> list[VectorDocument]: continue explanation = (row.get("evidence") or "").strip() tables = SQLHistoryStore.extract_tables(sql) + # Trust is gated on a named reviewer, matching + # ``SQLHistoryStore.import_success_stories``. This path used to + # stamp every row ``human_reviewed`` unconditionally, so any CSV + # with question/sql columns entered the knowledge base as trusted + # few-shot material without anyone having reviewed it — the + # opposite of what the three-tier verification model in + # ``domain/knowledge/governance.py`` exists for. + reviewer = str( + row.get("reviewer") or row.get("reviewed_by") or "" + ).strip() documents.append( VectorDocument.create( id=KnowledgeBaseBuilder._id("success_story", str(path), str(index), question, sql), @@ -691,9 +702,16 @@ def _csv_documents(path: Path) -> list[VectorDocument]: "explanation": explanation, "tables_used": tables, "row": row, - "verification_level": VerificationLevel.human_reviewed.value, - "review_status": "reviewed", - "owner": str(row.get("owner") or "business"), + # Unreviewed rows stay execution_success: the SQL ran, + # which says nothing about business correctness. + "verification_level": ( + VerificationLevel.human_reviewed.value + if reviewer + else VerificationLevel.execution_success.value + ), + "review_status": "reviewed" if reviewer else "draft", + "reviewer": reviewer or None, + "owner": reviewer or str(row.get("owner") or "business"), "version": row.get("version"), }, ) diff --git a/queryforge/infrastructure/storage/sql_history_store.py b/queryforge/infrastructure/storage/sql_history_store.py index 8bb17cf..98ef1a9 100644 --- a/queryforge/infrastructure/storage/sql_history_store.py +++ b/queryforge/infrastructure/storage/sql_history_store.py @@ -13,6 +13,7 @@ from pathlib import Path from typing import Any, Iterable +from queryforge.core.paths import workspace_root from queryforge.core.schemas.models import HistoryMatch from queryforge.domain.knowledge import ( VerificationLevel, @@ -22,7 +23,7 @@ from queryforge.infrastructure.tools.database_tool import DatabaseTool, UnsafeSQLError -PROJECT_ROOT = Path(__file__).resolve().parents[3] +PROJECT_ROOT = workspace_root() DEFAULT_HISTORY_DB_PATH = PROJECT_ROOT / ".queryforge/history.db" #: Hard bound on the rows one search may scan. Similarity is scored in Python, so #: the candidate window must be bounded; it is explicit (constructor override) and @@ -548,7 +549,7 @@ def import_success_stories(self, csv_path: str | Path) -> ImportSummary: skipped += 1 continue try: - sql = DatabaseTool.validate_readonly_sql(sql) + sql = DatabaseTool.validate_readonly_shape(sql) except UnsafeSQLError: skipped += 1 continue @@ -604,7 +605,7 @@ def import_reference_sql(self, source_path: str | Path) -> ImportSummary: raise SQLHistoryError(f"Could not read reference SQL {file}: {exc}") from exc for index, (comments, sql) in enumerate(self._parse_reference_text(text), 1): try: - sql = DatabaseTool.validate_readonly_sql(sql) + sql = DatabaseTool.validate_readonly_shape(sql) except UnsafeSQLError: skipped += 1 continue diff --git a/queryforge/infrastructure/storage/vector_store.py b/queryforge/infrastructure/storage/vector_store.py index d5315bf..4bcf9be 100644 --- a/queryforge/infrastructure/storage/vector_store.py +++ b/queryforge/infrastructure/storage/vector_store.py @@ -12,7 +12,10 @@ from typing import Any, Iterable, Mapping, Protocol, Sequence -PROJECT_ROOT = Path(__file__).resolve().parents[3] +from queryforge.core.paths import workspace_root + + +PROJECT_ROOT = workspace_root() DEFAULT_VECTOR_KB_PATH = PROJECT_ROOT / ".queryforge/lancedb" SQL_HISTORY_VECTORS = "sql_history_vectors" SCHEMA_DOC_VECTORS = "schema_doc_vectors" diff --git a/queryforge/infrastructure/tools/database_tool.py b/queryforge/infrastructure/tools/database_tool.py index 3194eb9..fd25018 100644 --- a/queryforge/infrastructure/tools/database_tool.py +++ b/queryforge/infrastructure/tools/database_tool.py @@ -46,7 +46,29 @@ def __init__( source_path=policy_source_path, dialect=self.dialect, ) - self.last_policy_decision: SqlPolicyDecision | None = None + self._last_policy_decision: SqlPolicyDecision | None = None + + @property + def last_policy_decision(self) -> SqlPolicyDecision | None: + """The decision from the most recent call made through *this tool*. + + Read-only on purpose. It used to be a plain attribute, which two callers + exploited: the planner node *assigned* a decision it had computed itself, + so the tool reported an audit record for a call that never happened, and + ``PlanOutputNode`` read it as a fallback for a decision it had just failed + to obtain (E-23). Only :meth:`_record_policy_decision` mutates it now. + + It is still per-instance state, so it is only meaningful immediately after + a call on this instance. Prefer the value returned by that call wherever a + decision has to be attributed: shared tools (the planner's and the + registry's) can be driven concurrently, and a decision read back later can + belong to another caller's statement. + """ + return self._last_policy_decision + + def _record_policy_decision(self, decision: SqlPolicyDecision | None) -> None: + """Record the decision for the call currently in flight.""" + self._last_policy_decision = decision @property def policy_summary(self) -> dict: @@ -61,7 +83,7 @@ def describe_table(self, table_name: str) -> TableSchema: self.connector.describe_table(table_name) ) except SQLPolicyViolation as exc: - self.last_policy_decision = exc.decision + self._record_policy_decision(exc.decision) raise UnsafeSQLError(str(exc), exc.decision) from exc def describe_table_for_validation(self, table_name: str) -> TableSchema: @@ -93,13 +115,26 @@ def execute_sql(self, sql: str) -> ExecutionResult: try: decision = self.policy_engine.evaluate(sql) except SQLPolicyViolation as exc: - self.last_policy_decision = exc.decision + self._record_policy_decision(exc.decision) raise UnsafeSQLError(str(exc), exc.decision) from exc - self.last_policy_decision = decision + self._record_policy_decision(decision) return self.connector.execute_sql(sql.strip()) def execute_sql_preview(self, sql: str, limit: int = 20) -> ExecutionResult: - """Execute a bounded read-only preview through the same policy engine.""" + """Execute a bounded read-only preview through the same policy engine. + + The policy engine is fed the **original** statement, before the preview + LIMIT is injected. That ordering matters: evaluating the rewritten text + meant the engine saw a LIMIT that the model never wrote, so a policy with + ``require_limit`` was satisfied by construction and its audit record + disagreed with what was actually enforced. + + ``unbounded_result`` is the one refusal that preview tolerates, because an + unbounded preview is the point — the injected LIMIT bounds it. Every other + refusal (table scope, column scope, read-only, dangerous functions, + ``max_limit``) still applies and still raises. + """ + if not isinstance(limit, int) or limit < 1: raise ValueError("preview limit must be a positive integer") bounded_limit = min(limit, 100) @@ -107,18 +142,33 @@ def execute_sql_preview(self, sql: str, limit: int = 20) -> ExecutionResult: tree = sqlglot.parse_one(sql, read=self.dialect) if tree is None or not tree.find(sqlglot.exp.Select): raise UnsafeSQLError("preview requires a SELECT query") - if tree.args.get("limit") is None: - tree = tree.limit(bounded_limit) - else: - existing = tree.args["limit"] - literal = existing.expression - current = int(literal.this) if literal and literal.is_int else bounded_limit - existing.set("expression", sqlglot.exp.Literal.number(min(current, bounded_limit))) - bounded_sql = tree.sql(dialect=self.dialect) except UnsafeSQLError: raise except Exception as exc: raise UnsafeSQLError(f"preview SQL could not be parsed: {exc}") from exc + + try: + self.policy_engine.evaluate(sql) + except SQLPolicyViolation as exc: + if getattr(exc.decision, "rule", None) != "unbounded_result": + self._record_policy_decision(exc.decision) + raise UnsafeSQLError(str(exc), exc.decision) from exc + # Tolerated for preview only. Recorded so the audit shows the engine + # was consulted on the original statement and why it was allowed. + self._record_policy_decision(exc.decision) + + if tree.args.get("limit") is None: + tree = tree.limit(bounded_limit) + else: + existing = tree.args["limit"] + literal = existing.expression + current = int(literal.this) if literal and literal.is_int else bounded_limit + existing.set( + "expression", sqlglot.exp.Literal.number(min(current, bounded_limit)) + ) + bounded_sql = tree.sql(dialect=self.dialect) + # ``execute_sql`` re-evaluates the bounded statement, which is what the + # engine actually runs; both verdicts are on the record now. return self.execute_sql(bounded_sql) def preview_distinct_values( @@ -150,8 +200,30 @@ def _quote_identifier(value: str) -> str: return f'"{value}"' @staticmethod - def validate_readonly_sql(sql: str) -> str: - """Compatibility API now backed by the same SQLGlot AST checks.""" + def validate_readonly_shape(sql: str) -> str: + """Refuse SQL that is not a single read-only statement. **Not** authorization. + + This check runs a :class:`SQLPolicyEngine` with an **empty schema** and a + default policy, so it enforces only what does not depend on knowing the + database: + + * exactly one statement, + * a read-only AST root (no INSERT/UPDATE/DDL/ATTACH/PRAGMA, no recursive CTE), + * no dangerous function, and + * ``LIMIT`` literals when the default policy asks for them. + + It cannot enforce ``allowed_tables``, ``allowed_columns`` or table scope, + because it is given no schema to compare against — and with an empty schema + the engine deliberately lets an unknown table through (see + ``SQLPolicyEngine._validate_scopes``). An earlier name, ``validate_readonly_sql``, + invited callers to treat it as a security gate; ``sql_history_store`` used it + as the *only* gate on its import path, which is why the name is now explicit + about what it does and does not do. + + For authorization use :meth:`execute_sql` (or a :class:`DatabaseTool` bound + to a real schema and the deployment's policy), which runs the full engine. + """ + if not isinstance(sql, str) or not sql.strip(): raise UnsafeSQLError( "SQL_SECURITY_ERROR run_id=- rule=ast_parse: empty SQL query" @@ -162,3 +234,14 @@ def validate_readonly_sql(sql: str) -> str: except SQLPolicyViolation as exc: raise UnsafeSQLError(str(exc), exc.decision) from exc return sql.strip() + + @staticmethod + def validate_readonly_sql(sql: str) -> str: + """Deprecated alias for :meth:`validate_readonly_shape`. + + Kept so existing callers keep working; new code should call the explicit + name so the difference from an authorization gate is visible at the call + site. + """ + + return DatabaseTool.validate_readonly_shape(sql) diff --git a/queryforge/orchestration/agents/entry_router.py b/queryforge/orchestration/agents/entry_router.py index ed26d42..6429f0d 100644 --- a/queryforge/orchestration/agents/entry_router.py +++ b/queryforge/orchestration/agents/entry_router.py @@ -4,6 +4,7 @@ import re from dataclasses import dataclass +from functools import lru_cache from queryforge.orchestration.schemas import RoutingDecision, TaskType @@ -128,8 +129,38 @@ def _classify(self, text: str) -> tuple[TaskType, float, str]: return "ask_sql", 0.75, "Defaulted a data question to the ask_sql pipeline." @staticmethod - def _contains(text: str, markers: tuple[str, ...]) -> bool: - return any(marker in text for marker in markers) + @lru_cache(maxsize=256) + def _marker_regex(marker: str) -> "re.Pattern[str]": + """Match a marker with arbitrary whitespace between its characters. + + ``route`` normalises the input with ``" ".join(text.split())``, which + collapses runs of whitespace but cannot remove a single space inside a + Chinese phrase. The marker tables were therefore internally inconsistent: + ``_SQL_REVIEW_MARKERS`` and ``_TROUBLESHOOT_MARKERS`` spelled each Chinese + marker twice ("审核sql" and "审核 sql") while ``_REPORT_MARKERS``, + ``_METADATA_MARKERS`` and ``_EXPLAIN_MARKERS`` listed only the unspaced + form — so "生成 报告" fell through to the default ``ask_sql`` while + "审核 sql" worked. Spelling every variant twice is unbounded (two spaces? + three?), so the separator is made whitespace-tolerant instead. + """ + + pattern = r"\s*".join(re.escape(char) for char in marker) + # An ASCII marker must match on a word boundary, or a bare keyword inside a + # longer identifier routes the whole request: "report" matched + # "sales_report", so "Show the first 10 rows of sales_report" was classified + # as a report-building task and the run skipped straight to report + # generation. CJK has no such boundary concept, so the anchors are applied + # only when the marker both starts and ends with an ASCII word character — + # which also leaves phrases like "sql review" matching as written. + ends_word = re.search(r"\w$", marker) is not None + starts_word = re.match(r"\w", marker) is not None + if marker.isascii() and starts_word and ends_word: + pattern = rf"\b{pattern}\b" + return re.compile(pattern) + + @classmethod + def _contains(cls, text: str, markers: tuple[str, ...]) -> bool: + return any(cls._marker_regex(marker).search(text) for marker in markers) @classmethod def _complexity( diff --git a/queryforge/orchestration/agents/product_analyst.py b/queryforge/orchestration/agents/product_analyst.py index b406e97..dec852e 100644 --- a/queryforge/orchestration/agents/product_analyst.py +++ b/queryforge/orchestration/agents/product_analyst.py @@ -48,20 +48,35 @@ class ProductAnalystAgent(RoleAgent): r"\b(?:growth|increase|decrease|compare|comparison|versus|vs|yoy|mom)\b|增长|下降|对比|比较|同比|环比", re.IGNORECASE, ) + # Older revisions of these patterns required whitespace after the marker, + # which works for Latin input ("by region") but silently failed for the most + # natural Chinese phrasing: "按地区" is one unspaced run, so it never matched + # while "按 地区" did. The same file already used the space-optional form in + # _TIME_FOLLOWUP / _TOP_FOLLOWUP, so the two conventions were inconsistent + # and the follow-up was lost entirely — the router then treated a bare + # follow-up as a fresh question and the model invented its own metrics. + # + # A single optional-whitespace separator now covers both scripts, and every + # pattern captures its payload in the named group "rest" so an unmatched + # alternative can never raise on a missing group. _BREAKDOWN_FOLLOWUP = re.compile( - r"^(?:by|per|break(?:\s+it)?\s+down\s+by|按|按照)\s+(.+?)(?:\s*(?:again|再|一下|呢))?$", + r"^(?:by|per|break(?:\s+it)?\s+down\s+by|按|按照)\s*" + r"(?P.+?)(?:\s*(?:again|再|一下|呢))?$", re.IGNORECASE, ) _FILTER_FOLLOWUP = re.compile( - r"^(?:only(?:\s+(?:show|include|look at))?|filter(?:\s+to)?|只看|仅看|只保留)\s+(.+)$", + r"^(?:only(?:\s+(?:show|include|look\s+at))?|filter(?:\s+to)?|只看|仅看|只保留)\s*" + r"(?P.+?)(?:\s*(?:again|再|一下|呢))?$", re.IGNORECASE, ) _ADD_FOLLOWUP = re.compile( - r"^(?:also\s+(?:include|add)|add|再加上|加上)\s+(.+)$", + r"^(?:also\s+(?:include|add)|add|再加上|加上)\s*" + r"(?P.+?)(?:\s*(?:again|再|一下|呢))?$", re.IGNORECASE, ) _REMOVE_FOLLOWUP = re.compile( - r"^(?:remove|drop|去掉|移除)\s+(.+)$", + r"^(?:remove|drop|去掉|移除|删除)\s*" + r"(?P.+?)(?:\s*(?:again|再|一下|呢))?$", re.IGNORECASE, ) _TOP_FOLLOWUP = re.compile( @@ -73,7 +88,37 @@ class ProductAnalystAgent(RoleAgent): re.IGNORECASE, ) _REFERENCE_FOLLOWUP = re.compile( - r"\b(?:that result|previous result|same result|刚才那个|那个结果|上一个结果)\b", + # \b cannot anchor CJK: 汉 characters are word characters in Python's + # Unicode re, so "\b刚才那个\b" only matched when the phrase happened to + # sit between non-word characters. That made "那个结果再按地区" (where the + # phrase is followed by another Han character) fail while "刚才那个" + # alone succeeded. Split the alternatives so Latin keeps \b and CJK uses + # no boundary assertion at all. + r"\b(?:that result|previous result|same result)\b" + r"|(?:刚才那个|那个结果|上一个结果|上一条结果)", + re.IGNORECASE, + ) + #: A bare request to repeat the previous query, with no new content. This is + #: the shape the benchmark's only multi-turn case uses ("再查一次"), and it + #: used to fall through to the router as a brand-new question. + _REPEAT_FOLLOWUP = re.compile( + r"^(?:再|重新|重|又)?\s*(?:查|查询|跑|执行|算|来|看)\s*(?:一遍|一次|一下|一回|下)?$" + r"|^(?:再来|重来)$", + re.IGNORECASE, + ) + #: Verbs that make the captured payload a complete instruction rather than a + #: fragment to attach to the previous request. Dropping the required space + #: after the CJK markers must not turn "按门店统计订单数" (a full question) + #: into a follow-up. + #: + #: Only words that are rare as *metric or dimension names* belong here. A + #: first attempt also listed count/total/average/sum/query/list, which broke a + #: legitimate follow-up ("also include order count") because "count" is a + #: perfectly ordinary metric name — the guard must not be more eager than the + #: pattern it guards. + _PAYLOAD_ACTION = re.compile( + r"统计|查询|计算|分析|汇总|列出|找出|求和|排名|排序|占比" + r"|\b(?:compute|calculate|compare)\b", re.IGNORECASE, ) @@ -101,35 +146,35 @@ def rewrite_followup( "add_time_dimension", ) breakdown = cls._BREAKDOWN_FOLLOWUP.match(original) - if breakdown: + if breakdown and not cls._payload_is_an_instruction(breakdown): return cls._rewrite( original, previous, - f"Break down the result by {breakdown.group(1).strip()}.", + f"Break down the result by {breakdown.group('rest').strip()}.", "add_dimension", ) filtered = cls._FILTER_FOLLOWUP.match(original) - if filtered: + if filtered and not cls._payload_is_an_instruction(filtered): return cls._rewrite( original, previous, - f"Only include {filtered.group(1).strip()}.", + f"Only include {filtered.group('rest').strip()}.", "add_filter", ) added = cls._ADD_FOLLOWUP.match(original) - if added: + if added and not cls._payload_is_an_instruction(added): return cls._rewrite( original, previous, - f"Also include {added.group(1).strip()} as an additional metric or field.", + f"Also include {added.group('rest').strip()} as an additional metric or field.", "add_metric", ) removed = cls._REMOVE_FOLLOWUP.match(original) - if removed: + if removed and not cls._payload_is_an_instruction(removed): return cls._rewrite( original, previous, - f"Remove {removed.group(1).strip()} from the grouping or requested metrics.", + f"Remove {removed.group('rest').strip()} from the grouping or requested metrics.", "remove_dimension_or_metric", ) top = cls._TOP_FOLLOWUP.match(original) @@ -148,12 +193,37 @@ def rewrite_followup( original, "resolve_reference", ) + if cls._REPEAT_FOLLOWUP.match(original): + # "再查一次" carries no new content: unlike the other branches there + # is nothing to add, so the previous request is re-issued verbatim + # instead of being passed through as a fresh question. + return cls._rewrite( + original, + previous, + "Repeat the previous request exactly.", + "repeat_previous_request", + ) return { "question": original, "is_followup": False, "reason": None, } + @classmethod + def _payload_is_an_instruction(cls, match: "re.Match[str]") -> bool: + """Whether a marker's captured payload is itself a complete instruction. + + The CJK markers now accept an optional space, so "按门店统计订单数" matches + the breakdown pattern. Its payload carries its own verb, which means the + text is a standalone question rather than a fragment to attach to the + previous request — treating it as a follow-up would silently discard the + user's new instruction. Returns True when the payload must NOT be treated + as a follow-up. + """ + + payload = (match.group("rest") or "").strip() + return bool(payload) and bool(cls._PAYLOAD_ACTION.search(payload)) + @staticmethod def _rewrite( original: str, diff --git a/queryforge/orchestration/orchestrator/orchestrator.py b/queryforge/orchestration/orchestrator/orchestrator.py index 9e47234..4b76229 100644 --- a/queryforge/orchestration/orchestrator/orchestrator.py +++ b/queryforge/orchestration/orchestrator/orchestrator.py @@ -19,6 +19,7 @@ from queryforge.orchestration.agents.visualization import VisualizationAgent from queryforge.orchestration.quality import append_warning from queryforge.orchestration.gates import QualityGateEvaluator +from queryforge.core.outcomes import TerminalOutcome, derive_outcome from queryforge.orchestration.runtime.session_store import SessionStore from queryforge.orchestration.runtime.state_store import AgentTeamStateStore from queryforge.orchestration.schemas import DeliveryReport, RoutingDecision, TaskState, utc_now @@ -39,6 +40,66 @@ DirectRun = Callable[[TaskState, "OrchestratorAgent"], dict[str, Any]] +#: Canonical terminal outcome -> the value persisted in ``TaskState.status``. +#: ``succeeded`` is spelled ``completed`` in the persisted vocabulary for backward +#: compatibility with existing ``state.json`` files and readers. +_TASK_STATUS_FOR_OUTCOME: dict[TerminalOutcome, str] = { + "succeeded": "completed", + "needs_clarification": "needs_clarification", + "blocked": "blocked", + "partial": "blocked", + "failed": "failed", + "cancelled": "cancelled", +} + + +def _phase_for( + outcome: TerminalOutcome, state: Any, blocked_phase: str | None +) -> str: + if outcome == "needs_clarification": + return "clarification" + if outcome in {"blocked", "partial"}: + return blocked_phase or "blocked" + if outcome == "cancelled": + return "cancelled" + if outcome == "failed": + return "failed" + return "delivery" + + +_PHASE_FOR_OUTCOME = { + outcome: (lambda state, blocked_phase, _o=outcome: _phase_for(_o, state, blocked_phase)) + for outcome in ( + "succeeded", + "needs_clarification", + "blocked", + "partial", + "failed", + "cancelled", + ) +} + +#: Delivery report status. A clarification and a block are both "degraded" rather +#: than "success", because neither delivered an answer. +_DELIVERY_STATUS_FOR_OUTCOME: dict[TerminalOutcome, str] = { + "succeeded": "success", + "needs_clarification": "degraded", + "blocked": "degraded", + "partial": "degraded", + "failed": "failed", + "cancelled": "failed", +} + +_SUMMARY_FOR_OUTCOME: dict[TerminalOutcome, str] = { + "succeeded": "The request completed through the integrated Agent Team workflow.", + "needs_clarification": "The request stopped for user clarification.", + "blocked": "The request was blocked by a quality gate.", + "partial": "The request completed partially; see the artifacts for what is missing.", + "failed": "The integrated Agent Team workflow failed.", + "cancelled": "The run was cancelled before it finished.", +} + + class OrchestratorAgent: """Coordinate role agents at explicit lifecycle points in WorkflowRunner.""" @@ -210,13 +271,23 @@ def run( self.state_store.save_state(state) raise - result_status = str(result.get("status") or "success") - if result_status == "blocked": - state.status = "blocked" - state.current_phase = state.blocked_phase or "blocked" - else: - state.status = "completed" - state.current_phase = "delivery" + # One derivation, one vocabulary (queryforge.core.outcomes). + # + # This block previously re-derived the status from the workflow's result + # dict and ignored a block the completion hook had already recorded, which + # is how defect E-02 happened: the QA gate set state.status = "blocked" and + # this code then overwrote it to "completed", so state.json and the caller + # disagreed. A recorded block now wins over the result status, and the + # persisted status is mapped onto the canonical vocabulary. + outcome = derive_outcome( + result_status=result.get("status"), + blocked_reason=state.blocked_reason, + cancelled=state.status == "cancelled", + ) + state.status = _TASK_STATUS_FOR_OUTCOME[outcome] + state.current_phase = _PHASE_FOR_OUTCOME[outcome]( + state, state.blocked_phase + ) self._start_phase(state, "delivery") self._complete_phase(state, "delivery") state.pending_phases = [] @@ -226,17 +297,10 @@ def run( run_id=run_id, task_id=state.task_id, task_type=decision.task_type, - status=( - "degraded" - if result_status == "blocked" - else - result_status - if result_status in {"planned", "success", "degraded", "failed"} - else "success" - ), + status=_DELIVERY_STATUS_FOR_OUTCOME[outcome], pipeline=list(pipeline), artifact_refs=list(state.artifacts), - summary="The request completed through the integrated Agent Team workflow.", + summary=_SUMMARY_FOR_OUTCOME[outcome], ) self._write_delivery(state, report) session = self._record_session_turn( diff --git a/queryforge/orchestration/planner/executor.py b/queryforge/orchestration/planner/executor.py index aa2ab9d..5b30056 100644 --- a/queryforge/orchestration/planner/executor.py +++ b/queryforge/orchestration/planner/executor.py @@ -251,6 +251,10 @@ class AnalysisExecutionResult(BaseModel): status: str = "pending" replan_reasons: list[str] = Field(default_factory=list) budgets: dict[str, Any] = Field(default_factory=dict) + #: Usage inherited from an earlier attempt of this same run, when resuming. + #: ``budgets.usage`` is cumulative across attempts; this field lets a reader + #: tell inherited consumption from what this attempt spent. + inherited_usage: dict[str, Any] = Field(default_factory=dict) stop_reason: str | None = None reused_steps: list[str] = Field(default_factory=list) recomputed_steps: list[str] = Field(default_factory=list) @@ -280,6 +284,7 @@ def __init__( max_steps: int = 32, clock: Callable[[], float] = time.monotonic, journal: ExecutionJournal | None = None, + run_versions: dict[str, Any] | None = None, worker_id: str = "worker", lease_ttl_seconds: float = 60.0, cancel_check: Callable[[], bool] | None = None, @@ -301,6 +306,10 @@ def __init__( self.max_steps = max_steps self._clock = clock self.journal = journal + #: Data / semantic / policy versions this run binds to, threaded into every + #: step fingerprint so a resume cannot reuse a result computed against a + #: different database snapshot or a changed semantic model. + self.run_versions: dict[str, Any] = dict(run_versions or {}) self.worker_id = worker_id self.lease_ttl_seconds = max(float(lease_ttl_seconds), 1.0) self.cancel_check = cancel_check @@ -499,6 +508,14 @@ def execute( ) state = _ExecutionState(plan=plan, generated_by=self.budget_manager) + # Remembered on the state so the payload can distinguish "spent by this + # attempt" from "inherited from the earlier attempt" — a bare cumulative + # figure cannot express the difference. + state.inherited_usage = dict( + (self.journal.budget() or {}).get("usage") or {} + if self.journal is not None + else {} + ) if self.journal is not None: if self.journal.journal.terminal() and not self.force_resume: raise RunNotResumable( @@ -506,7 +523,7 @@ def execute( f"{self.journal.journal.terminal_outcome!r}; refusing to revive it" ) self.journal.expire_leases() - self.journal.register_plan(plan) + self.journal.register_plan(plan, versions=self.run_versions) state.reuse_allowed, state.reuse_denied = self._reuse_plan(plan) plan.status = "running" base_context = self._base_context(tool_context) @@ -636,7 +653,10 @@ def _reuse_plan( assert self.journal is not None for step in PlanValidator.topological_order(plan): fingerprint = self.journal.fingerprint_step( - step.action, dict(step.inputs or {}), plan_version=plan.version + step.action, + dict(step.inputs or {}), + plan_version=plan.version, + versions=self.journal.versions, ) blocker = self._reuse_blocker(step) if blocker is not None: @@ -808,7 +828,10 @@ def _run_step( context = self._step_context(base_context, step, state) if self.journal is not None: fingerprint = self.journal.fingerprint_step( - step.action, dict(step.inputs or {}), plan_version=state.plan.version + step.action, + dict(step.inputs or {}), + plan_version=state.plan.version, + versions=self.journal.versions, ) if state.reuse_allowed.get(step_id): reused = self._reuse_step(step, step_id, context, state, result) @@ -2008,6 +2031,8 @@ def __init__(self, plan: AnalysisPlan, generated_by: BudgetManager) -> None: step.id: StepResult(step_id=step.id, action=step.action) for step in plan.steps } self.evidence: list[dict[str, Any]] = [] + #: Budget usage carried over from an earlier attempt of the same run. + self.inherited_usage: dict[str, Any] = {} self.evidence_by_kind: dict[str, str] = {} self.evidence_payloads: dict[str, dict[str, Any]] = {} self.answer: dict[str, Any] | None = None @@ -2108,6 +2133,7 @@ def result(self, budget_manager: BudgetManager) -> AnalysisExecutionResult: status=self.plan.status, replan_reasons=self.replan_reasons, budgets=budget_manager.snapshot(), + inherited_usage=dict(self.inherited_usage), stop_reason=self.stop_reason, ) diff --git a/queryforge/orchestration/runtime/execution_journal.py b/queryforge/orchestration/runtime/execution_journal.py index f7ad976..136687e 100644 --- a/queryforge/orchestration/runtime/execution_journal.py +++ b/queryforge/orchestration/runtime/execution_journal.py @@ -12,9 +12,11 @@ import hashlib import json +import logging import os import threading import time +from contextlib import contextmanager from enum import Enum from pathlib import Path from typing import Any, Callable, Literal @@ -22,6 +24,9 @@ from pydantic import BaseModel, Field +LOGGER = logging.getLogger("queryforge.execution_journal") + + JOURNAL_SCHEMA_VERSION = "1.0" TERMINAL_OUTCOMES = ("success", "partial", "blocked", "failed", "cancelled") @@ -132,6 +137,9 @@ class RunJournal(BaseModel): created_at: str = "" updated_at: str = "" budget: dict[str, Any] = Field(default_factory=dict) + #: Data / semantic / policy versions this run was recorded against. Persisted so + #: a resume recomputes steps whose inputs now resolve differently. + versions: dict[str, str] = Field(default_factory=dict) steps: dict[str, StepRecord] = Field(default_factory=dict) notes: list[str] = Field(default_factory=list) @@ -156,6 +164,10 @@ def __init__( self._lock = threading.RLock() self._run_id = run_id or self.run_dir.name self._journal = self._load_or_create() + #: The version set this journal binds to. Set by ``register_plan`` and read + #: by every later fingerprint computation so a resume cannot recompute a + #: step against a different version than it was originally recorded under. + self.versions: dict[str, Any] = dict(self._journal.versions) # ------------------------------------------------------------------ paths @@ -204,17 +216,53 @@ def save(self) -> Path: # ------------------------------------------------------------------ plan - @staticmethod - def fingerprint_step(action: str, inputs: dict[str, Any], *, plan_version: int) -> str: - payload = json.dumps( - {"action": action, "inputs": inputs, "plan_version": plan_version}, - ensure_ascii=False, - sort_keys=True, - default=str, - ) - return hashlib.sha256(payload.encode("utf-8")).hexdigest() - - def register_plan(self, plan: Any) -> list[str]: + #: Version dimensions a step fingerprint must bind, beyond the action itself. + #: + #: A fingerprint that covers only ``action``/``inputs``/``plan_version`` says + #: "the same request" while ignoring the data and definitions it ran against, so + #: a resume after the database or the semantic model changed would reuse a + #: result computed over different inputs. ``DomainContext`` and ``RunContext`` + #: already carry all three versions; this is where they start mattering. + FINGERPRINT_VERSIONS: tuple[str, ...] = ( + "data_version", + "semantic_version", + "policy_version", + ) + + @classmethod + def fingerprint_step( + cls, + action: str, + inputs: dict[str, Any], + *, + plan_version: int, + versions: dict[str, Any] | None = None, + ) -> str: + payload: dict[str, Any] = { + "action": action, + "inputs": inputs, + "plan_version": plan_version, + } + if versions: + # Only the declared dimensions, and only when actually supplied: an + # undeclared version is "unknown", which is different from "unchanged", + # so it must not silently equal a known value. + declared = { + key: str(versions[key]) + for key in cls.FINGERPRINT_VERSIONS + if versions.get(key) not in (None, "") + } + if declared: + payload["versions"] = declared + return hashlib.sha256( + json.dumps( + payload, ensure_ascii=False, sort_keys=True, default=str + ).encode("utf-8") + ).hexdigest() + + def register_plan( + self, plan: Any, *, versions: dict[str, Any] | None = None + ) -> list[str]: """Register (or refresh) every step of ``plan``; returns changed step ids. A step whose fingerprint changed (new inputs or a new plan version) is @@ -228,7 +276,10 @@ def register_plan(self, plan: Any) -> list[str]: self._journal.plan_version = max(self._journal.plan_version, version) for step in getattr(plan, "steps", []): fingerprint = self.fingerprint_step( - step.action, dict(step.inputs or {}), plan_version=version + step.action, + dict(step.inputs or {}), + plan_version=version, + versions=versions, ) existing = self._journal.steps.get(step.id) if existing is None: @@ -253,6 +304,13 @@ def register_plan(self, plan: Any) -> list[str]: existing.outcome_certain = True existing.plan_version = version changed.append(step.id) + if versions: + self._journal.versions = { + key: str(versions[key]) + for key in self.FINGERPRINT_VERSIONS + if versions.get(key) not in (None, "") + } + self.versions = dict(self._journal.versions) self.save() return changed @@ -352,28 +410,107 @@ def record_failure( record.outcome_certain = False self.save() + # ------------------------------------------------------- cross-instance lock + # + # The in-process ``threading.RLock`` only serialises access within one + # ``ExecutionJournal`` object. Two instances built on the same run directory — + # a second worker, or a resumed process — each keep their own in-memory + # snapshot, so both could pass the "is this lease still live?" check against a + # stale copy and both write their own lease: measured, worker A and worker B + # both acquired the same step. ``os.replace`` makes each write atomic but + # provides no mutual exclusion between the writers. + # + # A file lock does. Same approach as the data-asset builder lock: an advisory + # ``fcntl.flock`` on a sidecar file, degrading to a logged no-op where + # ``fcntl`` is unavailable. + + @property + def _lock_path(self) -> Path: + return self.path.with_name(self.path.name + ".lock") + + @contextmanager + def _cross_instance_lock(self) -> Any: + """Serialise a read-modify-write against other instances of this run.""" + + try: + import fcntl + except ImportError: # pragma: no cover - platform dependent + LOGGER.warning( + "fcntl unavailable; lease exclusion between journal instances is " + "not enforced on this platform (%s)", + self.path, + ) + yield + return + self.run_dir.mkdir(parents=True, exist_ok=True) + handle = self._lock_path.open("a+") + try: + fcntl.flock(handle.fileno(), fcntl.LOCK_EX) + yield + finally: + try: + fcntl.flock(handle.fileno(), fcntl.LOCK_UN) + finally: + handle.close() + + def reload(self) -> None: + """Refresh the in-memory snapshot from disk. + + Called while holding the cross-instance lock, so a read-modify-write cycle + decides on the state another instance actually committed rather than on + the copy this instance loaded at construction time. + """ + + with self._lock: + self._journal = self._load_or_create() + # ------------------------------------------------------------------ leases def acquire_lease( self, step_id: str, *, owner: str, ttl_seconds: float = 60.0 ) -> StepLease | None: - """Claim a step for one worker; returns None when another lease is live.""" + """Claim a step for one worker; returns None when another lease is live. + + The claim is decided under a cross-instance file lock and on a freshly + reloaded snapshot, so a second worker cannot win the same step by checking + its own stale copy. Measured before this: two instances in the same + process both acquired the same step. + """ + now = self.clock() - with self._lock: - record = self._journal.steps.get(step_id) - if record is None: - return None - lease = record.lease - if lease is not None and lease.expires_at > now and lease.owner != owner: - return None - token = hashlib.sha256( - f"{self._journal.run_id}:{step_id}:{owner}:{now}".encode("utf-8") - ).hexdigest()[:16] - record.lease = StepLease( - owner=owner, token=token, acquired_at=now, expires_at=now + ttl_seconds - ) - self.save() - return record.lease + with self._cross_instance_lock(): + # Decide on what is actually on disk, not on what this instance loaded + # when it was constructed. + self.reload() + with self._lock: + record = self._journal.steps.get(step_id) + if record is None: + return None + lease = record.lease + if ( + lease is not None + and lease.expires_at > now + and lease.owner != owner + ): + LOGGER.info( + "lease_denied run_id=%s step=%s held_by=%s expires_in=%.1fs", + self._journal.run_id, + step_id, + lease.owner, + lease.expires_at - now, + ) + return None + token = hashlib.sha256( + f"{self._journal.run_id}:{step_id}:{owner}:{now}".encode("utf-8") + ).hexdigest()[:16] + record.lease = StepLease( + owner=owner, + token=token, + acquired_at=now, + expires_at=now + ttl_seconds, + ) + self.save() + return record.lease def release_lease(self, step_id: str, *, owner: str) -> None: with self._lock: diff --git a/queryforge/orchestration/runtime/resume.py b/queryforge/orchestration/runtime/resume.py index 54af74b..3b3c86a 100644 --- a/queryforge/orchestration/runtime/resume.py +++ b/queryforge/orchestration/runtime/resume.py @@ -195,7 +195,10 @@ def resume_decisions(self, plan: Any) -> list[ResumeDecision]: reusable: dict[str, bool] = {} for step in plan.steps: fingerprint = journal.fingerprint_step( - step.action, dict(step.inputs or {}), plan_version=plan.version + step.action, + dict(step.inputs or {}), + plan_version=plan.version, + versions=journal.versions, ) record = journal.reusable_step(step.id, fingerprint) upstream_ok = all(reusable.get(item, False) for item in step.depends_on) diff --git a/queryforge/orchestration/schemas/__init__.py b/queryforge/orchestration/schemas/__init__.py index 294a4c0..beef8a6 100644 --- a/queryforge/orchestration/schemas/__init__.py +++ b/queryforge/orchestration/schemas/__init__.py @@ -21,6 +21,11 @@ "blocked", "failed", "completed", + # A run that stopped to ask the user a question (reflection returned + # NEED_USER_REVIEW, or the planner could not resolve a governed metric) is + # terminal for this attempt but is neither a success nor a failure. The value + # has to stay in this vocabulary so the persisted state round-trips. + "needs_clarification", # A run stopped because its client went away (step 14) is persisted with this # status by ``agent_service.persist_cancelled_outcome``. The value has to stay # in this vocabulary, otherwise the cancelled ``state.json`` cannot be diff --git a/queryforge/orchestration/schemas/knowledge_versions.py b/queryforge/orchestration/schemas/knowledge_versions.py index beb3afb..f8b799f 100644 --- a/queryforge/orchestration/schemas/knowledge_versions.py +++ b/queryforge/orchestration/schemas/knowledge_versions.py @@ -131,7 +131,18 @@ def knowledge_retrieval_version_refs( if kind is None: continue identifier = _governed_identifier(document, kind) - version = content_version(str(_field(document, "text") or "")) + # Digest the governance state alongside the text. ``content_version``'s own + # docstring promises that "any edit to a definition, a synonym, an owner or + # a review status changes it", but only the retrieval text was passed, and + # that text contains none of these fields: demoting an entry to + # ``deprecated`` or narrowing its permissions left the version identical, so + # no session was told its knowledge had changed. Each part is hashed + # separately by ``content_version``, so adding parts cannot collide with a + # different partition of the same content. + version = content_version( + str(_field(document, "text") or ""), + *_governance_parts(document), + ) if not identifier or not version: continue references.append( @@ -166,6 +177,41 @@ def merge_version_refs( return merged +#: Governance fields that must participate in a content version. Order is fixed so +#: the digest is stable. +_GOVERNANCE_VERSION_KEYS: tuple[str, ...] = ( + "review_status", + "verification_level", + "permissions", + "domain_id", + "owner", +) + + +def _governance_parts(document: Any) -> list[str]: + """The governance state of one document, as digestable strings. + + Only values that are actually set contribute, so an absent permission list does + not make two otherwise-identical documents differ from a document that never + carried the key at all. + """ + + metadata = _field(document, "metadata") + if not isinstance(metadata, Mapping): + metadata = {} + parts: list[str] = [] + for key in _GOVERNANCE_VERSION_KEYS: + value = metadata.get(key, _field(document, key)) + if value in (None, "", [], ()): + continue + if isinstance(value, (list, tuple, set)): + rendered = ",".join(sorted(str(item) for item in value)) + else: + rendered = str(value) + parts.append(f"{key}={rendered}") + return parts + + def _field(document: Any, name: str) -> Any: """Read one field of a retrieval match, whether it is a model or a mapping.""" if isinstance(document, Mapping): diff --git a/queryforge/orchestration/tools/budget.py b/queryforge/orchestration/tools/budget.py index fe3fc62..b613d84 100644 --- a/queryforge/orchestration/tools/budget.py +++ b/queryforge/orchestration/tools/budget.py @@ -184,6 +184,41 @@ def remaining(self, key: str) -> float: 0.0, float(getattr(self.limits, key)) - float(getattr(self._usage, key)) ) + def restore(self, usage: Mapping[str, Any] | None) -> list[str]: + """Re-apply a previously persisted usage snapshot. + + A resumed run must inherit what the earlier attempt already spent, + otherwise the budget boundary resets on every resume and a run can spend + its whole allowance again after a crash. Returns the keys that were + restored, so a caller can report what carried over rather than assuming. + + Values are clamped to the configured limits: a snapshot that exceeds the + current limits (because they were tightened) must not leave the manager in + a state where ``remaining`` is negative and nothing can run. + """ + + if not usage: + return [] + applied: list[str] = [] + with self._lock: + for key in BUDGET_KEYS: + if key not in usage: + continue + try: + amount = float(usage[key]) + except (TypeError, ValueError): + continue + if amount <= 0: + continue + limit = float(getattr(self.limits, key)) + # Preserve the spent amount, but never above the limit: the point + # is that the allowance is consumed, not that it is invalid. + setattr(self._usage, key, min(amount, limit) if limit > 0 else amount) + applied.append(key) + if applied: + self._reservations += 1 + return applied + def snapshot(self) -> dict[str, Any]: """Serialize limits, usage, and remaining budget for a result payload.""" diff --git a/queryforge/workflow/budgeted_model.py b/queryforge/workflow/budgeted_model.py new file mode 100644 index 0000000..267d4b8 --- /dev/null +++ b/queryforge/workflow/budgeted_model.py @@ -0,0 +1,162 @@ +"""Bound every model call by the run's shared budget and deadline. + +The conversational path had no model budget at all. ``BudgetLimits`` declared +``model_deadline_ms`` and ``max_estimated_tokens``, and ``BudgetManager`` could +enforce both, but nothing on the ``/ask`` path ever constructed a manager or called +``reserve`` for a model request: only the tool loop and the planner did. The result +was that a run's model spend and wall-clock time were unbounded, and +``model_deadline_ms`` had no consumer anywhere. + +This module closes that gap with a provider decorator rather than with changes at +each of the seven decision points, so every model call is covered by construction +instead of by remembering to add a call site. + +Design notes: + +* A call **reserves** before the request is sent and **settles** with the tokens the + provider actually reported. Reserving first is what makes the cap real: a + reservation that cannot be satisfied raises before any spend happens. +* The remaining deadline is published through + :func:`queryforge.core.observability.model_deadline`, so adapters can bound the + HTTP request instead of discovering the overrun afterwards. +* A refusal is recorded on the run context, because "the run stopped" and "the run + stopped because the budget ran out" are different reports. +""" + +from __future__ import annotations + +from contextlib import contextmanager +from typing import Any + +from queryforge.core.observability import ( + _call_with_optional_timeout, + current_model_deadline, + model_deadline, +) +from queryforge.orchestration.tools.budget import BudgetManager +from queryforge.orchestration.tools.specs import ToolBudgetError + + +class BudgetedModelProvider: + """Provider decorator that charges every model call to a shared budget. + + Wraps the observed provider, so both the span and the budget see the call. + """ + + #: Tokens reserved before a call, reconciled to the reported total afterwards. + #: + #: A first version reserved the *whole* per-call cap (20k by default) on every + #: call, so three model calls exhausted the 100k global allowance and every + #: workflow run failed with a budget refusal — the enforcement was real but the + #: accounting was wrong. An expectation is reserved and then settled to the + #: truth; the per-call cap still applies as a hard ceiling on any single call. + DEFAULT_EXPECTED_TOKENS = 4_000.0 + + def __init__( + self, + provider: Any, + budget: BudgetManager, + *, + refusal_sink: dict[str, Any] | None = None, + expected_tokens: float | None = None, + ) -> None: + self._provider = provider + self._budget = budget + self._refusal_sink = refusal_sink + self._expected_tokens = float( + self.DEFAULT_EXPECTED_TOKENS + if expected_tokens is None + else expected_tokens + ) + self.provider = getattr(provider, "provider", None) + self.model = getattr(provider, "model", None) + + def __getattr__(self, name: str) -> Any: + return getattr(self._provider, name) + + @contextmanager + def _charged(self, *, entry_point: str): + """Reserve budget, publish the deadline, settle with reported usage. + + One implementation for every entry point. An earlier version overrode only + ``generate_with_messages`` and re-implemented ``generate_json`` on top of + it, which silently bypassed each provider's own ``generate_json`` — every + adapter and test double that customises it stopped being called. + """ + + try: + reservation = self._budget.reserve( + category="model", + calls=1, + estimated_tokens=self._expected_tokens, + require_remaining=("model_deadline_ms",), + ) + except ToolBudgetError as exc: + self._record_refusal(exc) + raise + try: + with model_deadline(self._budget.deadline_seconds()): + yield + except Exception: + # A failed call still consumed its reservation (and may have been + # billed), so settle at the reserved amount rather than releasing it. + reservation.settle(max_estimated_tokens=self._expected_tokens) + raise + usage = self._last_usage() + actual: dict[str, float] = {} + if usage is not None: + tokens = int(getattr(usage, "total_tokens", 0) or 0) + if tokens > 0: + actual["max_estimated_tokens"] = float(tokens) + reservation.settle(**actual) + + # ------------------------------------------------------- entry points + # + # The nodes call ``generate_json`` (and occasionally ``generate_text``), not + # ``generate_with_messages``. Each entry point therefore charges the budget and + # then delegates to the *inner* provider's same method, so a provider that + # customises one of them keeps its behaviour. + + def generate_with_messages( + self, + messages: list[dict[str, str]], + json_mode: bool = False, + **_ignored: Any, + ) -> str: + with self._charged(entry_point="generate_with_messages"): + return self._provider.generate_with_messages( + messages, + json_mode=json_mode, + timeout=current_model_deadline(), + ) + + def generate_json(self, prompt: str) -> dict[str, Any]: + with self._charged(entry_point="generate_json"): + return _call_with_optional_timeout( + self._provider.generate_json, prompt, current_model_deadline() + ) + + def generate_text(self, prompt: str) -> str: + with self._charged(entry_point="generate_text"): + return _call_with_optional_timeout( + self._provider.generate_text, prompt, current_model_deadline() + ) + + # ------------------------------------------------------------------ helpers + + def _last_usage(self) -> Any: + inner = getattr(self._provider, "_provider", None) + return getattr(inner, "last_usage", None) or getattr( + self._provider, "last_usage", None + ) + + def _record_refusal(self, exc: ToolBudgetError) -> None: + if self._refusal_sink is None: + return + self._refusal_sink["budget_refusal"] = { + "limit": getattr(exc, "limit", None), + "reason": str(exc), + "usage": self._budget.snapshot().get("usage"), + "deadline_exhausted": current_model_deadline() is not None + and float(current_model_deadline() or 0) <= 0, + } diff --git a/queryforge/workflow/node/gen_sql_node.py b/queryforge/workflow/node/gen_sql_node.py index 77896cc..df5e2f9 100644 --- a/queryforge/workflow/node/gen_sql_node.py +++ b/queryforge/workflow/node/gen_sql_node.py @@ -2,6 +2,7 @@ import json import logging +from typing import Any import sqlglot from sqlglot import expressions as exp @@ -41,18 +42,31 @@ def execute(self, context: Context) -> NodeResult: context.sql_context = SQLContext.model_validate(payload) if payload.get("reasoning") is not None: try: - context.sql_context.reasoning_result = ReasoningResult.model_validate( + normalized, coercions = self._normalize_reasoning( payload["reasoning"] ) + context.sql_context.reasoning_result = ReasoningResult.model_validate( + normalized + ) context.sql_context.reasoning_validation = self._validate_reasoning( context.sql_context.sql, context.sql_context.reasoning_result, ) + if coercions: + context.sql_context.reasoning_validation["coercions"] = coercions context.reasoning_result = context.sql_context.reasoning_result context.reasoning_validation = ( context.sql_context.reasoning_validation ) except ValidationError as exc: + # Recorded on the context, not only in a nested warning: the + # payload is dropped entirely here, and a silent drop makes a + # model that consistently emits an incompatible shape + # indistinguishable from one that emits nothing. + context.reasoning_discarded = "; ".join( + f"{'.'.join(str(part) for part in error['loc'])}: {error['msg']}" + for error in exc.errors()[:5] + ) context.reasoning_validation = { "status": "warning", "warnings": [f"Invalid reasoning payload: {exc}"], @@ -77,6 +91,116 @@ def execute(self, context: Context) -> NodeResult: return self.failure(f"Could not generate SQL: {exc}") return self.success("Generated SQLite query") + #: Words a model may use instead of a number for ``confidence``. + #: + #: Mapped strictly *below* the self-verification threshold (0.8) on purpose: a + #: word is not a measurement, so it must never by itself be enough to skip the + #: review pass. A model that genuinely means "high" can say 0.9. + _CONFIDENCE_WORDS: dict[str, float] = { + "high": 0.75, + "very high": 0.79, + "medium": 0.5, + "moderate": 0.5, + "low": 0.2, + "very low": 0.1, + } + + @classmethod + def _normalize_reasoning(cls, payload: Any) -> tuple[dict[str, Any], list[str]]: + """Coerce the shapes models actually emit into the declared contract. + + Observed from a real run, the same payload violated the schema in four + places at once — ``metrics`` as bare strings instead of objects, + ``sorting``/``limit`` as the *string* ``"None"`` instead of null, and + ``confidence`` as the word ``"high"`` instead of a number. Every run + therefore failed validation and the whole reasoning payload was discarded, + which made the audit summary dead weight in production. + + Coercion is recorded and surfaced (``reasoning_validation.coercions``) so a + drifting prompt cannot hide behind tolerant parsing. Anything that cannot + be coerced is left alone and still fails validation, which keeps the + original behaviour: no reasoning rather than wrong reasoning. + """ + + if not isinstance(payload, dict): + return payload, [] + normalized = dict(payload) + coercions: list[str] = [] + + nullish = {"none", "null", "n/a", "na", "nan"} + + def _is_nullish(value: Any) -> bool: + return isinstance(value, str) and value.strip().lower() in nullish + + for key in ("time_range", "limit", "strategy"): + if _is_nullish(normalized.get(key)): + normalized[key] = None + coercions.append(f"{key}: string-null -> null") + + if _is_nullish(normalized.get("sorting")): + normalized["sorting"] = [] + coercions.append("sorting: string-null -> []") + + # Each of these is "a list of objects" in the contract, and the model + # reliably flattens some of them to bare strings. Confirmed on real runs: + # ``metrics`` and ``sorting`` both arrive as ["COUNT(x)", "col DESC"]. A + # single unconvertible entry discarded the entire payload, so the whole + # audit summary was dead weight in production. + list_shapes: tuple[tuple[str, str, Any], ...] = ( + ("metrics", "expression", None), + ("sorting", "column", None), + ("dimensions", None, None), + ("tables", None, None), + ) + for key, field, _unused in list_shapes: + value = normalized.get(key) + if not isinstance(value, list): + continue + if not any(isinstance(item, str) for item in value): + continue + if field is None: + continue + normalized[key] = [ + {field: item} if isinstance(item, str) else item for item in value + ] + coercions.append(f"{key}: string entries -> {{{field}}}") + + if isinstance(normalized.get("sorting"), list): + parsing = [] + changed = False + for item in normalized["sorting"]: + if not isinstance(item, dict) or "direction" in item: + parsing.append(item) + continue + column = str(item.get("column") or "") + upper = column.upper() + for suffix, direction in ((" DESC", "DESC"), (" ASC", "ASC")): + if upper.endswith(suffix): + item = {"column": column[: -len(suffix)].strip(), "direction": direction} + changed = True + break + parsing.append(item) + if changed: + normalized["sorting"] = parsing + coercions.append("sorting: 'col DESC' -> {column, direction}") + + confidence = normalized.get("confidence") + if isinstance(confidence, str): + text = confidence.strip().lower() + if text in cls._CONFIDENCE_WORDS: + normalized["confidence"] = cls._CONFIDENCE_WORDS[text] + coercions.append( + f"confidence: {confidence!r} -> {normalized['confidence']}" + ) + else: + try: + normalized["confidence"] = float(text) + coercions.append(f"confidence: {confidence!r} -> float") + except ValueError: + pass + + return normalized, coercions + @staticmethod def _validate_reasoning( sql: str, @@ -264,7 +388,9 @@ def _build_prompt(context: Context) -> str: except for a scalar aggregate query guaranteed to return one row. - Return only one JSON object with fields sql, explanation, tables_used, and optional reasoning. The reasoning must be a concise structured audit summary, not hidden - chain-of-thought: + chain-of-thought. If you include it, report "confidence" as a number between 0 and + 1 and list any unresolved assumption in "assumptions" and any known weakness in + "risks": {{"sql": "SELECT ...", "explanation": "...", "tables_used": ["..."], "reasoning": {{"goal": "...", "grain": "...", "tables": ["..."], "joins": [], "metrics": [], "dimensions": [], "filters": [], "time_range": null, diff --git a/queryforge/workflow/node/output_node.py b/queryforge/workflow/node/output_node.py index 047fa1d..7ba915f 100644 --- a/queryforge/workflow/node/output_node.py +++ b/queryforge/workflow/node/output_node.py @@ -134,6 +134,7 @@ def execute(self, context: Context) -> NodeResult: if context.reflection_result else None ), + "reasoning_discarded": context.reasoning_discarded, "retry_count": context.retry_count, "execution_errors": context.execution_errors, "fix_attempts": [ @@ -176,6 +177,10 @@ def execute(self, context: Context) -> NodeResult: "columns": execution.columns, "rows": execution.rows, "row_count": execution.row_count, + # The bound is now visible to callers: without it a capped result was + # indistinguishable from a complete one. + "truncated": execution.truncated, + "fetched_row_count": execution.fetched_row_count or execution.row_count, "sql_execution_duration_ms": context.sql_execution_duration_ms, "sql_security": { **context.sql_policy, @@ -191,6 +196,7 @@ def execute(self, context: Context) -> NodeResult: "history": context.tool_loop_history, }, "candidate_selection": context.candidate_selection, + "candidate_strategy": context.candidate_strategy, "reasoning": ( context.reasoning_result.model_dump(mode="json") if context.reasoning_result @@ -198,6 +204,12 @@ def execute(self, context: Context) -> NodeResult: ), "reasoning_validation": context.reasoning_validation, "task_evidence": _task_evidence(context), + # First-class, not buried in task_evidence: a semantic verdict of + # ``unsupported`` means "this SQL was never proved to answer the + # question". It used to reach only a log line, so a run whose business + # semantics were never checked was delivered exactly like one that + # passed every check. `verified` states that difference explicitly. + "semantic_validation": _semantic_validation_summary(context), # Step 12 evidence layer: every key number traceable to its source. "evidence": evidence_layer["evidence"], "evidence_issues": evidence_layer["evidence_issues"], @@ -253,6 +265,15 @@ def _evidence_layer(self, context: Context) -> dict[str, Any]: f"result evidence could not be recorded: {type(exc).__name__}: {exc}" ) + # Compose an answer when the pipeline did not supply one (E-10). The + # conversational path never wrote ``final_answer``, so a run delivered + # evidence and a completeness marker but no statement of what the result + # says — ``final_answer`` was always None here. The composer derives every + # number from the evidence store, so this adds no new source of truth. + if task_context.get("final_answer") is None and execution is not None: + composed = _compose_answer(context, store, execution, sql_context) + if composed is not None: + task_context["final_answer"] = composed.model_dump(mode="json") answer = _final_answer(task_context.get("final_answer"), store, issues) problems: list[str] = [] if answer is not None: @@ -268,16 +289,23 @@ def _evidence_layer(self, context: Context) -> dict[str, Any]: if returned >= row_count else f"The JSON 'rows' list already holds only {returned} of {row_count} reported rows." ) + semantic = _semantic_validation_summary(context) + completeness_notes = [ + f"Display truncation follows the report's report_max_rows={display_limit}. " + f"{json_note}", + *issues, + ] + if not semantic["verified"]: + # Stated in the completeness record as well as in its own field, so a + # consumer reading only one of the two still sees that the business + # semantics were not proved. + completeness_notes.append(f"Business semantics: {semantic['reason']}") completeness = summarize_completeness( total_row_count=row_count, displayed_row_count=min(row_count, display_limit), evidence=store.all(), answer=answer, - extra_notes=[ - f"Display truncation follows the report's report_max_rows={display_limit}. " - f"{json_note}", - *issues, - ], + extra_notes=completeness_notes, ) return { "evidence": store.to_list(), @@ -305,8 +333,20 @@ def _result_evidence(self, context: Context, execution: Any, sql_context: Any) - if execution.row_count == returned else COMPLETENESS_TRUNCATED ) - if task_context.get("result_truncated") is True: + if task_context.get("result_truncated") is True or execution.truncated: completeness = COMPLETENESS_TRUNCATED + # The bound is stated in the evidence method so the reader sees that the + # aggregate was computed over a capped row set. + method = ( + "executed SQL over the run's data version (aggregates computed over the " + "complete returned result set)" + ) + if execution.truncated: + method = ( + f"executed SQL over the run's data version; the adapter's row bound " + f"returned {execution.row_count} of {execution.fetched_row_count} " + f"rows, so aggregates cover only the returned rows" + ) return build_execution_evidence( sql=sql_context.sql, source=context.task.database_path, @@ -317,8 +357,7 @@ def _result_evidence(self, context: Context, execution: Any, sql_context: Any) - grain=grain, range_=_resolved_range(context, request), completeness=completeness, - method="executed SQL over the run's data version (aggregates computed over the " - "complete returned result set)", + method=method, kind=KIND_SQL_RESULT, ) @@ -363,6 +402,113 @@ def _resolved_range(context: Context, request: dict[str, Any]) -> dict[str, Any] return {"expression": str(time_range)} if time_range else None +def _compose_answer( + context: Context, + store: EvidenceStore, + execution: Any, + sql_context: Any, +) -> FinalAnswer | None: + """Build a :class:`FinalAnswer` from the run's own evidence. + + Every number is resolved against the evidence payloads, so a value that cannot + be found becomes ``None`` and is reported as unknown rather than invented. A + failure here is recorded as an issue and never fails the run: the query result + is the primary product. + """ + + from queryforge.domain.analysis.evidence import AnswerComposer, Finding + + try: + result_evidence_id = next( + ( + item.id + for item in reversed(list(store)) + if item.kind == KIND_SQL_RESULT + and (item.sql or "").strip() == (sql_context.sql or "").strip() + ), + None, + ) + if result_evidence_id is None: + return None + columns = list(execution.columns or []) + row_count = int(execution.row_count or 0) + numbers: dict[str, Any] = {"row_count": row_count} + if row_count == 1 and len(columns) == 1: + # Name the single scalar so the answer states it. The key must be one + # the evidence payload resolves to a *number*: a bare column name + # resolves to that column's aggregate dict, which the composer reports + # as unknown, whereas the evidence also exposes a flat ``total_`` + # scalar. Declared as None so the composer takes the evidence value as + # the only source of truth. + numbers[f"total_{columns[0]}"] = None + finding = Finding( + kind=KIND_SQL_RESULT, + statement=( + f"The query returned {row_count} row(s) over columns " + f"{', '.join(columns) or ''}." + ), + numbers=numbers, + evidence_ids=[result_evidence_id], + ) + composer = AnswerComposer(store) + composed = composer.compose(context.task.question, [finding]) + problems = validate_answer(composed, store) + return apply_validation(composed, problems) if problems else composed + except Exception: + # Reported by the caller through the completeness notes; an answer layer + # problem must not cost the caller the result. + return None + + +def _semantic_validation_summary(context: Context) -> dict[str, Any]: + """Report whether the business semantics of this SQL were actually proved. + + Three distinct situations must not look alike: + + * ``passed`` / ``violation`` — the governed validator reached a verdict; + * ``unsupported`` — the validator could not prove anything (an unlisted SQL + shape, an unparsable statement, or a question that matched no governed + metric). Execution continues under SQL policy, but "not disproved" is not + "proved", and the answer has to say so; + * no verdict at all — no semantic model, or no governed metric matched, so the + semantic gate never ran. This is the common case for a follow-up that does + not restate a metric. + + ``verified`` is true only in the first group, so a consumer can gate on one + boolean instead of re-deriving it from whichever field happens to be present. + """ + + raw = context.task_context.get("semantic_validation") + if not isinstance(raw, dict): + return { + "status": "not_run", + "verified": False, + "reason": ( + "No semantic verdict was produced for this run: the question matched " + "no governed metric, or no semantic model was in scope. Table scope " + "and the SQL policy still applied; business semantics were not checked." + ), + "rules": [], + } + status = str(raw.get("status") or "unsupported") + verified = status == "passed" + if status == "unsupported": + reason = str( + raw.get("unsupported_reason") + or "the governed validator could not prove this SQL answers the question" + ) + elif status == "violation": + reason = "The governed validator rejected this SQL." + else: + reason = "The governed validator proved this SQL answers the question." + return { + "status": status, + "verified": verified, + "reason": reason, + "rules": list(raw.get("rule_names") or []), + } + + def _task_evidence(context: Context) -> dict[str, Any]: """Summarize step 04-09 task evidence for audit without dumping raw dumps. diff --git a/queryforge/workflow/node/plan_output_node.py b/queryforge/workflow/node/plan_output_node.py index 9abf0c8..7319fba 100644 --- a/queryforge/workflow/node/plan_output_node.py +++ b/queryforge/workflow/node/plan_output_node.py @@ -20,15 +20,16 @@ def execute(self, context: Context) -> NodeResult: context.sql_context.sql ) except Exception as exc: - decision = self.database_tool.last_policy_decision + # This node preflights the statement itself instead of calling the + # tool, so the only decision describing *this* statement is the one the + # engine attached to the exception. Reading — or worse, assigning — + # ``database_tool.last_policy_decision`` here would attribute another + # caller's decision to this plan, and the tool's audit record would + # claim a call that never happened (E-23). policy_violation = getattr(exc, "decision", None) if policy_violation is not None: - decision = policy_violation - self.database_tool.last_policy_decision = policy_violation - if decision is not None: - context.sql_policy_decisions.append(decision) + context.sql_policy_decisions.append(policy_violation) return self.failure(f"SQL policy preflight failed: {exc}") - self.database_tool.last_policy_decision = decision context.sql_policy_decisions.append(decision) context.final_output = { "status": "planned", diff --git a/queryforge/workflow/node/visualization_node.py b/queryforge/workflow/node/visualization_node.py index 06774c2..78cd8aa 100644 --- a/queryforge/workflow/node/visualization_node.py +++ b/queryforge/workflow/node/visualization_node.py @@ -11,11 +11,12 @@ from pathlib import Path from typing import Any, Literal +from queryforge.core.paths import workspace_root from queryforge.workflow.node.base import Node from queryforge.core.schemas.models import Context, NodeResult, VisualizationResult -PROJECT_ROOT = Path(__file__).resolve().parents[3] +PROJECT_ROOT = workspace_root() DEFAULT_CHART_OUTPUT_DIR = PROJECT_ROOT / ".queryforge/charts" VEGA_LITE_SCHEMA = "https://vega.github.io/schema/vega-lite/v5.json" LOGGER = logging.getLogger("queryforge.visualization") diff --git a/queryforge/workflow/workflow.py b/queryforge/workflow/workflow.py index 62f038d..f800fd8 100644 --- a/queryforge/workflow/workflow.py +++ b/queryforge/workflow/workflow.py @@ -223,16 +223,41 @@ def run(self) -> dict: if getattr(self.context, "final_output", None) is not None: assert self.context.final_output is not None return self.context.final_output + # Candidate strategy. Exactly one SQL producer runs per attempt, and the + # precedence is deliberate: + # + # 1. a tool-loop ``final_answer`` already produced SQL, so it is used as-is; + # 2. otherwise, if several candidates are configured, they are generated and + # the selector picks one; + # 3. otherwise a single generation runs. + # + # The consequence of (1) is that candidates are skipped whenever the tool + # loop produced SQL — which is most complex requests, because complexity + # routing enables both the tool loop *and* extra candidates. That is not a + # dead branch, it is the intended precedence (observations gathered first, + # then one answer), but it means "complex" does not imply "candidates ran". + # Recorded here because the behaviour was previously inferable only from the + # control flow, and a run could not report which strategy it used. + tool_loop_produced_sql = False if self.tool_loop_node is not None and self.context.sql_context is None: self._run_required(self.tool_loop_node) + tool_loop_produced_sql = self.context.sql_context is not None if getattr(self.context, "final_output", None) is not None: assert self.context.final_output is not None return self.context.final_output - if self.context.sql_context is None: + if not tool_loop_produced_sql and self.context.sql_context is None: if self.parallel_candidates_node is not None: self._run_required(self.parallel_candidates_node) else: self._run_required(self.gen_sql_node) + self.context.candidate_strategy = ( + "tool_loop" + if tool_loop_produced_sql + else "parallel_candidates" + if self.parallel_candidates_node is not None + and self.context.candidate_selection is not None + else "single_generation" + ) while True: self._check_cancelled() @@ -284,6 +309,16 @@ def run(self) -> dict: continue self.context.last_execution_error = None + # Reflection runs on every execution, unconditionally. + # + # A conditional skip was implemented and measured, then removed: the + # gate required a self-verified generation (high confidence, no declared + # assumptions, no declared risks) and it fired on **0 of 40** cases, + # because the model correctly reports high confidence *together with* + # material assumptions on essentially every question. Skipping on that + # signal would have meant not asking for assumptions at all, which + # trades audit visibility for latency. Measurements and the three + # alternatives considered are recorded in docs/evaluation_baselines.md. self._run_required(self.reflect_node) reflection = self.context.reflection_result if reflection is None: @@ -300,39 +335,21 @@ def run(self) -> dict: ) if reflection.strategy == "SUCCESS": - self._run_required(self.output_node) - if self.visualization_node is not None: - visualization_result = self._run(self.visualization_node) - if not visualization_result.success: - assert self.context.final_output is not None - self.context.final_output["visualization"] = { - "chart_type": "table", - "chart_config": { - "format": "table", - "columns": ( - self.context.execution_result.columns - if self.context.execution_result - else [] - ), - "rows": ( - self.context.execution_result.rows - if self.context.execution_result - else [] - ), - }, - "chart_path": None, - "reason": "Visualization failed; SQL output remains valid.", - "error": visualization_result.error, - } + self._finish_successfully() assert self.context.final_output is not None return self.context.final_output if reflection.strategy == "NEED_USER_REVIEW": - raise WorkflowError( - "reflect", - f"Human review required: {reflection.reason}", - self.context, - ) + # The reflection verdict is "I cannot decide whether this answers + # the question". That is a clarification request, not a failure: + # raising here produced zero payload and recorded the run as + # failed, even though the model had correctly identified an + # ambiguity (observed on the benchmark's only multi-turn case). + # ``needs_clarification`` is already part of the platform's + # terminal vocabulary (event protocol, REST mapping, gateway + # wording, evaluator outcome set), so surface it as a result. + self.context.final_output = self._clarification_output(reflection) + return self.context.final_output self._require_retry(reflection.strategy, reflection.reason) self.context.retry_count += 1 @@ -364,6 +381,62 @@ def run(self) -> dict: self.context, ) + def _finish_successfully(self) -> None: + """Assemble the final output, with the table fallback if a chart fails. + + Shared by the reflected-success path and the self-verified path so the two + cannot drift apart in how a failed visualization is reported. + """ + + self._run_required(self.output_node) + if self.visualization_node is None: + return + visualization_result = self._run(self.visualization_node) + if visualization_result.success: + return + assert self.context.final_output is not None + execution = self.context.execution_result + self.context.final_output["visualization"] = { + "chart_type": "table", + "chart_config": { + "format": "table", + "columns": execution.columns if execution else [], + "rows": execution.rows if execution else [], + }, + "chart_path": None, + "reason": "Visualization failed; SQL output remains valid.", + "error": visualization_result.error, + } + + def _clarification_output(self, reflection: Any) -> dict[str, Any]: + """Structured clarification result for a NEED_USER_REVIEW verdict. + + Shaped to match the payload ``AnalysisPlannerService`` already returns + for a clarification, so both execution paths report the same terminal + status and reason vocabulary. The SQL and its result are kept: the + ambiguity is about meaning, not about whether the query ran. + """ + + sql_context = self.context.sql_context + execution = self.context.execution_result + self.context.last_execution_error = None + return { + "status": "needs_clarification", + "run_id": self.context.run_id, + "question": self.context.task.question, + "reason": reflection.reason, + "strategy": reflection.strategy, + "unresolved_questions": [reflection.reason], + "sql": sql_context.sql if sql_context else None, + "explanation": sql_context.explanation if sql_context else None, + "tables_used": list(sql_context.tables_used) if sql_context else [], + "columns": execution.columns if execution else [], + "rows": execution.rows if execution else [], + "row_count": execution.row_count if execution else 0, + "retry_count": self.context.retry_count, + "execution_errors": list(self.context.execution_errors), + } + def _register_attempt_signature(self) -> None: """Track normalized SQL per attempt and stop A -> B -> A repair cycles.""" signature = normalize_sql_signature( diff --git a/queryforge/workflow/workflow_runner.py b/queryforge/workflow/workflow_runner.py index 89456e8..cafbafc 100644 --- a/queryforge/workflow/workflow_runner.py +++ b/queryforge/workflow/workflow_runner.py @@ -29,6 +29,7 @@ from queryforge.workflow.node.subject_selection_node import SubjectSelectionNode from queryforge.workflow.node.tool_loop_node import ToolLoopNode from queryforge.workflow.node.visualization_node import VisualizationNode +from queryforge.workflow.budgeted_model import BudgetedModelProvider from queryforge.workflow.event_emitter import EventEmitter from queryforge.workflow.workflow import ( ReflectiveWorkflow, @@ -50,7 +51,13 @@ stable_digest, start_span_recorder, ) -from queryforge.core.schemas.models import Context, SQLContext, SqlTask, VectorMatch +from queryforge.core.schemas.models import ( + Context, + RunContext, + SQLContext, + SqlTask, + VectorMatch, +) from queryforge.domain.security import load_sql_policy from queryforge.domain.skills.manager import SkillManager from queryforge.infrastructure.storage import ( @@ -312,6 +319,8 @@ def __init__( tool_loop_preview_limit: int = 20, tool_budget_manager: BudgetManager | None = None, tool_budget_limits: dict[str, float] | None = None, + model_budget_manager: BudgetManager | None = None, + model_budget_limits: dict[str, float] | None = None, parallel_candidates: int = 1, parallel_max_preview: int = 2, parallel_preview_limit: int = 20, @@ -373,6 +382,11 @@ def __init__( # Step 09: the tool loop shares one atomic budget per run. self.tool_budget_manager = tool_budget_manager self.tool_budget_limits = dict(tool_budget_limits or {}) + # Separate from the tool budget on purpose: tool calls and model calls are + # different resources with different failure modes, and sharing one + # allowance would let a long tool loop silently starve generation. + self._model_budget_manager = model_budget_manager + self.model_budget_limits = dict(model_budget_limits or {}) if parallel_candidates < 1 or parallel_candidates > 3: raise ValueError("parallel_candidates must be between 1 and 3") self.parallel_candidates = parallel_candidates @@ -491,6 +505,14 @@ def _run_inner(self, task: SqlTask, run_id: str) -> tuple[dict, Context]: retrieval_scope = self._retrieval_scope() if retrieval_scope: context.task_context["retrieval_scope"] = retrieval_scope + # Unified run identity + versions, populated once here so downstream + # consumers (artifacts, evidence, recovery) read one object instead of + # reconstructing identity and versions from whichever layer is asking. + context.run_context = RunContext.from_context( + context, + domain_id=self.history_domain_id, + data_version=self.history_data_version, + ) if self.initial_sql: context.sql_context = SQLContext( sql=self.initial_sql, @@ -515,13 +537,24 @@ def _run_inner(self, task: SqlTask, run_id: str) -> tuple[dict, Context]: context.sql_policy = database_tool.policy_summary context.task_context["sql_dialect"] = database_tool.dialect raw_llm = self.llm_factory(self.config) - llm = ObservedModelProvider( + observed_llm = ObservedModelProvider( raw_llm, provider_name=self.config.llm_provider, model_name=self.config.llm_model, debug_prompts=self.debug_prompts, trace_dir=self.trace_dir, ) + # Every model call on this path is charged to one shared budget, and + # the run's remaining deadline reaches the adapter (see + # workflow/budgeted_model.py). Previously only the tool loop and the + # planner had a budget, so /ask model spend was unbounded. + model_budget = self.model_budget_manager() + budget_refusal: dict[str, Any] = {} + llm = BudgetedModelProvider( + observed_llm, model_budget, refusal_sink=budget_refusal + ) + context.model_budget = model_budget + context.budget_refusal = budget_refusal vector_store = self.vector_store if self.enable_vector_kb and vector_store is None: try: @@ -693,6 +726,15 @@ def _run_inner(self, task: SqlTask, run_id: str) -> tuple[dict, Context]: ) from exc return output, context + def model_budget_manager(self) -> BudgetManager: + """The per-run budget that bounds every model call.""" + + if self._model_budget_manager is None: + self._model_budget_manager = BudgetManager( + limits=self.model_budget_limits or None + ) + return self._model_budget_manager + def _tool_budget_manager(self) -> BudgetManager: """Return the per-run budget manager used by the bounded tool loop.""" diff --git a/scripts/evaluate_sql.py b/scripts/evaluate_sql.py index 524bd57..b3f37d0 100644 --- a/scripts/evaluate_sql.py +++ b/scripts/evaluate_sql.py @@ -9,6 +9,14 @@ - Row ORDER is ignored (multiset comparison); DUPLICATE rows are preserved — the comparison never deduplicates with a set. - Column ORDER is tolerated when the column name sets match. +- Column SETS may differ when the widths differ: comparison is done on the + columns the two projections share by name, so an answer that is correct but + projects a different *number* of columns is not scored semantically wrong. + The difference itself is reported per case as ``extra_columns`` / + ``missing_columns`` and in aggregate as ``projection_difference_rate``. + Equal widths with disjoint names keep the positional fallback, so a renamed + column holding the same values is not penalised. Different widths *and* no + shared column name is a different query shape and is not equivalent. - NULL compares equal to NULL only. - Floats: integral floats compare equal to ints; non-integral floats are rounded to 10 decimal places before comparison (float-noise tolerance); @@ -43,6 +51,7 @@ from queryforge.data_assets import DataAssetBuilder from queryforge.domain.security import SQLPolicyViolation, load_sql_policy from queryforge.infrastructure.db.sqlite_connector import SQLiteConnector +from queryforge.infrastructure.models.factory import ModelFactory from queryforge.infrastructure.tools.database_tool import DatabaseTool # Comparison tolerance for non-integral floats (see module docstring). @@ -85,6 +94,30 @@ def parse_args() -> argparse.Namespace: default=1.0, help="Exit-code gate: minimum sql_execution_success_rate (0..1)", ) + parser.add_argument( + "--parallel-candidates", + type=int, + choices=(1, 2, 3), + default=None, + help="Force the candidate count for every query case, overriding the " + "per-case candidate_selection flag. This makes the multi-candidate " + "ablation a controlled experiment over identical inputs: run the same " + "case set with 1 and again with 2 and compare. Without it the flag " + "tracks case category, so a candidate/no-candidate comparison is " + "confounded with task difficulty.", + ) + parser.add_argument( + "--skill-mode", + choices=("auto", "off"), + default="auto", + help="How the evaluated run resolves prompt skills. 'auto' (default) matches " + "the production path: AgentOptions.skills stays unset, so the workflow runs " + "its automatic skill selection and may load optional skills from the " + "catalogue. 'off' passes an empty explicit list, which disables skill " + "loading entirely. The evaluator used to hardcode 'off' without saying so, " + "so every skill in the catalogue except the enabled default was inert in " + "every measured run.", + ) parser.add_argument( "--min-semantic-correct", type=float, @@ -182,9 +215,14 @@ def evaluate_cases( model: str | None = None, input_cost_per_million: float = 0.0, output_cost_per_million: float = 0.0, + usage_log: list[dict[str, int]] | None = None, + parallel_candidates_override: int | None = None, + skill_selection: list[str] | None = None, ) -> dict[str, Any]: results = [] for index, case in enumerate(cases, start=1): + if usage_log is not None: + usage_log.clear() database, semantic_model = environment.resolve( case, default_database, default_semantic_model ) @@ -202,15 +240,33 @@ def evaluate_cases( model_provider=model_provider, model=model, index=index, + usage_log=usage_log, + parallel_candidates_override=parallel_candidates_override, + skill_selection=skill_selection, ) - input_tokens = _estimate_tokens(case["question"]) - output_tokens = _estimate_tokens( - str(result.get("sql") or "") - + "".join( - str(candidate.get("sql") or "") - for candidate in result.get("candidates", []) + # Token accounting: prefer provider-reported usage over the character + # heuristic. ``_estimate_tokens`` measured only the question text (a few + # dozen characters), so the reported average came out around 10 "input + # tokens" while the real prompt is thousands — the number was not merely + # imprecise, it was off by two orders of magnitude and could not support + # any cost bound. Measured usage is recorded when the run reports it, and + # the source is stated explicitly so a report can never be read as if the + # estimate were billing data. + usage = _measured_usage(result, usage_log) + if usage is not None: + input_tokens = usage["input_tokens"] + output_tokens = usage["output_tokens"] + result["token_source"] = "measured" + else: + input_tokens = _estimate_tokens(case["question"]) + output_tokens = _estimate_tokens( + str(result.get("sql") or "") + + "".join( + str(candidate.get("sql") or "") + for candidate in result.get("candidates", []) + ) ) - ) + result["token_source"] = "estimated" result["estimated_input_tokens"] = input_tokens result["estimated_output_tokens"] = output_tokens result["estimated_cost_usd"] = round( @@ -224,11 +280,19 @@ def evaluate_cases( "question": case.get("question") or "", "expected_sql": case.get("expected_sql") or case.get("policy_probe_sql") or "", } + result["failure_class"] = classify_outcome(result) + result["failure_detail"] = failure_detail(result) result["case_fingerprint"] = hashlib.sha256( json.dumps(identity, ensure_ascii=False, sort_keys=True).encode("utf-8") ).hexdigest() results.append(result) - return _build_report(results) + return _build_report( + results, + parallel_candidates_override=parallel_candidates_override, + cases=cases, + default_semantic_model=default_semantic_model, + default_database=default_database, + ) def isolated_config(config: Config, state_root: str | Path) -> Config: @@ -256,6 +320,9 @@ def _evaluate_query( model_provider: str | None, model: str | None, index: int, + usage_log: list[dict[str, int]] | None = None, + parallel_candidates_override: int | None = None, + skill_selection: list[str] | None = None, ) -> dict[str, Any]: started = time.perf_counter() session_id = f"evaluation_{case.get('id', index)}" if case.get("follow_up_context") else None @@ -269,7 +336,7 @@ def _evaluate_query( sql_policy_path=case.get("sql_policy") or default_sql_policy, model_provider=model_provider, model=model, - skills=[], + skills=skill_selection, session_id=session_id, run_id=f"evaluation_warmup_{index}_{turn}", ), @@ -284,9 +351,13 @@ def _evaluate_query( sql_policy_path=case.get("sql_policy") or default_sql_policy, model_provider=model_provider, model=model, - skills=[], + skills=skill_selection, session_id=session_id, - parallel_candidates=2 if case.get("candidate_selection") else 1, + parallel_candidates=( + parallel_candidates_override + if parallel_candidates_override is not None + else (2 if case.get("candidate_selection") else 1) + ), run_id=f"evaluation_{index}", ), ) @@ -305,6 +376,9 @@ def _evaluate_query( expected_rows, expected_columns, ) + extra_columns, missing_columns = _projection_difference( + output.get("columns"), expected_columns + ) selection = output.get("candidate_selection") or {} candidates = selection.get("candidates", []) first_candidate_correct = _candidate_correct( @@ -320,19 +394,67 @@ def _evaluate_query( "expected_outcome": "query", "status": output.get("status"), "execution_success": output.get("status") == "success", + # Coverage markers, derived from the case definition (not the output) + # so they are stable across runs and survive a failed case. + "multi_turn": bool(case.get("follow_up_context")), + "requires_context": bool(case.get("requires_context")), + "compound": bool(case.get("compound")), + "environment_error": False, + "usage": dict(usage_log[-1]) if usage_log else None, "semantic_correct": semantic_correct, + "extra_columns": extra_columns, + "missing_columns": missing_columns, + # False when the comparison could not use column names (either side + # missing them) and fell back to positional values. Projection + # difference is only meaningful when names were available. + "columns_compared": bool(output.get("columns")) and bool(expected_columns), "sql_exact_match": normalize_sql(output.get("sql")) == normalize_sql(case.get("expected_sql")), "policy_rejected": policy_rejected, "policy_expected_rejection": False, "row_count": output.get("row_count"), "latency_ms": round((time.perf_counter() - started) * 1000, 3), + # SQL execution time of the *generated* query, recorded separately + # from end-to-end latency. Candidate selection can only make the + # engine part faster, and the engine part is tiny, so this field is + # what settles whether "a faster candidate" could ever matter. + "sql_execution_ms": output.get("sql_execution_duration_ms"), "oracle_latency_ms": oracle_latency_ms, "sql": output.get("sql"), "selected_index": selection.get("selected_index"), "candidates": candidates, "first_candidate_semantic_correct": first_candidate_correct, "candidate_selection_used": bool(candidates), + # Which prompt skills the run actually loaded, and how they were + # chosen. Without this a skill ablation cannot be audited at all: the + # only trace was inside a log line, and a benchmark that silently + # suppresses selection looks identical to one where selection ran and + # chose nothing. + "loaded_skills": list(output.get("skills_used") or []), + # Whether the reflective second model call was skipped, and on what + # basis. Without these two fields a conditional-reflection change + # cannot be verified at all: a run that skipped reflection and a run + # that reflected successfully look identical from the outside. + "reasoning_confidence": ( + (output.get("reasoning") or {}).get("confidence") + if isinstance(output.get("reasoning"), dict) + else None + ), + # The two conditions that most often keep the self-verification gate + # closed. Recorded so a gate that never fires can be explained rather + # than guessed at. + "reasoning_assumptions": len( + (output.get("reasoning") or {}).get("assumptions") or [] + ) + if isinstance(output.get("reasoning"), dict) + else None, + "reasoning_risks": len( + (output.get("reasoning") or {}).get("risks") or [] + ) + if isinstance(output.get("reasoning"), dict) + else None, + "skill_selection_mode": (output.get("skill_selection") or {}).get("mode"), + "skill_selection_reason": (output.get("skill_selection") or {}).get("reason"), "model_provider": output.get("model_provider"), "model": output.get("model"), "error": None, @@ -345,6 +467,23 @@ def _evaluate_query( "expected_outcome": "query", "status": "failed", "execution_success": False, + "multi_turn": bool(case.get("follow_up_context")), + "requires_context": bool(case.get("requires_context")), + "compound": bool(case.get("compound")), + # Present on the failure path too, so a skipped or discarded reasoning + # payload does not vanish from the bucket counts when the run fails. + "reasoning_confidence": None, + "reasoning_discarded": None, + # A transport/provider failure is not a model-quality failure. A + # depleted balance (HTTP 402) previously landed here as a plain + # "failed" and dragged sql_execution_success_rate from 1.00 to 0.75, + # which reads as a capability regression. Such cases are flagged and + # excluded from the accuracy metrics instead. + "environment_error": is_environment_error(exc), + "environment_error_reason": ( + _environment_error_reason(exc) if is_environment_error(exc) else None + ), + "usage": dict(usage_log[-1]) if usage_log else None, "semantic_correct": False, "sql_exact_match": False, "policy_rejected": None, @@ -353,6 +492,9 @@ def _evaluate_query( "sql": None, "candidates": [], "candidate_selection_used": False, + "columns_compared": False, + "extra_columns": [], + "missing_columns": [], "error": str(exc), } @@ -436,6 +578,78 @@ def _evaluate_policy_probe( } +def _schemas_for(database: Path) -> list[Any]: + with SQLiteConnector(str(database)) as connector: + tool = DatabaseTool(connector) + return [tool.describe_table(name) for name in tool.list_tables()] + + +def _governance_coverage( + cases: list[dict[str, Any]], + default_semantic_model: str | None, + default_database: str | None, +) -> dict[str, Any]: + """How many cases the business-semantic layer can actually see. + + ``SemanticSQLValidator.for_context`` returns None when the question matches no + governed metric, which means the entire semantic gate is skipped for that run: + unknown tables and columns are still caught by table scope, but nothing checks + grain, join keys, fan-out or default filters. + + Metric matching is deterministic term matching, so a follow-up that names no + metric ("Break that down by region.") is exactly the case that needs context + and also exactly the case the semantic layer cannot see. Measured on this + repository: 11 of 12 context-dependent cases and 8 of 12 of the checked-in + multi-turn cases match no governed metric, i.e. they run ungoverned. + + Reported rather than asserted: this is a coverage boundary of the semantic + layer, not a failure of the evaluation. + """ + + from queryforge.domain.semantic import SemanticModelLoader + + ungoverned: list[str] = [] + ungoverned_requiring_context: list[str] = [] + checked = 0 + for case in cases: + if case.get("expected_outcome") == "policy_rejection": + continue + model_path = case.get("semantic_model") or default_semantic_model + database = case.get("database") or default_database + if not model_path or not database: + continue + try: + schemas = _schemas_for(Path(database)) + semantic = SemanticModelLoader.load_and_validate( + Path(model_path), schemas, str(case.get("question") or "") + ) + matches = SemanticModelLoader.match_metrics( + semantic.model, str(case.get("question") or "") + ) + except Exception: # noqa: BLE001 - a coverage probe must never fail a run + continue + checked += 1 + if not matches: + case_id = str(case.get("id")) + ungoverned.append(case_id) + if case.get("requires_context") or case.get("follow_up_context"): + ungoverned_requiring_context.append(case_id) + return { + "cases_checked": checked, + "ungoverned_cases": ungoverned, + "ungoverned_count": len(ungoverned), + #: The intersection that matters: questions that need context, and that the + #: semantic layer therefore cannot see. + "ungoverned_requiring_context": ungoverned_requiring_context, + "note": ( + "A case listed here matched no governed metric, so SemanticSQLValidator " + "was skipped for it: no grain, join-key, fan-out or default-filter check " + "ran. Metric matching is term-based, so context-dependent follow-ups that " + "do not restate the metric are systematically in this set." + ), + } + + def _execute_expected( database: Path, sql: str | None ) -> tuple[list[str], list[list[Any]]]: @@ -488,14 +702,88 @@ def _semantic_equivalent( ) -> bool: """Compare result sets by content, tolerant to column order and row order. - When the actual and expected column name sets match but their order - differs, actual rows are reordered to the expected column order before - comparison; otherwise comparison falls back to positional values. + Comparison happens on the columns the two projections **share by name**: + + * Equal name sets in a different order are reordered and compared in full. + * Different column counts (the defect this rule exists for) compare on the + shared columns only, so an answer that is correct but projects a + different *number* of columns is not scored as semantically wrong. The + difference is reported separately by the caller (see + ``_projection_difference``) rather than silently ignored. + * Equal column counts with disjoint names fall back to positional + comparison — this keeps the historical behaviour for a renamed column + (``anime_count`` vs ``action_anime_count``) with identical values. + * When the two projections share no column name *and* their counts differ, + this is a different query shape, not a projection difference, so the + result is not equivalent. + + Known limitation: a renamed column whose count also differs cannot be + matched by name and is still scored as wrong. Alias normalisation is not + attempted here. + + Row order is never significant and duplicate rows are preserved. When + column names are unavailable on the actual side, comparison falls back to + positional values (the historical behaviour). """ - reordered = _reorder_columns( - list(actual_columns or []), actual_rows, expected_columns + actual_columns = list(actual_columns or []) + if actual_columns and expected_columns: + if set(actual_columns) == set(expected_columns): + reordered = _reorder_columns(actual_columns, actual_rows, expected_columns) + return _canonical_rows(reordered) == _canonical_rows(expected_rows) + if len(actual_columns) == len(expected_columns) and not ( + set(actual_columns) & set(expected_columns) + ): + # Same width, completely different names: not a projection + # difference. Keep the positional comparison so a renamed column + # holding the same values is not newly penalised. + return _canonical_rows(actual_rows) == _canonical_rows(expected_rows) + actual_projection, expected_projection = _shared_column_projection( + actual_columns, expected_columns + ) + if not actual_projection: + # Different widths AND no shared column name: the two results do not + # describe the same quantity, so this is a genuine mismatch. + return False + actual_rows = [[row[index] for index in actual_projection] for row in actual_rows] + expected_rows = [ + [row[index] for index in expected_projection] for row in expected_rows + ] + return _canonical_rows(actual_rows) == _canonical_rows(expected_rows) + + +def _shared_column_projection( + actual_columns: list[str], + expected_columns: list[str], +) -> tuple[list[int], list[int]]: + """Index positions of the shared column names, in the expected order.""" + + actual_positions: list[int] = [] + expected_positions: list[int] = [] + for position, column in enumerate(expected_columns): + if column in actual_columns: + expected_positions.append(position) + actual_positions.append(actual_columns.index(column)) + return actual_positions, expected_positions + + +def _projection_difference( + actual_columns: list[str] | None, + expected_columns: list[str], +) -> tuple[list[str], list[str]]: + """Column names the actual result has additionally / is missing. + + Reported as its own signal so projection tolerance cannot hide a wrong + query shape: a result that shares no column name is not equivalent at all, + and a partial overlap is still visible in the report. + """ + + actual = list(actual_columns or []) + if not actual or not expected_columns: + return [], [] + return ( + [column for column in actual if column not in expected_columns], + [column for column in expected_columns if column not in actual], ) - return _canonical_rows(reordered) == _canonical_rows(expected_rows) def _reorder_columns( @@ -538,6 +826,237 @@ def _canonical_value(value: Any) -> Any: return value +def _recording_model_factory( + usage_log: list[dict[str, int]], +): + """Wrap the model factory so provider-reported usage survives the run. + + The evaluator used to estimate tokens from the *question text*, which is a + few dozen characters — so it reported roughly 10 input tokens while the real + prompt is thousands. The provider already reports normalized usage + (``last_usage`` with ``estimated=False`` when the API returned counts), so it + is recorded here and preferred over the heuristic. + """ + + def factory(config: Any): + return wrap_provider_for_usage(ModelFactory.create(config), usage_log) + + return factory + + +def wrap_provider_for_usage(model: Any, usage_log: list[dict[str, int]]) -> Any: + """Record provider-reported usage for every model call on ``model``. + + Split out from the factory so it can be tested directly against an adapter: + it must satisfy the *current* provider contract, and a wrapper that silently + falls behind that contract makes every evaluated run fail before any spend. + + ``timeout`` is forwarded rather than swallowed — it carries the run's + remaining model deadline (feat-012). + """ + + class _UsageRecordingModel(type(model)): # type: ignore[misc] + def __init__(self, inner: Any) -> None: + self._inner = inner + + def __getattr__(self, name: str) -> Any: + return getattr(self._inner, name) + + def generate_with_messages(self, messages, json_mode=False, timeout=None): + result = self._inner.generate_with_messages( + messages, json_mode=json_mode, timeout=timeout + ) + usage = getattr(self._inner, "last_usage", None) + if usage is not None: + prompt = int(getattr(usage, "prompt_tokens", 0) or 0) + completion = int(getattr(usage, "completion_tokens", 0) or 0) + if prompt or completion: + usage_log.append( + { + "input_tokens": prompt, + "output_tokens": completion, + "estimated": bool(getattr(usage, "estimated", False)), + } + ) + return result + + return _UsageRecordingModel(model) + + + return factory + + +def _measured_usage( + result: dict[str, Any], + usage_log: list[dict[str, int]] | None, +) -> dict[str, int] | None: + """Sum the usage recorded for this case, or None when nothing was reported. + + Returns None when the provider reported nothing, and for estimates that the + adapter had to synthesize — a char-count guess must not be presented as + measured usage. + """ + + recorded = result.get("usage") + entries = [recorded] if isinstance(recorded, dict) else list(usage_log or []) + measured = [item for item in entries if item and not item.get("estimated")] + if not measured: + return None + return { + "input_tokens": sum(int(item.get("input_tokens", 0)) for item in measured), + "output_tokens": sum(int(item.get("output_tokens", 0)) for item in measured), + } + + +#: Markers of a transport / account problem rather than a model-quality problem. +_ENVIRONMENT_ERROR_MARKERS: tuple[str, ...] = ( + "insufficient balance", + "insufficient_quota", + "quota exceeded", + "rate limit", + "too many requests", + "error code: 402", + "error code: 429", + "error code: 401", + "error code: 403", + "unauthorized", + "authentication", + "connection error", + "connect timeout", + "read timeout", + "service unavailable", + "bad gateway", + "gateway timeout", + "no api key is configured", +) + + +def _environment_error_reason(exc: BaseException) -> str | None: + """The matched environment-failure marker, or None when it is not one.""" + + text = str(exc).lower() + for marker in _ENVIRONMENT_ERROR_MARKERS: + if marker in text: + return marker + return None + + +#: Failure buckets. A single aggregate accuracy number over a case set this +#: varied is misleading, because it mixes "the system answered the wrong +#: question" with "a deterministic gate refused the answer" and "the system +#: correctly asked for clarification". Those need different responses, so they are +#: separated here rather than collapsed into one rate. +FAILURE_CLASSES: tuple[str, ...] = ( + "context_lost", + "clarification_requested", + "governance_refusal", + "retry_exhausted", + "policy_blocked", + "wrong_projection", + "wrong_value", + "execution_failed", + "environment_error", +) + +#: Textual markers of a deterministic governance refusal in an error chain. +_GOVERNANCE_REFUSAL_MARKERS: tuple[str, ...] = ( + "semantic sql validation failed", + "unsupported metric dimension combination", + "governance rejected", +) + + +def classify_outcome(result: dict[str, Any]) -> str | None: + """Bucket one case result. ``None`` means the case passed. + + Order matters. A governance refusal is reported as ``governance_refusal`` even + when it *led* to retry exhaustion, because the refusal is the cause and the + exhaustion is the symptom — reporting only the symptom would hide that the + deterministic layer, not the model, ended the run. + """ + + if result.get("environment_error"): + return "environment_error" + if result.get("semantic_correct"): + return None + + error = str(result.get("error") or "").lower() + status = str(result.get("status") or "") + + # Structural causes are classified on their own terms, even for a + # context-dependent case. A first version of this function swept every failure + # of a `requires_context` case into `context_lost`, which was misleading: a + # deterministic gate refusing the answer, or the system correctly asking for + # clarification, is NOT a context failure. Only a confident wrong answer is. + if status == "needs_clarification": + # The system asked instead of answering. Checked before the + # comparison-based buckets because a clarification run can also show a + # column difference (the clause that ran produced rows), and asking is the + # headline outcome. + return "clarification_requested" + if any(marker in error for marker in _GOVERNANCE_REFUSAL_MARKERS): + return "governance_refusal" + if "retry_limit" in error or "maximum sql retries" in error: + return "retry_exhausted" + if status == "blocked" or "policy" in error: + return "policy_blocked" + if result.get("requires_context"): + # SQL ran to completion and differs from the reference: a confident wrong + # answer on a question that needed the prior turn. + return "context_lost" + if status != "success": + # SQL never produced a comparable result for a reason not covered above. + return "execution_failed" + # SQL ran and produced a result that differs from the reference. + if result.get("extra_columns") or result.get("missing_columns"): + return "wrong_projection" + return "wrong_value" + + +def failure_detail(result: dict[str, Any]) -> str | None: + """The concrete mechanism behind a bucket, for attribution inside it. + + ``context_lost`` deliberately absorbs every failure mode of a + context-dependent case; this keeps the mechanism visible so the bucket does + not become a place where causes go to hide. + """ + + if not result.get("failure_class"): + return None + error = str(result.get("error") or "").lower() + if "semantic sql validation failed" in error: + return "semantic_validator_refusal" + if "unsupported metric dimension combination" in error: + return "metric_search_refusal" + if "retry_limit" in error or "maximum sql retries" in error: + return "retry_exhausted" + if "fix response repeated" in error or "fix response must contain" in error: + return "repair_loop_stalled" + status = str(result.get("status") or "") + if status == "needs_clarification": + return "clarification_requested" + if status == "blocked": + return "policy_blocked" + if status != "success": + return f"status_{status}" + if result.get("extra_columns") or result.get("missing_columns"): + return "wrong_projection" + return "wrong_value" + + +def is_environment_error(exc: BaseException) -> bool: + """Whether a failure is an environment/transport problem, not model quality. + + The first baseline attempt was contaminated by ``Error code: 402 - + Insufficient Balance`` mid-run: those cases were recorded as model failures + and lowered ``sql_execution_success_rate`` to 0.75. Re-running with a funded + account produced 1.00 for the same code, which is the whole point — the + number must not move because an account ran dry. + """ + + return _environment_error_reason(exc) is not None + + def _estimate_tokens(text: str) -> int: return math.ceil(len(text) / 4) if text else 0 @@ -553,12 +1072,39 @@ def _percentile(values: list[float], percentile: float) -> float | None: return round(values[lower] + (values[upper] - values[lower]) * (position - lower), 3) +def _failure_class_counts(results: list[dict[str, Any]]) -> dict[str, int]: + """Count failures per bucket, so one number cannot hide three causes.""" + + counts: dict[str, int] = {} + for item in results: + bucket = item.get("failure_class") or classify_outcome(item) + if bucket: + counts[bucket] = counts.get(bucket, 0) + 1 + return dict(sorted(counts.items(), key=lambda pair: (-pair[1], pair[0]))) + + def _rate(values: list[bool]) -> float | None: return round(sum(values) / len(values), 6) if values else None -def _build_report(results: list[dict[str, Any]]) -> dict[str, Any]: +def _build_report( + results: list[dict[str, Any]], + *, + parallel_candidates_override: int | None = None, + cases: list[dict[str, Any]] | None = None, + default_semantic_model: str | None = None, + default_database: str | None = None, +) -> dict[str, Any]: query_results = [item for item in results if item["expected_outcome"] == "query"] + # Environment failures (no balance, rate limit, transport) are excluded from + # the accuracy denominators and counted separately, so an account running dry + # can never be read as a capability regression. + environment_failures = [ + item for item in query_results if item.get("environment_error") + ] + scored_results = [ + item for item in query_results if not item.get("environment_error") + ] probe_results = [ item for item in results if item["expected_outcome"] == "policy_rejection" ] @@ -626,16 +1172,70 @@ def _build_report(results: list[dict[str, Any]]) -> dict[str, Any]: category: sum(item["category"] == category for item in results) for category in sorted({str(item["category"]) for item in results}) }, + # Explicit buckets so a coverage gap is visible in the report instead of + # having to be derived by hand. `multi_turn` counts cases that carry + # follow_up_context; `requires_context` counts the stricter subset whose + # question cannot be answered without it (an elliptical or referential + # follow-up), which is the bucket that actually measures context handling + # rather than merely exercising a multi-turn code path. + "coverage_buckets": { + "total": len(results), + "query": len(query_results), + "policy_rejection": len(probe_results), + "multi_turn": sum(1 for item in results if item.get("multi_turn")), + "requires_context": sum( + 1 for item in results if item.get("requires_context") + ), + "compound": sum(1 for item in results if item.get("compound")), + }, + "governance_coverage": _governance_coverage( + cases or [], default_semantic_model, default_database + ), + "failure_classes": _failure_class_counts(results), "per_domain": per_domain, "metrics": { + # Environment failures are reported, counted and excluded from every + # accuracy rate below: an unfunded account must not look like a + # capability regression. + "environment_error_cases": len(environment_failures), + "environment_error_ids": [ + item.get("id") for item in environment_failures + ], + "scored_case_count": len(scored_results), "sql_execution_success_rate": _rate( - [bool(item["execution_success"]) for item in query_results] + [bool(item["execution_success"]) for item in scored_results] ), "semantic_correctness_rate": _rate( - [bool(item["semantic_correct"]) for item in query_results] + [bool(item["semantic_correct"]) for item in scored_results] + ), + # Whether the token numbers come from provider usage or from the + # character heuristic. "estimated" means the figures cannot support a + # cost bound. + "token_source": ( + "measured" + if any(item.get("token_source") == "measured" for item in results) + else "estimated" + ), + "token_source_measured_cases": sum( + 1 for item in results if item.get("token_source") == "measured" + ), + # Projection tolerance (see module docstring): semantic_correctness_rate + # compares the shared columns, so a differing projection is reported + # here instead of being silently absorbed into the correctness rate. + "projection_difference_rate": _rate( + [ + bool(item.get("extra_columns") or item.get("missing_columns")) + for item in query_results + if item.get("columns_compared") + ] + ), + "projection_difference_cases": sum( + 1 + for item in query_results + if item.get("extra_columns") or item.get("missing_columns") ), "sql_exact_match_rate": _rate( - [bool(item["sql_exact_match"]) for item in query_results] + [bool(item["sql_exact_match"]) for item in scored_results] ), "policy_rejection_precision": round( true_positives / (true_positives + false_positives), 6 @@ -664,6 +1264,19 @@ def _build_report(results: list[dict[str, Any]]) -> dict[str, Any]: [float(item["latency_ms"]) for item in query_results] ), "average_oracle_latency_ms": _average(oracle_latencies), + # Engine time of the generated query. Bounded by the oracle latency, + # i.e. microseconds-to-milliseconds against a multi-second run — the + # field that settles whether "a faster candidate" could ever matter. + "average_sql_execution_ms": _average( + [ + float(item["sql_execution_ms"]) + for item in query_results + if item.get("sql_execution_ms") is not None + ] + ), + "sql_execution_measured_cases": sum( + 1 for item in query_results if item.get("sql_execution_ms") is not None + ), "average_estimated_input_tokens": _average( [int(item["estimated_input_tokens"]) for item in results] ), @@ -679,7 +1292,20 @@ def _build_report(results: list[dict[str, Any]]) -> dict[str, Any]: else None ), "candidate_selection_cases": len(first_correct), + # Stated so an ablation run is self-describing: None means the + # per-case candidate_selection flag was honoured (which tracks case + # category), an integer means every query case was forced to that + # candidate count. + "parallel_candidates_override": parallel_candidates_override, }, + "multi_candidate_method": ( + "candidate_selection_uplift compares the selected candidate against the first " + "candidate over the same cases. NOTE: the per-case candidate_selection flag in " + "the gold set correlates perfectly with category (multi_table/metric set it, " + "single_table/time do not), so a candidate-vs-no-candidate comparison across " + "cases is confounded with task difficulty. Use --parallel-candidates to force " + "one candidate count across identical cases." + ), "token_cost_method": "heuristic char_count/4; configure per-million prices for estimated USD only", "policy_evaluation_method": ( "probes run through the real SQLPolicyEngine with the case's " @@ -708,9 +1334,13 @@ def main() -> int: model_override=args.model, ) evaluation_config = isolated_config(base_config, args.asset_state_root) + usage_log: list[dict[str, int]] = [] report = evaluate_cases( cases, - service=AgentService(config_loader=lambda **_: evaluation_config), + service=AgentService( + config_loader=lambda **_: evaluation_config, + llm_factory=_recording_model_factory(usage_log), + ), environment=EvaluationEnvironment(args.asset_state_root), default_database=args.database, default_semantic_model=args.semantic_model, @@ -719,6 +1349,9 @@ def main() -> int: model=args.model, input_cost_per_million=args.input_cost_per_million, output_cost_per_million=args.output_cost_per_million, + usage_log=usage_log, + parallel_candidates_override=args.parallel_candidates, + skill_selection=[] if args.skill_mode == "off" else None, ) report["state_isolation"] = { "mode": "isolated", diff --git a/scripts/generate_context_gold.py b/scripts/generate_context_gold.py new file mode 100644 index 0000000..fdab0ea --- /dev/null +++ b/scripts/generate_context_gold.py @@ -0,0 +1,482 @@ +"""Generate the context-dependent and compound NL2SQL gold set. + +Why this file exists +-------------------- +The original 120-case gold set has 12 cases carrying ``follow_up_context``, but **none of them +need it**: "Now count anime assigned to the Action genre" states its own metric and filter, so a +model that never reads the session history answers it correctly. Multi-turn context handling was +therefore effectively unmeasured (0 real cases), which makes any claim about it unfounded. + +Two further problems were found while building this set, and both are why the references here are +deliberately written against the governed semantic model rather than against what the data +supports: + +1. A reference SQL that joins a fact to a dimension through a path the semantic model does not + declare (e.g. ``fact_watch_session`` -> ``bridge_anime_genre`` -> ``dim_genre``) is rejected by + ``SemanticSQLValidator`` as a fan-out/join-key violation. A case whose own reference cannot pass + the governed chain measures nothing. +2. A dimension outside the metric's ``allowed_dimensions`` (e.g. ``studio.tier`` for + ``watch_hours``) is refused by ``metric_search`` with "Unsupported metric dimension + combination". Same problem. + +Both were caught by this generator's validation and by running the references through the real +governed tool, not by inspection. + +Case shape +---------- +* ``context_dependent`` — the follow-up is elliptical ("Break that down by region.") or + referential ("Why is that so high?"). Without the prior turn the question has no metric or its + referent is undefined. ``requires_context: true``. +* ``compound`` — two asks in one question. The reference encodes the measurement half; a complete + answer must not silently drop the second ask. ``compound: true``. +* ``causal`` — a why/how-come question. No single SQL is correct, so the reference is the neutral + current-state aggregate a defensible answer must be consistent with. ``causal: true``. These + measure whether the system grounds an answer or invents a cause. + +Regenerate with:: + + python scripts/generate_context_gold.py + +The generator refuses to write unless every reference (a) executes read-only through the governed +``DatabaseTool``, (b) returns a column set, (c) returns at least one row, and (d) passes +``SemanticSQLValidator`` when the case is governed. A reference that cannot satisfy the system's +own governance is not a usable gold answer. +""" + +from __future__ import annotations + +import json +import sys +from collections import Counter +from pathlib import Path +from typing import Any + + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +if str(PROJECT_ROOT) not in sys.path: + sys.path.insert(0, str(PROJECT_ROOT)) + +OUTPUT = PROJECT_ROOT / "evaluation" / "gold" / "context_and_compound.jsonl" + +BASE = { + "database": "sample_data/anime_streaming/anime_streaming.sqlite", + "semantic_model": "sample_data/anime_streaming/semantic_model.yml", + "sql_policy": "sample_data/anime_streaming/sql_policy.yml", +} + +#: Joins the semantic model declares as safe, expressed as SQL. Every reference +#: below is built from these and nothing else. +JOIN_STUDIO = ( + "JOIN dim_episode e ON e.episode_id = w.episode_id " + "JOIN dim_anime a ON a.anime_id = e.anime_id " + "JOIN dim_studio s ON s.studio_id = a.studio_id" +) +JOIN_USER = ( + "JOIN dim_episode e ON e.episode_id = w.episode_id " + "JOIN dim_anime a ON a.anime_id = e.anime_id " + "JOIN dim_user u ON u.user_id = w.user_id" +) +#: Raw, unrounded expression. The evaluator declares a 10-decimal float policy, so a +#: reference that rounds first destroys the very precision the comparison relies on: +#: a model answering 3937.6125 would be scored wrong against a pre-rounded 3937.61. +#: References therefore state the expression faithfully and let the comparator apply +#: its own tolerance. +WATCH_HOURS = "SUM(w.watch_seconds) / 3600.0" + + +def _case( + case_id: str, + category: str, + question: str, + sql: str, + *, + context: list[str], + requires_context: bool = False, + compound: bool = False, + causal: bool = False, + notes: str = "", +) -> dict[str, Any]: + return { + "id": case_id, + **BASE, + "domain": "anime_streaming", + "category": category, + "expected_outcome": "query", + "question": question, + "expected_sql": sql, + "follow_up_context": context, + # Candidate generation is off: these cases measure context and completeness, + # not candidate selection, and the candidate mechanism was shown to add + # latency without accuracy benefit (docs/evaluation_baselines.md). + "candidate_selection": False, + "requires_context": requires_context, + "compound": compound, + "causal": causal, + "notes": notes, + } + + +# --------------------------------------------------------------------------- cases + +# Every dimension below is taken from the metric's allowed_dimensions, and every +# join from the declared join_paths. See the module docstring. +_CONTEXT_DEPENDENT: list[dict[str, Any]] = [ + _case( + "ctx_01", + "context_dependent", + "Break that down by playback region.", + f"SELECT w.playback_region, {WATCH_HOURS} AS watch_hours " + "FROM fact_watch_session w GROUP BY w.playback_region " + "ORDER BY watch_hours DESC", + context=["Show total watch hours."], + requires_context=True, + notes="Metric only in the prior turn. playback_region is an allowed dimension.", + ), + _case( + "ctx_02", + "context_dependent", + "And by anime release year?", + f"SELECT a.release_year, {WATCH_HOURS} AS watch_hours " + f"FROM fact_watch_session w JOIN dim_episode e ON e.episode_id = w.episode_id " + f"JOIN dim_anime a ON a.anime_id = e.anime_id " + "GROUP BY a.release_year ORDER BY a.release_year", + context=["Show total watch hours."], + requires_context=True, + notes="Elliptical: 'And by ...' carries no metric. release_year is allowed.", + ), + _case( + "ctx_03", + "context_dependent", + "Only APAC.", + f"SELECT {WATCH_HOURS} AS watch_hours FROM fact_watch_session w " + "WHERE w.playback_region = 'APAC'", + context=["Show total watch hours by playback region."], + requires_context=True, + notes="Neither metric nor aggregation named; the referent is the prior request.", + ), + _case( + "ctx_04", + "context_dependent", + "Only Major studios.", + f"SELECT {WATCH_HOURS} AS watch_hours FROM fact_watch_session w " + f"{JOIN_STUDIO} WHERE s.studio_tier = 'Major'", + context=["Show total watch hours by studio."], + requires_context=True, + notes="Filter-only follow-up on studio.tier, which is allowed for watch_hours.", + ), + _case( + "ctx_05", + "context_dependent", + "Top 5 only.", + f"SELECT a.title, {WATCH_HOURS} AS watch_hours FROM fact_watch_session w " + "JOIN dim_episode e ON e.episode_id = w.episode_id " + "JOIN dim_anime a ON a.anime_id = e.anime_id " + "GROUP BY a.title ORDER BY watch_hours DESC LIMIT 5", + context=["Show watch hours by anime title."], + requires_context=True, + notes="A bare ranking follow-up: neither metric nor entity is restated.", + ), + _case( + "ctx_06", + "context_dependent", + "Why is that so high?", + f"SELECT {WATCH_HOURS} AS apac_watch_hours FROM fact_watch_session w " + "WHERE w.playback_region = 'APAC'", + context=[ + "Show total watch hours by playback region.", + "APAC is the highest region.", + ], + requires_context=True, + causal=True, + notes="'that' resolves only against the previous turns; no metric is named.", + ), + _case( + "ctx_07", + "context_dependent", + "Same thing but for Europe.", + f"SELECT {WATCH_HOURS} AS watch_hours FROM fact_watch_session w " + "WHERE w.playback_region = 'Europe'", + context=["Show total watch hours by playback region. APAC is highest."], + requires_context=True, + notes="'Same thing' requires the prior metric AND aggregation.", + ), + _case( + "ctx_08", + "context_dependent", + "Compare that with the other release cohort.", + f"SELECT a.release_year, {WATCH_HOURS} AS watch_hours " + "FROM fact_watch_session w " + "JOIN dim_episode e ON e.episode_id = w.episode_id " + "JOIN dim_anime a ON a.anime_id = e.anime_id " + "WHERE a.release_year >= 2012 " + "GROUP BY a.release_year ORDER BY a.release_year", + context=["Show watch hours for anime released in 2019."], + requires_context=True, + notes="Comparison target and metric both come from context. Years present in the sample " + "data are 1998/2005/2012/2019.", + ), + _case( + "ctx_09", + "context_dependent", + "Break that down by product category.", + "SELECT p.product_category, SUM(i.net_amount_usd) AS net_revenue " + "FROM fact_merch_order_item i " + "JOIN dim_merch_product p ON p.product_id = i.product_id " + "GROUP BY p.product_category ORDER BY net_revenue DESC", + context=["Show total merchandise net revenue."], + requires_context=True, + notes="Second domain, to avoid tuning to one shape. merch_product.category is allowed.", + ), + _case( + "ctx_10", + "context_dependent", + "Only rewatched sessions.", + f"SELECT {WATCH_HOURS} AS watch_hours FROM fact_watch_session w " + "WHERE w.rewatch_flag = 1", + context=["Show total watch hours."], + requires_context=True, + ), + _case( + "ctx_11", + "context_dependent", + "Now average rating instead.", + "SELECT CAST(SUM(r.score) AS REAL) / NULLIF(COUNT(*), 0) AS average_rating " + "FROM fact_rating r", + context=["Show the total number of ratings."], + requires_context=True, + notes="Metric replacement on the same entity, which the question leaves implicit.", + ), + _case( + "ctx_12", + "context_dependent", + "Just that one.", + f"SELECT a.title, {WATCH_HOURS} AS watch_hours FROM fact_watch_session w " + "JOIN dim_episode e ON e.episode_id = w.episode_id " + "JOIN dim_anime a ON a.anime_id = e.anime_id " + "GROUP BY a.title ORDER BY watch_hours DESC LIMIT 1", + context=[ + "Show watch hours by anime title.", + "The top title is the one we want to drill into.", + ], + requires_context=True, + notes="Referential restriction with no metric, no dimension and no entity restated.", + ), +] + +_COMPOUND: list[dict[str, Any]] = [ + _case( + "cmp_01", + "compound", + "How many anime are there in total, and how many ratings do they have?", + "SELECT (SELECT COUNT(*) FROM dim_anime) AS anime_count, " + "(SELECT COUNT(*) FROM fact_rating) AS rating_count", + context=[], + compound=True, + notes="Two aggregates in one question; answering only the first is incomplete.", + ), + _case( + "cmp_02", + "compound", + "Show total watch hours and the number of distinct viewers.", + f"SELECT {WATCH_HOURS} AS watch_hours, " + "COUNT(DISTINCT w.user_id) AS unique_viewers FROM fact_watch_session w", + context=[], + compound=True, + notes="Volume plus reach.", + ), + _case( + "cmp_03", + "compound", + "Compare average rating and average review length by anime format, and tell me " + "which format does best.", + "SELECT a.content_format, " + "CAST(SUM(r.score) AS REAL) / NULLIF(COUNT(*), 0) AS average_rating, " + "AVG(r.review_length) AS average_review_length " + "FROM fact_rating r JOIN dim_anime a ON a.anime_id = r.anime_id " + "GROUP BY a.content_format ORDER BY average_rating DESC", + context=[], + compound=True, + notes="The measurement half is the oracle; 'which does best' is a judgement the answer " + "must state rather than silently omit.", + ), + _case( + "cmp_04", + "compound", + "What is merchandise net revenue by product category, and which category should we " + "invest in next quarter?", + "SELECT p.product_category, SUM(i.net_amount_usd) AS net_revenue " + "FROM fact_merch_order_item i " + "JOIN dim_merch_product p ON p.product_id = i.product_id " + "GROUP BY p.product_category ORDER BY net_revenue DESC", + context=[], + compound=True, + notes="Measurement plus a recommendation that the data does not settle.", + ), + _case( + "cmp_05", + "compound", + "List the top 5 anime by watch hours, and give me the session counts too.", + f"SELECT a.title, {WATCH_HOURS} AS watch_hours, COUNT(*) AS session_count " + "FROM fact_watch_session w " + "JOIN dim_episode e ON e.episode_id = w.episode_id " + "JOIN dim_anime a ON a.anime_id = e.anime_id " + "GROUP BY a.title ORDER BY watch_hours DESC LIMIT 5", + context=[], + compound=True, + notes="Ranking plus a second measure; dropping the counts is a partial answer.", + ), + _case( + "cmp_06", + "compound", + "How many users signed up in 2024, and how many of them activated a subscription?", + "SELECT (SELECT COUNT(*) FROM dim_user u WHERE u.signup_date_key >= 20240101 " + "AND u.signup_date_key <= 20241231) AS signups_2024, " + "(SELECT COUNT(DISTINCT s.user_id) FROM fact_subscription s " + "JOIN dim_user u ON u.user_id = s.user_id " + "WHERE u.signup_date_key >= 20240101 AND u.signup_date_key <= 20241231) " + "AS activated_subscribers", + context=[], + compound=True, + notes="Funnel: two dependent counts in one ask.", + ), +] + +_CAUSAL: list[dict[str, Any]] = [ + _case( + "cau_01", + "causal", + "Why did watch hours drop last month?", + "SELECT d.year, d.month_number, " + "SUM(w.watch_seconds) / 3600.0 AS watch_hours " + "FROM fact_watch_session w " + "JOIN dim_date d ON d.date_key = w.watch_date_key " + "GROUP BY d.year, d.month_number ORDER BY d.year, d.month_number", + context=[], + causal=True, + notes="No cause exists in the data. A defensible answer states the observed trend and " + "either asks what changed or lists candidate explanations as hypotheses with evidence.", + ), + _case( + "cau_02", + "causal", + "What is driving the growth in merchandise revenue?", + "SELECT d.year, d.month_number, SUM(i.net_amount_usd) AS net_revenue " + "FROM fact_merch_order_item i " + "JOIN fact_merch_order o ON o.order_id = i.order_id " + "JOIN dim_date d ON d.date_key = o.order_date_key " + "GROUP BY d.year, d.month_number ORDER BY d.year, d.month_number", + context=[], + causal=True, + notes="Same shape: measures the trend, must not invent a driver.", + ), + _case( + "cau_03", + "causal", + "Users who watch more rate higher, right?", + "SELECT CAST(SUM(r.score) AS REAL) / NULLIF(COUNT(*), 0) " + "AS overall_average_rating FROM fact_rating r", + context=[], + causal=True, + notes="Leading question with a false presupposition and no comparison in the data. A " + "defensible answer declines to confirm it and states what would be needed to test it.", + ), +] + + +def build() -> list[dict[str, Any]]: + return [*_CONTEXT_DEPENDENT, *_COMPOUND, *_CAUSAL] + + +def validate(cases: list[dict[str, Any]]) -> list[str]: + """Validate every reference through the real governed chain.""" + + from queryforge.domain.semantic import SemanticModelLoader + from queryforge.domain.semantic.sql_validator import SemanticSQLValidator + from queryforge.infrastructure.db.sqlite_connector import SQLiteConnector + from queryforge.infrastructure.tools.database_tool import DatabaseTool + + problems: list[str] = [] + database = PROJECT_ROOT / BASE["database"] + model_path = PROJECT_ROOT / BASE["semantic_model"] + with SQLiteConnector(str(database)) as connector: + tool = DatabaseTool(connector) + schemas = [tool.describe_table(name) for name in tool.list_tables()] + for case in cases: + sql = case["expected_sql"] + try: + result = tool.execute_sql(sql) + except Exception as exc: # noqa: BLE001 - any reference failure is a finding + problems.append(f"{case['id']}: reference SQL failed: {exc}") + continue + if not result.columns: + problems.append(f"{case['id']}: reference returned no columns") + if not result.rows: + problems.append( + f"{case['id']}: reference returned zero rows (unanswerable)" + ) + if case.get("requires_context") and not case.get("follow_up_context"): + problems.append(f"{case['id']}: requires_context with empty context") + + # Governance check: build a governed context so the semantic validator + # can run exactly as ExecuteSqlNode would run it. + try: + from queryforge.core.schemas.models import Context, SqlTask + + semantic = SemanticModelLoader.load_and_validate( + model_path, schemas, case["question"] + ) + matches = SemanticModelLoader.match_metrics( + semantic.model, case["question"] + ) + governed = Context( + task=SqlTask( + question=case["question"], database_path=str(database) + ), + semantic_model=semantic, + metric_matches=[ + m for m in matches if case.get("expected_sql") + ], + ) + validator = SemanticSQLValidator.for_context(governed) + if validator is not None: + verdict = validator.validate(sql) + if verdict.status == "violation": + problems.append( + f"{case['id']}: reference violates governance " + f"({','.join(verdict.rule_names)}): {verdict.unsupported_reason or verdict.error_message()}" + ) + except Exception as exc: # noqa: BLE001 - surface, do not mask + problems.append(f"{case['id']}: governance check errored: {exc}") + return problems + + +def main() -> int: + cases = build() + problems = validate(cases) + if problems: + print("REFUSING TO WRITE — reference validation failed:", file=sys.stderr) + for problem in problems: + print(f" - {problem}", file=sys.stderr) + return 1 + OUTPUT.write_text( + "".join(json.dumps(case, ensure_ascii=False) + "\n" for case in cases), + encoding="utf-8", + ) + buckets = Counter( + "requires_context" + if c.get("requires_context") + else "compound" + if c.get("compound") + else "causal" + if c.get("causal") + else "other" + for c in cases + ) + print(f"wrote {len(cases)} cases to {OUTPUT.relative_to(PROJECT_ROOT)}") + for name, count in sorted(buckets.items()): + print(f" {name:18} {count}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_agent_task_gold.py b/tests/test_agent_task_gold.py index b380788..5136e47 100644 --- a/tests/test_agent_task_gold.py +++ b/tests/test_agent_task_gold.py @@ -59,10 +59,21 @@ def test_repeat_is_executed_and_recompute_ignores_stored_verdict(self): self.assertEqual(recompute(report,selected)['task_success_rate'],1) def test_evaluator_does_not_import_runtime_being_graded(self): - for p in (ROOT/'queryforge/evaluation').glob('*.py'): + # The evaluator must not import the machinery it grades, or a change to + # the runtime could silently change what "passing" means. Check plain + # `import` as well as `from ... import`, and recurse into subpackages: + # ast.walk already descends into function/class bodies. + forbidden=('queryforge.application','queryforge.workflow','queryforge.orchestration','queryforge.interfaces') + checked=0 + for p in sorted((ROOT/'queryforge/evaluation').rglob('*.py')): + checked+=1 for node in ast.walk(ast.parse(p.read_text())): if isinstance(node,ast.ImportFrom): - self.assertFalse((node.module or '').startswith(('queryforge.application','queryforge.workflow','queryforge.orchestration','queryforge.interfaces')),str(p)) + self.assertFalse((node.module or '').startswith(forbidden),f'{p}:{node.lineno}') + elif isinstance(node,ast.Import): + for a in node.names: + self.assertFalse(a.name.startswith(forbidden),f'{p}:{node.lineno}') + self.assertGreater(checked,0,'the evaluator package was not found') def test_live_dispatch_does_not_use_planner_or_reference_sql(self): spec=next(s for s in self.specs if s.expected_outcome=='query') diff --git a/tests/test_agent_team_router_orchestrator.py b/tests/test_agent_team_router_orchestrator.py index ff51116..d646792 100644 --- a/tests/test_agent_team_router_orchestrator.py +++ b/tests/test_agent_team_router_orchestrator.py @@ -14,8 +14,16 @@ register_pipeline, ) from queryforge.orchestration.runtime.state_store import AgentTeamStateStore +from queryforge.orchestration.orchestrator.orchestrator import OrchestratorAgent from queryforge.orchestration.schemas import RoutingDecision, TaskState, TaskStatus from queryforge.core.config import Config +from queryforge.core.schemas.models import ( + Context, + ExecutionResult, + ReflectionResult, + SQLContext, + SqlTask, +) from queryforge.domain.semantic import ( SemanticEntity, SemanticMetric, @@ -128,6 +136,86 @@ def test_router_classifies_required_task_types_without_execution_dependencies(se self.assertEqual(decision.task_type, expected) self.assertEqual(decision.entrypoint, "cli") + def test_router_cjk_markers_are_whitespace_insensitive(self): + """D-3: the Chinese marker tables were internally inconsistent. + + ``route`` normalises whitespace with ``" ".join(text.split())``, which + collapses runs but cannot remove a single space inside a Chinese phrase. + ``_SQL_REVIEW_MARKERS`` / ``_TROUBLESHOOT_MARKERS`` therefore spelled each + Chinese marker twice ("审核sql" and "审核 sql") while ``_REPORT_MARKERS``, + ``_METADATA_MARKERS`` and ``_EXPLAIN_MARKERS`` listed only the unspaced + form — so "生成 报告" fell through to the default ``ask_sql`` while + "审核 sql" worked. Marker matching is now whitespace-tolerant for every + marker, so no table can drift again. + """ + router = EntryRouterAgent() + cases = { + "生成报告": "build_report", + "生成 报告": "build_report", + "生成 报告": "build_report", + "制作 报告": "build_report", + "生成 看板": "build_report", + "表结构": "metadata_query", + "表 结构": "metadata_query", + "指标 列表": "metadata_query", + "有哪些表": "metadata_query", + "为什么": "explain_result", + "为 什么": "explain_result", + "解释 结果": "explain_result", + "审核sql": "sql_review", + "审核 sql": "sql_review", + "审 核 sql": "sql_review", + "修复 sql": "troubleshoot_sql", + "sql 报错": "troubleshoot_sql", + } + for question, expected in cases.items(): + with self.subTest(question=question): + self.assertEqual(router.route(question).task_type, expected) + + def test_router_latin_markers_still_classify_as_before(self): + """The whitespace-tolerant matcher must not disturb Latin behaviour.""" + router = EntryRouterAgent() + cases = { + "generate report": "build_report", + "create report": "build_report", + "dashboard": "build_report", + "show tables": "metadata_query", + "database schema": "metadata_query", + "explain the output": "explain_result", + "review sql": "sql_review", + "fix this sql": "troubleshoot_sql", + "how many orders were placed": "ask_sql", + } + for question, expected in cases.items(): + with self.subTest(question=question): + self.assertEqual(router.route(question).task_type, expected) + + def test_a_keyword_inside_an_identifier_does_not_route_the_request(self): + """E-08: "report" matched "sales_report" and hijacked the whole request. + + Marker matching is substring-based, so a bare keyword appearing inside a + table or column name was enough to classify the request: "Show the first 10 + rows of sales_report" became a report-building task, and the run skipped + straight to report generation instead of answering the question. ASCII + markers now match on a word boundary; CJK is unaffected because it has no + such boundary. + """ + router = EntryRouterAgent() + cases = { + "Show the first 10 rows of sales_report": "ask_sql", + "显示 sales_report 的前 10 行": "ask_sql", + "List rows from a metadata_table": "ask_sql", + "How many rows are in the explain_log": "ask_sql", + # Genuine report requests must still route as before. + "build report for monthly sales": "build_report", + "generate report": "build_report", + "dashboard": "build_report", + "生成报告": "build_report", + } + for question, expected in cases.items(): + with self.subTest(question=question): + self.assertEqual(router.route(question).task_type, expected) + def test_router_assigns_simple_and_complex_execution_profiles(self): router = EntryRouterAgent() simple = router.route("List item names") @@ -189,7 +277,15 @@ def test_router_and_pipeline_support_deployment_specific_task_types(self): ("product_analyst", "schema_architect", "delivery"), ) - def test_auto_complex_profile_enables_expensive_stages_only_for_complex_requests(self): + def test_auto_complex_profile_enables_the_tool_loop_but_not_extra_candidates(self): + """Complexity routing boosts the tool loop only. + + It used to also raise parallel_candidates to 2. A controlled ablation over + identical inputs found zero accuracy benefit for the second candidate with + ~20% higher p50 latency (docs/evaluation_baselines.md), and the boost was + often wasted anyway because the tool loop short-circuits the candidate + node. Candidates are now opt-in via the caller. + """ service = self.service() service.ask("List item names", self.options(run_id="simple_profile")) self.assertFalse(self.runners[-1].kwargs["tool_loop_enabled"]) @@ -200,7 +296,8 @@ def test_auto_complex_profile_enables_expensive_stages_only_for_complex_requests self.options(run_id="complex_profile"), ) self.assertTrue(self.runners[-1].kwargs["tool_loop_enabled"]) - self.assertEqual(self.runners[-1].kwargs["parallel_candidates"], 2) + # Not raised implicitly — see the docstring. + self.assertEqual(self.runners[-1].kwargs["parallel_candidates"], 1) service.ask( "List item names", @@ -214,7 +311,17 @@ def test_auto_complex_profile_enables_expensive_stages_only_for_complex_requests self.options(run_id="forced_complex", complexity_mode="complex"), ) self.assertTrue(self.runners[-1].kwargs["tool_loop_enabled"]) + self.assertEqual(self.runners[-1].kwargs["parallel_candidates"], 1) + + def test_an_explicit_candidate_count_is_still_honoured(self): + """The opt-in path must survive: a caller that wants candidates gets them.""" + service = self.service() + service.ask( + "Compare monthly revenue by region and product category with top ranking", + self.options(run_id="explicit_candidates", parallel_candidates=2), + ) self.assertEqual(self.runners[-1].kwargs["parallel_candidates"], 2) + self.assertTrue(self.runners[-1].kwargs["tool_loop_enabled"]) def test_pipeline_is_declarative_and_contains_expected_ask_sql_roles(self): self.assertEqual( @@ -538,7 +645,7 @@ def blocked_state(run_id: str) -> TaskState: # The same first-writer-wins rule covers every terminal status; a run that # is still running is the only one a cancellation may claim. - for status in ("completed", "failed", "cancelled"): + for status in ("completed", "failed", "cancelled", "needs_clarification"): with self.subTest(status=status): terminal = blocked_state(f"terminal_{status}") terminal.status = status @@ -564,6 +671,163 @@ def blocked_state(run_id: str) -> TaskState: self.assertIsNotNone(persisted) self.assertEqual(store.load_state(running.run_id).status, "cancelled") + def test_a_recorded_gate_block_is_not_overwritten_by_a_stale_result_status(self): + """E-02 regression: the QA gate's block must survive the delivery path. + + The completion hook runs *after* ``ReflectiveWorkflow.run()`` has already + returned its output, so the workflow's result dict can still say "success" + while a gate has blocked the run. The orchestrator used to re-derive the + status from that stale dict and write ``completed``, leaving ``state.json`` + saying ``blocked`` (with a recorded reason) and the caller being told the + run succeeded. + + This test drives the real gate and the real derivation rather than + asserting the mapping in isolation. + """ + from queryforge.core.outcomes import derive_outcome + from queryforge.orchestration.agents.data_qa import DataQAAgent + from queryforge.orchestration.orchestrator.orchestrator import ( + _DELIVERY_STATUS_FOR_OUTCOME, + _TASK_STATUS_FOR_OUTCOME, + ) + + root = self.root / ".queryforge" / "runs" + store = AgentTeamStateStore(root) + state = TaskState( + run_id="gate_block_survives_delivery", + entrypoint="test", + classification=RoutingDecision( + task_type="ask_sql", + entrypoint="test", + confidence=1, + reason="test", + pipeline="ask_sql", + ), + status="running", + ) + store.initialize(state) + + # A QA report the deterministic gate must treat as blocking. + DataQAAgent(store).emit( + state, + { + "passed": False, + "row_count": 3, + "columns": ["a"], + "row_count_consistent": False, + "answers_question": False, + "empty_result": False, + "issues": [ + { + "rule": "row_count_mismatch", + "severity": "error", + "reason": "row_count does not match the returned rows", + } + ], + "quality_checks": [], + "quality_status": "skipped", + "reflection": None, + "retry_recommendation": "REGENERATE", + "sql_attempts": [], + }, + ) + + orchestrator = OrchestratorAgent(store) + context = Context( + task=SqlTask(question="q", database_path="/tmp/x.sqlite"), + run_id=state.run_id, + sql_context=SQLContext(sql="SELECT 1", explanation="e", tables_used=[]), + execution_result=ExecutionResult( + columns=["a"], rows=[[1], [2]], row_count=3 + ), + reflection_result=ReflectionResult( + success=True, strategy="SUCCESS", reason="ok" + ), + ) + context.final_output = {"status": "success", "run_id": state.run_id} + state.pending_phases = ["completion"] + orchestrator._completion_hook(state)(context) + + self.assertEqual(state.blocked_reason, "qa_report found severe data quality issue") + self.assertEqual(state.status, "blocked") + + # The workflow still reports its earlier, pre-hook status. + stale_result = {"status": "success", "run_id": state.run_id} + outcome = derive_outcome( + result_status=stale_result.get("status"), + blocked_reason=state.blocked_reason, + ) + self.assertEqual(outcome, "blocked") + self.assertEqual(_TASK_STATUS_FOR_OUTCOME[outcome], "blocked") + # A block is reported as degraded delivery, never as success. + self.assertEqual(_DELIVERY_STATUS_FOR_OUTCOME[outcome], "degraded") + + def test_run_context_carries_identity_and_versions(self): + """feat-008: one object holds run identity and the versions it depends on.""" + from queryforge.core.schemas.models import RunContext + + context = Context( + task=SqlTask(question="q", database_path="/tmp/x.sqlite"), + run_id="qf_identity", + ) + context.task_context["retrieval_scope"] = { + "domain_id": "retail", + "data_version": "2024-01", + } + run_context = RunContext.from_context( + context, semantic_version="7", entrypoint="api" + ) + self.assertEqual(run_context.run_id, "qf_identity") + self.assertEqual(run_context.domain_id, "retail") + self.assertEqual(run_context.data_version, "2024-01") + self.assertEqual(run_context.semantic_version, "7") + self.assertEqual(run_context.entrypoint, "api") + # An undeclared version stays None: "not declared" is not "unchanged". + self.assertIsNone(run_context.policy_version) + + def test_needs_clarification_is_a_terminal_status_that_round_trips(self): + """D-2: a clarification stop must be persistable and not overwritten. + + Before this, a NEED_USER_REVIEW verdict raised and produced no payload; + ``needs_clarification`` was also absent from ``TaskStatus``, so persisting + it would have failed validation on the next read. + """ + from queryforge.application.agent_service import persist_cancelled_outcome + + root = self.root / ".queryforge" / "runs" + store = AgentTeamStateStore(root) + state = TaskState( + run_id="clarify_round_trip", + entrypoint="test", + classification=RoutingDecision( + task_type="ask_sql", + entrypoint="test", + confidence=1, + reason="test", + pipeline="ask_sql", + ), + status="needs_clarification", + current_phase="clarification", + ) + store.initialize(state) + self.assertIn("needs_clarification", TaskStatus.__args__) + + loaded = store.load_state("clarify_round_trip") + self.assertIsNotNone(loaded) + self.assertEqual(loaded.status, "needs_clarification") + self.assertEqual(loaded.current_phase, "clarification") + + # A late disconnect must not relabel a question as a cancellation. + self.assertIsNone( + persist_cancelled_outcome( + state_root=root, + run_id="clarify_round_trip", + reason="client disconnected", + ) + ) + still = store.load_state("clarify_round_trip") + self.assertEqual(still.status, "needs_clarification") + def test_plan_uses_integrated_analysis_candidate_governance_and_ops(self): output = self.real_service().plan( "List item names", diff --git a/tests/test_analysis_planner.py b/tests/test_analysis_planner.py index 16cd7a1..bdde6fa 100644 --- a/tests/test_analysis_planner.py +++ b/tests/test_analysis_planner.py @@ -1267,7 +1267,17 @@ def test_finished_run_is_terminal_and_a_crashed_run_resumes_without_recomputing( ["check_data_quality", "compose_answer", "query_metric", "resolve_metric"], ) self.assertEqual(resumed["recomputed_steps"], []) - self.assertEqual(resumed["budgets"]["usage"]["max_tool_calls"], 0) + # E-04: budget usage is cumulative across attempts of one run, so the + # inherited amount is reported separately from what this attempt spent. + # This assertion previously read ``usage == 0``, which encoded the defect: + # the budget reset on every resume, so a crashed-and-restarted run could + # spend its whole allowance again. Here every step was reused, so this + # attempt spent nothing and the whole figure is inheritance. + self.assertGreater(resumed["inherited_usage"]["max_tool_calls"], 0) + self.assertEqual( + resumed["budgets"]["usage"]["max_tool_calls"], + resumed["inherited_usage"]["max_tool_calls"], + ) self.assertEqual(resumed["answer"]["value"], len(ITEMS)) diff --git a/tests/test_answer_evidence.py b/tests/test_answer_evidence.py index 3eb6666..04c4c4a 100644 --- a/tests/test_answer_evidence.py +++ b/tests/test_answer_evidence.py @@ -652,3 +652,87 @@ def test_summarize_completeness_never_derives_one_from_the_other(self) -> None: if __name__ == "__main__": unittest.main() + + +class SemanticValidationPropagationTest(unittest.TestCase): + """E-11: a semantic verdict must travel with the answer. + + ``SemanticSQLValidator`` reports three states, and only one of them means the + SQL was *proved* to answer the question. ``unsupported`` (an unlisted SQL + shape, an unparsable statement) used to reach a log line and nothing else, so a + run whose business semantics were never checked was delivered exactly like one + that passed every check. + """ + + def _output(self, task_context): + from queryforge.core.schemas.models import ( + Context, + ExecutionResult, + ReflectionResult, + SQLContext, + SqlTask, + ) + from queryforge.workflow.node.output_node import OutputNode + + context = Context( + task=SqlTask(question="q", database_path="/tmp/x.sqlite") + ) + context.task_context.update(task_context) + context.sql_context = SQLContext( + sql="SELECT 1", explanation="e", tables_used=[] + ) + context.execution_result = ExecutionResult( + columns=["a"], rows=[[1]], row_count=1 + ) + context.reflection_result = ReflectionResult( + success=True, strategy="SUCCESS", reason="ok" + ) + OutputNode().execute(context) + return context.final_output + + def test_a_passed_verdict_is_reported_as_verified(self): + output = self._output( + {"semantic_validation": {"status": "passed", "rule_names": []}} + ) + summary = output["semantic_validation"] + self.assertEqual(summary["status"], "passed") + self.assertTrue(summary["verified"]) + # Nothing to warn about, so no completeness note is added. + notes = output["completeness"].get("notes") or [] + self.assertFalse(any("Business semantics" in str(n) for n in notes)) + + def test_unsupported_is_not_verified_and_is_stated_in_the_answer(self): + output = self._output( + { + "semantic_validation": { + "status": "unsupported", + "unsupported_reason": "SQLite SQL could not be parsed", + "rule_names": [], + } + } + ) + summary = output["semantic_validation"] + self.assertEqual(summary["status"], "unsupported") + self.assertFalse(summary["verified"]) + self.assertIn("could not be parsed", summary["reason"]) + notes = output["completeness"].get("notes") or [] + self.assertTrue( + any("Business semantics" in str(n) for n in notes), + "an unproved verdict must be visible in the completeness record too", + ) + + def test_a_violation_is_not_verified(self): + output = self._output( + {"semantic_validation": {"status": "violation", "rule_names": ["fanout"]}} + ) + summary = output["semantic_validation"] + self.assertFalse(summary["verified"]) + self.assertEqual(summary["rules"], ["fanout"]) + + def test_a_run_with_no_verdict_says_so_rather_than_looking_verified(self): + """The common case: a follow-up that matched no governed metric.""" + output = self._output({}) + summary = output["semantic_validation"] + self.assertEqual(summary["status"], "not_run") + self.assertFalse(summary["verified"]) + self.assertIn("no governed metric", summary["reason"]) diff --git a/tests/test_cli_kb_governance.py b/tests/test_cli_kb_governance.py index 69cba21..30a133c 100644 --- a/tests/test_cli_kb_governance.py +++ b/tests/test_cli_kb_governance.py @@ -54,11 +54,17 @@ def __init__(self, vector_store, manifest_path=None) -> None: self.manifest_path = manifest_path self.schemas = None self.sources = None + self.knowledge = None RecordingKnowledgeBaseBuilder.last = self - def rebuild(self, *, history_store, schemas, sources) -> dict: + def rebuild(self, *, history_store, schemas, sources, knowledge=None) -> dict: self.schemas = list(schemas) self.sources = list(sources) + # Recorded so a test can assert the governed path is actually reachable + # from the CLI. Previously `knowledge` was never passed, so + # build_governed_documents (verification tiers, conflict detection, + # holdout isolation) only ever ran in tests. + self.knowledge = knowledge return {"documents": len(self.schemas), "sources": len(self.sources)} @property @@ -157,6 +163,97 @@ def test_rebuild_accepts_an_explicit_sql_policy_flag(self): self.assertNotIn("email", builder.document_text) self.assertIn("user_handle", builder.document_text) + # ---------------------------------------------------------- governed knowledge (E-05) + + def test_rebuild_without_knowledge_leaves_the_governed_path_unused(self): + """Documents the gap this flag closes: with no knowledge source the CLI + never reaches build_governed_documents.""" + config = self.config(sql_policy_path=str(ANIME_POLICY)) + exit_code, builder = self.run_rebuild( + config, + [ + "--rebuild-vector-kb", + "--database", + str(ANIME_DATABASE), + "--kb-source", + str(ANIME_SOURCE), + ], + ) + self.assertEqual(exit_code, 0) + self.assertIsNone(builder.knowledge) + + def test_rebuild_passes_a_structured_knowledge_base_to_the_builder(self): + """``--kb-knowledge`` makes the governed path reachable from the CLI. + + Before this flag there was no way to supply structured knowledge, so the + three-tier verification, conflict detection and holdout isolation in + domain/knowledge/governance.py could only ever run in tests. + """ + config = self.config(sql_policy_path=str(ANIME_POLICY)) + with tempfile.TemporaryDirectory() as directory: + knowledge_path = Path(directory) / "knowledge.json" + knowledge_path.write_text( + json.dumps( + { + "metrics": { + "watch_hours::v1": { + "metric_id": "watch_hours", + "name": "Watch Hours", + "expression": "SUM(fact_watch_session.watch_seconds) / 3600.0", + "aggregation": "sum", + "entity": "watch_session", + "version": "v1", + } + }, + "glossary": {}, + "sources": {}, + "documents": {}, + }, + ensure_ascii=False, + ), + encoding="utf-8", + ) + exit_code, builder = self.run_rebuild( + config, + [ + "--rebuild-vector-kb", + "--database", + str(ANIME_DATABASE), + "--kb-source", + str(ANIME_SOURCE), + "--kb-knowledge", + str(knowledge_path), + ], + ) + self.assertEqual(exit_code, 0) + self.assertIsNotNone(builder.knowledge) + self.assertIn("watch_hours::v1", builder.knowledge.metrics) + + def test_kb_knowledge_requires_rebuild_and_an_existing_file(self): + config = self.config(sql_policy_path=str(ANIME_POLICY)) + # Without --rebuild-vector-kb the flag must be rejected as a usage error. + with patch.object(cli, "load_config", lambda **_: config), patch.object( + sys, "argv", ["queryforge", "--kb-knowledge", "somewhere.json"] + ): + stderr = io.StringIO() + with redirect_stderr(stderr): + exit_code = cli.main() + self.assertEqual(exit_code, 2) + self.assertIn("--kb-knowledge requires --rebuild-vector-kb", stderr.getvalue()) + + # A missing file is also a usage error, not a crash. + exit_code, _builder = self.run_rebuild( + config, + [ + "--rebuild-vector-kb", + "--database", + str(ANIME_DATABASE), + "--kb-knowledge", + "/nonexistent/knowledge.json", + ], + ) + self.assertEqual(exit_code, 2) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_conversation_memory.py b/tests/test_conversation_memory.py index da69eec..e145745 100644 --- a/tests/test_conversation_memory.py +++ b/tests/test_conversation_memory.py @@ -96,6 +96,80 @@ def test_followup_rules_cover_required_operations(self): self.assertEqual(rewrite["reason"], expected_reason) self.assertIn(memory.last_question, str(rewrite["question"])) + def test_followup_rules_do_not_depend_on_whitespace_after_a_cjk_marker(self): + """D-3: "按地区" is one unspaced run and used to be invisible. + + The CJK markers required whitespace after the marker, so "按地区" and + "只看华东" were not recognised while "按 地区" and "只看 华东" were. A bare + follow-up then reached the router as a brand-new question; on the + benchmark's only multi-turn case the model answered by inventing a set of + metrics it had not been asked for. + """ + memory = SessionMemory( + session_id="cjk_rules", + last_question="Show revenue by region for last month", + ) + cases = { + "按地区": "add_dimension", + "按 地区": "add_dimension", + "按照地区": "add_dimension", + "只看华东": "add_filter", + "只看 华东": "add_filter", + "仅看华东": "add_filter", + "加上评分": "add_metric", + "去掉地区": "remove_dimension_or_metric", + } + for question, expected_reason in cases.items(): + with self.subTest(question=question): + rewrite = ProductAnalystAgent.rewrite_followup(question, memory) + self.assertTrue(rewrite["is_followup"]) + self.assertEqual(rewrite["reason"], expected_reason) + + def test_bare_repeat_request_reissues_the_previous_question(self): + """D-3: "再查一次" carries no new content and must re-issue the prior ask.""" + memory = SessionMemory( + session_id="repeat", + last_question="order count", + ) + for question in ("再查一次", "再查一遍", "重新查一次", "再来一次", "再跑一次"): + with self.subTest(question=question): + rewrite = ProductAnalystAgent.rewrite_followup(question, memory) + self.assertTrue(rewrite["is_followup"]) + self.assertEqual(rewrite["reason"], "repeat_previous_request") + self.assertIn("order count", str(rewrite["question"])) + + def test_a_complete_instruction_is_not_absorbed_as_a_followup(self): + """Making the CJK separator optional must not swallow a standalone ask. + + "按门店统计订单数" matches the breakdown marker but carries its own verb, + so it is a fresh question rather than a fragment to attach to the previous + one. Absorbing it would silently discard the user's new instruction. + """ + memory = SessionMemory( + session_id="instruction", + last_question="Show revenue by region for last month", + ) + for question in ( + "按门店统计订单数", + "按地区计算平均评分", + "查询订单总额", + "统计每个类别的订单数", + ): + with self.subTest(question=question): + rewrite = ProductAnalystAgent.rewrite_followup(question, memory) + self.assertFalse(rewrite["is_followup"]) + + def test_reference_followup_matches_cjk_phrases_inside_longer_text(self): + """``\\b`` cannot anchor CJK, so mid-sentence reference phrases were missed.""" + memory = SessionMemory( + session_id="reference", + last_question="Show revenue by region for last month", + ) + for question in ("那个结果再按地区", "上一个结果再查一次", "刚才那个"): + with self.subTest(question=question): + rewrite = ProductAnalystAgent.rewrite_followup(question, memory) + self.assertTrue(rewrite["is_followup"]) + def test_followup_is_rewritten_persisted_and_does_not_store_rows(self): service = self.service() first = service.ask( diff --git a/tests/test_database_tool.py b/tests/test_database_tool.py index 45dfe65..32125cb 100644 --- a/tests/test_database_tool.py +++ b/tests/test_database_tool.py @@ -1,4 +1,7 @@ +import sqlite3 +import tempfile import unittest +from pathlib import Path from queryforge.infrastructure.tools.database_tool import DatabaseTool, UnsafeSQLError @@ -36,3 +39,109 @@ def test_rejects_empty_write_and_multiple_statements(self) -> None: if __name__ == "__main__": unittest.main() + + +class PreviewPolicyOrderingTest(unittest.TestCase): + """E-12: the preview path must not neutralise the policy it reports honouring. + + ``execute_sql_preview`` injected its bounding LIMIT *before* calling the policy + engine, so with ``require_limit: true`` the engine saw a LIMIT the model never + wrote. The audit record then said the statement satisfied the policy while the + statement itself would have been refused by it. + + Preview still tolerates exactly one refusal — ``unbounded_result``, because an + unbounded preview is the point and the injected LIMIT bounds it. Every other + refusal still applies. + """ + + def setUp(self) -> None: + import sqlite3 + import tempfile + + from queryforge.domain.security.sql_policy import SQLSecurityPolicy + from queryforge.infrastructure.db.sqlite_connector import SQLiteConnector + from queryforge.infrastructure.tools.database_tool import DatabaseTool + + self._directory = tempfile.TemporaryDirectory() + database = Path(self._directory.name) / "preview.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE fact_sales (id INTEGER, amount REAL)") + connection.executemany( + "INSERT INTO fact_sales VALUES (?, ?)", + [(index, float(index)) for index in range(50)], + ) + self.connector = SQLiteConnector(str(database)) + self.connector.__enter__() + self.tool = DatabaseTool( + self.connector, SQLSecurityPolicy(require_limit=True, max_limit=100) + ) + + def tearDown(self) -> None: + self.connector.__exit__(None, None, None) + self._directory.cleanup() + + def test_preview_is_bounded_but_the_policy_saw_the_original_statement(self): + sql = "SELECT s.id, s.amount FROM fact_sales s" + result = self.tool.execute_sql_preview(sql, limit=20) + self.assertEqual(len(result.rows), 20) + # The engine consulted on the original, unbounded statement is on the + # record; the final decision is the bounded statement it actually ran. + self.assertIsNotNone(self.tool.last_policy_decision) + + def test_the_same_statement_is_still_refused_outside_the_preview_path(self): + from queryforge.infrastructure.tools.database_tool import UnsafeSQLError + + with self.assertRaises(UnsafeSQLError): + self.tool.execute_sql("SELECT s.id, s.amount FROM fact_sales s") + + def test_preview_does_not_excuse_non_limit_refusals(self): + from queryforge.infrastructure.tools.database_tool import UnsafeSQLError + + with self.assertRaises(UnsafeSQLError) as caught: + self.tool.execute_sql_preview( + "SELECT amount FROM a_table_that_is_not_authorized", limit=20 + ) + self.assertNotIn("unbounded_result", str(caught.exception)) + + +class PolicyDecisionOwnershipTest(unittest.TestCase): + """E-23: the decision record must describe a call this tool actually made. + + ``last_policy_decision`` was a plain public attribute, so a caller could + *assign* a decision it had computed itself (``PlanOutputNode`` did), leaving the + tool reporting an audit record for a call that never happened — and a reader + could pick up an unrelated caller's decision from a shared tool instance. + """ + + def setUp(self) -> None: + from queryforge.infrastructure.db.sqlite_connector import SQLiteConnector + from queryforge.infrastructure.tools.database_tool import DatabaseTool + + self._directory = tempfile.TemporaryDirectory() + database = Path(self._directory.name) / "policy.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE items (id INTEGER, amount REAL)") + connection.execute("INSERT INTO items VALUES (1, 2.5)") + connector = SQLiteConnector(str(database)) + self.addCleanup(self._directory.cleanup) + self.addCleanup(connector.close) + self.tool = DatabaseTool(connector) + + def test_the_decision_cannot_be_overwritten_from_outside(self): + self.tool.execute_sql("SELECT id FROM items LIMIT 1") + recorded = self.tool.last_policy_decision + self.assertIsNotNone(recorded) + self.assertTrue(recorded.allowed) + + with self.assertRaises(AttributeError): + self.tool.last_policy_decision = None + self.assertIs(self.tool.last_policy_decision, recorded) + + def test_the_decision_is_written_only_by_a_real_call(self): + self.assertIsNone(self.tool.last_policy_decision) + with self.assertRaises(UnsafeSQLError) as caught: + self.tool.execute_sql("SELECT amount FROM not_authorized LIMIT 1") + refusal = self.tool.last_policy_decision + self.assertIsNotNone(refusal) + self.assertFalse(refusal.allowed) + self.assertIs(refusal, caught.exception.decision) diff --git a/tests/test_db_adapter_contract.py b/tests/test_db_adapter_contract.py index d37a5d6..ffddcca 100644 --- a/tests/test_db_adapter_contract.py +++ b/tests/test_db_adapter_contract.py @@ -46,6 +46,7 @@ from unittest.mock import patch from queryforge.infrastructure.db import ( + CAPABILITY_REGISTRY, DATE_FUNCTION_VOCABULARY, AdapterCancelledError, AdapterCapabilities, @@ -59,6 +60,7 @@ DuckDBConnector, SQLiteConnector, adapt_connector, + capabilities_for_dialect, normalize_type, normalize_value, ) @@ -473,6 +475,23 @@ def test_type_conversion_normalizes_driver_values(self): # ---- capabilities ---------------------------------------------------- + def test_connector_declaration_is_the_frozen_matrix_entry(self): + """A connector must not re-derive its own capability declaration. + + ``SQLiteConnector`` used to build ``AdapterCapabilities("sqlite")`` from + dataclass defaults, which silently disagreed with ``SQLITE_CAPABILITIES`` + on ``explain_prefix`` (``"EXPLAIN"`` vs ``"EXPLAIN QUERY PLAN"``), + ``date_functions`` and ``integer_division``. Anyone reading the connector + got a different capability set from the one the adapter enforced (E-23). + """ + self.assertIs(self.adapter.capabilities, CAPABILITY_REGISTRY[self.DIALECT]) + self.assertIs( + self.CONNECTOR_CLASS.capabilities, CAPABILITY_REGISTRY[self.DIALECT] + ) + self.assertIs( + self.CONNECTOR_CLASS.capabilities, capabilities_for_dialect(self.DIALECT) + ) + def test_declared_capabilities_are_truthful(self): capabilities = self.adapter.capabilities probes = ( @@ -582,6 +601,7 @@ class SQLiteAdapterConformanceTest(AdapterConformanceMixin, unittest.TestCase): DIALECT = "sqlite" build_fixture = staticmethod(build_sqlite_fixture) open_adapter = staticmethod(sqlite_adapter) + CONNECTOR_CLASS = SQLiteConnector DATE_FUNCTION_PROBE = ( "SELECT strftime('%Y-%m', order_date) AS bucket FROM fact_orders" ) @@ -592,6 +612,7 @@ class DuckDBAdapterConformanceTest(AdapterConformanceMixin, unittest.TestCase): DIALECT = "duckdb" build_fixture = staticmethod(build_duckdb_fixture) open_adapter = staticmethod(duckdb_adapter) + CONNECTOR_CLASS = DuckDBConnector # DuckDB's read-only role additionally refuses ATTACH and config PRAGMA. ENGINE_REFUSED_EXTRA = ("ATTACH 'probe.db' AS other",) DATE_FUNCTION_PROBE = ( diff --git a/tests/test_evaluate_sql.py b/tests/test_evaluate_sql.py index f40201d..e45287e 100644 --- a/tests/test_evaluate_sql.py +++ b/tests/test_evaluate_sql.py @@ -16,6 +16,8 @@ evaluate_sql = importlib.util.module_from_spec(MODULE_SPEC) MODULE_SPEC.loader.exec_module(evaluate_sql) +from queryforge.infrastructure.models.base import BaseModelProvider + class FakeService: def ask(self, question, options): @@ -189,6 +191,239 @@ def _run( default_database=str(database), ) + # ------------------------------------------------- environment vs model (D-5) + + def test_account_failure_is_not_counted_as_a_model_failure(self): + """A depleted balance returned HTTP 402 and was recorded as a model + failure, dragging sql_execution_success_rate from 1.00 to 0.75. It must + be excluded from the accuracy denominators instead.""" + + class OutOfBalanceService: + def ask(self, question, options): + raise RuntimeError( + "Error code: 402 - {'error': {'message': 'Insufficient Balance'}}" + ) + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "numbers.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE numbers (value INTEGER)") + connection.execute("INSERT INTO numbers VALUES (1)") + case = self._query_case(str(database)) + report = self._run( + OutOfBalanceService(), [case], database, root / "assets" + ) + metrics = report["metrics"] + self.assertEqual(metrics["environment_error_cases"], 1) + self.assertEqual(metrics["scored_case_count"], 0) + # No scored cases left, so the rate is unmeasurable rather than 0.0. + self.assertIsNone(metrics["sql_execution_success_rate"]) + self.assertIsNone(metrics["semantic_correctness_rate"]) + self.assertTrue(report["results"][0]["environment_error"]) + self.assertEqual( + report["results"][0]["environment_error_reason"], "insufficient balance" + ) + + def test_environment_error_classifier_covers_transport_and_account_failures(self): + for message in ( + "Error code: 402 - {'message': 'Insufficient Balance'}", + "Error code: 429 - rate limit reached", + "Error code: 401 - unauthorized", + "No API key is configured for provider 'deepseek'", + "ConnectTimeout: connection error", + "503 service unavailable", + ): + with self.subTest(message=message): + self.assertTrue(evaluate_sql.is_environment_error(RuntimeError(message))) + # A genuine model failure must NOT be excused. + for message in ( + "Model response is not valid JSON: Expecting value", + "node=reflect: Human review required", + "no such column: foo", + ): + with self.subTest(message=message): + self.assertFalse(evaluate_sql.is_environment_error(RuntimeError(message))) + + def test_measured_usage_prefers_provider_counts_over_the_heuristic(self): + """A real provider reports estimated=False; a char-count guess must not + be presented as measured usage.""" + measured = {"input_tokens": 4820, "output_tokens": 96, "estimated": False} + self.assertEqual( + evaluate_sql._measured_usage({"usage": measured}, None), + {"input_tokens": 4820, "output_tokens": 96}, + ) + # Estimated values are refused. + estimated = {"input_tokens": 12, "output_tokens": 30, "estimated": True} + self.assertIsNone(evaluate_sql._measured_usage({"usage": estimated}, None)) + # Nothing recorded at all. + self.assertIsNone(evaluate_sql._measured_usage({}, None)) + # Falls back to the run-level log when the per-case record is absent. + self.assertEqual( + evaluate_sql._measured_usage({}, [measured]), + {"input_tokens": 4820, "output_tokens": 96}, + ) + + def test_token_source_is_declared_in_the_report(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "numbers.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE numbers (value INTEGER)") + connection.execute("INSERT INTO numbers VALUES (1)") + report = self._run( + FakeService(), [self._query_case(str(database))], database, root / "assets" + ) + # FakeService reports no usage, so the report must say so rather than + # implying the heuristic figures are billing data. + self.assertEqual(report["metrics"]["token_source"], "estimated") + self.assertEqual(report["metrics"]["token_source_measured_cases"], 0) + + # ------------------------------------------------- multi-candidate ablation + + def test_parallel_candidates_override_forces_the_count_for_every_case(self): + """The per-case flag correlates with category, so an ablation needs to be + able to override it over identical inputs.""" + + seen: list[int] = [] + + class CountingService: + def ask(self, question, options): + seen.append(int(options.parallel_candidates)) + return { + "status": "success", + "rows": [[1]], + "columns": ["value"], + "sql": "SELECT 1", + "model_provider": "fake", + "model": "fake-model", + } + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "numbers.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE numbers (value INTEGER)") + connection.execute("INSERT INTO numbers VALUES (1)") + # One case that would normally ask for 2, one that would not. + cases = [ + self._query_case(str(database), candidate_selection=True), + {**self._query_case(str(database)), "id": "query_2"}, + ] + for forced, expected in ((1, 1), (2, 2), (3, 3)): + seen.clear() + report = evaluate_sql.evaluate_cases( + cases, + service=CountingService(), + environment=evaluate_sql.EvaluationEnvironment(root / f"assets{forced}"), + default_database=str(database), + parallel_candidates_override=forced, + ) + with self.subTest(forced=forced): + self.assertEqual(seen, [expected, expected]) + self.assertEqual( + report["metrics"]["parallel_candidates_override"], forced + ) + + def test_without_override_the_per_case_flag_is_honoured(self): + seen: list[int] = [] + + class CountingService: + def ask(self, question, options): + seen.append(int(options.parallel_candidates)) + return { + "status": "success", + "rows": [[1]], + "columns": ["value"], + "sql": "SELECT 1", + } + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "numbers.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE numbers (value INTEGER)") + connection.execute("INSERT INTO numbers VALUES (1)") + cases = [ + self._query_case(str(database), candidate_selection=True), + {**self._query_case(str(database)), "id": "query_2"}, + ] + report = self._run(CountingService(), cases, database, root / "assets") + # candidate_selection=True -> 2, absent -> 1 + self.assertEqual(seen, [2, 1]) + self.assertIsNone(report["metrics"]["parallel_candidates_override"]) + # The report must warn that this mode is confounded. + self.assertIn("confounded", report["multi_candidate_method"]) + + # ------------------------------------------------- governance coverage (feat-007) + + def test_governance_coverage_reports_cases_the_semantic_layer_cannot_see(self): + """A question matching no governed metric skips SemanticSQLValidator entirely. + + Measured on this repository: 11 of 12 context-dependent cases and 8 of 12 + checked-in multi-turn cases match no governed metric, because metric + matching is term-based and a follow-up like "Break that down by region." + names no metric. The report must state that boundary instead of leaving it + to be discovered by hand. + """ + cases = [ + # Names no metric -> ungoverned, and needs context. + { + "id": "ungoverned_ctx", + "expected_outcome": "query", + "question": "Break that down by region.", + "follow_up_context": ["Show total watch hours."], + "requires_context": True, + "semantic_model": "sample_data/anime_streaming/semantic_model.yml", + "database": "sample_data/anime_streaming/anime_streaming.sqlite", + }, + # Names a governed metric -> counted as checked, not ungoverned. + { + "id": "governed_1", + "expected_outcome": "query", + "question": "What are the watch hours by playback region?", + "semantic_model": "sample_data/anime_streaming/semantic_model.yml", + "database": "sample_data/anime_streaming/anime_streaming.sqlite", + }, + # Probes are excluded from the coverage count. + { + "id": "probe_1", + "expected_outcome": "policy_rejection", + "question": "Drop the table", + }, + ] + coverage = evaluate_sql._governance_coverage( + cases, + "sample_data/anime_streaming/semantic_model.yml", + "sample_data/anime_streaming/anime_streaming.sqlite", + ) + self.assertEqual(coverage["cases_checked"], 2) + self.assertEqual(coverage["ungoverned_count"], 1) + self.assertEqual(coverage["ungoverned_cases"], ["ungoverned_ctx"]) + # The intersection that matters: needs context AND invisible to governance. + self.assertEqual( + coverage["ungoverned_requiring_context"], ["ungoverned_ctx"] + ) + self.assertIn("grain", coverage["note"]) + + def test_governance_coverage_is_resilient_to_bad_paths(self): + """A coverage probe must never fail the run.""" + coverage = evaluate_sql._governance_coverage( + [ + { + "id": "bad", + "expected_outcome": "query", + "question": "anything", + "semantic_model": "/nonexistent/model.yml", + "database": "/nonexistent/db.sqlite", + } + ], + None, + None, + ) + self.assertEqual(coverage["cases_checked"], 0) + self.assertEqual(coverage["ungoverned_cases"], []) + def test_probes_run_through_the_real_policy_engine_with_case_policy(self): with tempfile.TemporaryDirectory() as directory: root = Path(directory) @@ -263,6 +498,212 @@ def test_semantic_equivalence_is_tolerant_to_column_order(self): ) self.assertEqual(report["metrics"]["semantic_correctness_rate"], 1.0) + # ------------------------------------------------- projection tolerance (D-1) + # + # The evaluator compares the columns the two projections share by name, so an + # answer that is correct but projects a different set of columns is no longer + # scored as semantically wrong. The difference is reported instead, as + # extra_columns / missing_columns and as projection_difference_rate. + # + # Regression origin: 8 of the 40 cases in the Step 0 baseline asked only for + # anime titles while the gold reference SQL additionally projected + # studio_name and studio_tier, so a correct answer was scored wrong and the + # reported accuracy understated the model by ~19 percentage points. + + def test_narrower_projection_than_reference_is_not_a_semantic_error(self): + """Real shape: question asks for titles, reference SQL also projects the + filter columns. The answer is correct and must not be scored wrong.""" + + class NarrowProjectionService: + def ask(self, question, options): + return { + "status": "success", + "rows": [["Neon Genesis"], ["Akira"]], + "columns": ["title"], + "sql": "SELECT title FROM anime WHERE tier = 'Major'", + "model_provider": "fake", + "model": "fake-model", + } + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "anime.sqlite" + with sqlite3.connect(database) as connection: + connection.execute( + "CREATE TABLE anime (title TEXT, studio_name TEXT, tier TEXT)" + ) + connection.execute( + "INSERT INTO anime VALUES " + "('Neon Genesis', 'Gainax', 'Major'), ('Akira', 'TMS', 'Major')" + ) + case = self._query_case(str(database)) + case["expected_sql"] = ( + "SELECT title, studio_name, tier FROM anime WHERE tier = 'Major'" + ) + report = self._run( + NarrowProjectionService(), [case], database, root / "assets" + ) + metrics = report["metrics"] + self.assertEqual(metrics["semantic_correctness_rate"], 1.0) + self.assertEqual(metrics["projection_difference_cases"], 1) + self.assertEqual(metrics["projection_difference_rate"], 1.0) + result = report["results"][0] + self.assertEqual(result["extra_columns"], []) + self.assertEqual(result["missing_columns"], ["studio_name", "tier"]) + self.assertTrue(result["columns_compared"]) + + def test_wider_projection_than_reference_is_not_a_semantic_error(self): + """Real shape (tier-3 reg_3): the reference projects one aggregate, the + answer projects that aggregate plus extra descriptive aggregates.""" + + class WideProjectionService: + def ask(self, question, options): + return { + "status": "success", + # n matches the oracle; the other two columns are extra. + "rows": [[2, 2374, 165041.07]], + "columns": ["n", "quantity", "amount"], + "sql": "SELECT COUNT(*), SUM(q), SUM(a) FROM items", + "model_provider": "fake", + "model": "fake-model", + } + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "items.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE items (q INTEGER, a REAL)") + connection.execute("INSERT INTO items VALUES (1, 10.0), (2, 20.0)") + case = self._query_case(str(database)) + case["expected_sql"] = "SELECT COUNT(*) AS n FROM items" + report = self._run( + WideProjectionService(), [case], database, root / "assets" + ) + metrics = report["metrics"] + self.assertEqual(metrics["semantic_correctness_rate"], 1.0) + result = report["results"][0] + self.assertEqual(result["extra_columns"], ["quantity", "amount"]) + self.assertEqual(result["missing_columns"], []) + + def test_projection_with_no_shared_column_name_is_a_wrong_answer(self): + """Different width AND no shared column name = a different query shape. + + Projection tolerance must not excuse an answer that describes a + different quantity. + """ + + class DifferentShapeService: + def ask(self, question, options): + return { + "status": "success", + "rows": [[1]], + "columns": ["left_id"], + "sql": "SELECT left_id FROM pairs", + "model_provider": "fake", + "model": "fake-model", + } + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "pairs.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE pairs (a INTEGER, b INTEGER)") + connection.execute("INSERT INTO pairs VALUES (1, 2)") + case = self._query_case(str(database)) + case["expected_sql"] = "SELECT a, b FROM pairs" + report = self._run( + DifferentShapeService(), [case], database, root / "assets" + ) + self.assertEqual(report["metrics"]["semantic_correctness_rate"], 0.0) + + def test_renamed_column_of_equal_width_keeps_positional_fallback(self): + """Equal width, disjoint names: a renamed column with the same values is + not newly penalised. This preserves the pre-fix behaviour.""" + + class RenamedColumnService: + def ask(self, question, options): + return { + "status": "success", + "rows": [[1]], + "columns": ["action_anime_count"], + "sql": "SELECT COUNT(*) FROM pairs", + "model_provider": "fake", + "model": "fake-model", + } + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "pairs.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE pairs (a INTEGER, b INTEGER)") + connection.execute("INSERT INTO pairs VALUES (1, 2)") + case = self._query_case(str(database)) + case["expected_sql"] = "SELECT COUNT(*) AS anime_count FROM pairs" + report = self._run( + RenamedColumnService(), [case], database, root / "assets" + ) + self.assertEqual(report["metrics"]["semantic_correctness_rate"], 1.0) + + def test_wrong_values_under_a_shared_projection_are_still_a_semantic_error(self): + """Shared column names must not rescue values that genuinely differ.""" + + class WrongValueService: + def ask(self, question, options): + return { + "status": "success", + "rows": [["Card", 120]], # expected: a single overall count + "columns": ["payment_method", "n"], + "sql": "SELECT payment_method, COUNT(*) FROM orders GROUP BY 1", + "model_provider": "fake", + "model": "fake-model", + } + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "orders.sqlite" + with sqlite3.connect(database) as connection: + connection.execute( + "CREATE TABLE orders (payment_method TEXT, order_id INTEGER)" + ) + connection.execute( + "INSERT INTO orders VALUES ('Card', 1), ('Cash', 2)" + ) + case = self._query_case(str(database)) + case["expected_sql"] = ( + "SELECT COUNT(DISTINCT payment_method) AS n FROM orders" + ) + report = self._run( + WrongValueService(), [case], database, root / "assets" + ) + # Shared column "n": 120 != 2, so this stays a genuine mismatch. + self.assertEqual(report["metrics"]["semantic_correctness_rate"], 0.0) + + def test_projection_difference_rate_is_none_without_column_names(self): + """Positional fallback (no column names) cannot report a projection diff.""" + + class NoColumnsService: + def ask(self, question, options): + return { + "status": "success", + "rows": [[1]], + "sql": "SELECT 1", + "model_provider": "fake", + "model": "fake-model", + } + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "numbers.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE numbers (value INTEGER)") + connection.execute("INSERT INTO numbers VALUES (1)") + case = self._query_case(str(database)) + report = self._run(NoColumnsService(), [case], database, root / "assets") + metrics = report["metrics"] + self.assertEqual(metrics["semantic_correctness_rate"], 1.0) + self.assertIsNone(metrics["projection_difference_rate"]) + self.assertFalse(report["results"][0]["columns_compared"]) + def test_followup_warmup_turns_are_excluded_from_latency(self): with tempfile.TemporaryDirectory() as directory: root = Path(directory) @@ -416,3 +857,111 @@ def test_oracle_latency_is_recorded_separately(self): if __name__ == "__main__": unittest.main() + + +class RecordingModelFactoryTest(unittest.TestCase): + """The evaluator's provider wrapper must match the provider contract. + + Regression origin: ``_recording_model_factory`` wrapped the real adapter with a + ``generate_with_messages`` that still had the pre-feat-012 two-argument + signature. Once the model budget started passing ``timeout`` through, every + evaluated case failed at ``gen_sql`` with an unexpected keyword argument — and + the existing tests did not notice, because they inject an ``AgentService`` + double and never build this wrapper against a real adapter. + + A run that fails in 60 ms with zero measured tokens is the signature of this + class of bug, so the test asserts the wrapper actually completes a call. + """ + + def test_the_recording_wrapper_accepts_the_provider_contract(self): + import inspect + + contract = inspect.signature(BaseModelProvider.generate_with_messages) + self.assertIn("timeout", contract.parameters) + + class Adapter(BaseModelProvider): + provider = "test" + model = "test-1" + + def generate_with_messages( + self, messages, json_mode=False, timeout=None + ) -> str: + self.last_usage = None + return '{"sql": "SELECT 1", "explanation": "e", "tables_used": []}' + + usage_log: list[dict] = [] + wrapped = evaluate_sql.wrap_provider_for_usage(Adapter(), usage_log) + + # The deadline is ambient; the wrapper must forward it rather than choke. + from queryforge.core.observability import model_deadline + + with model_deadline(30.0): + raw = wrapped.generate_with_messages( + [{"role": "user", "content": "hi"}], timeout=30.0 + ) + self.assertIn("SELECT 1", raw) + + def test_the_wrapper_is_used_by_a_real_agent_service(self): + """Guard the path the doubles never touch: a real service, real workflow.""" + import sqlite3 + import tempfile + from pathlib import Path + + from queryforge.application import AgentService, AgentOptions + from queryforge.core.config import Config + + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + database = root / "w.sqlite" + with sqlite3.connect(database) as connection: + connection.execute("CREATE TABLE items (name TEXT)") + connection.execute("INSERT INTO items VALUES ('a')") + config = Config( + llm_provider="openai", + llm_api_key=None, + llm_model="offline", + llm_base_url=None, + database_path=str(database), + history_db_path=str(root / "history.db"), + orchestration_state_root=str(root / "runs"), + ) + + import json as _json + + class Adapter(BaseModelProvider): + provider = "test" + model = "test-1" + + def generate_with_messages( + self, messages, json_mode=False, timeout=None + ) -> str: + self.last_usage = None + prompt = _json.dumps(messages, ensure_ascii=False) + if "Select local QueryForge skills" in prompt: + return _json.dumps({"skills": [], "reason": "none"}) + if "Evaluate whether the SQL and result" in prompt: + return _json.dumps( + {"success": True, "strategy": "SUCCESS", "reason": "ok"} + ) + return _json.dumps( + {"sql": "SELECT 1", "explanation": "e", "tables_used": []} + ) + + usage_log: list[dict] = [] + + def factory(_config): + return evaluate_sql.wrap_provider_for_usage(Adapter(), usage_log) + + service = AgentService( + config_loader=lambda **_: config, llm_factory=factory + ) + output = service.ask( + "How many items are there?", + AgentOptions( + database=str(database), + skills=None, + run_id="wrapper_smoke", + orchestration_state_root=str(root / "runs"), + ), + ) + self.assertEqual(output.get("status"), "success") diff --git a/tests/test_knowledge_csv_trust.py b/tests/test_knowledge_csv_trust.py new file mode 100644 index 0000000..c67560b --- /dev/null +++ b/tests/test_knowledge_csv_trust.py @@ -0,0 +1,106 @@ +"""E-05: the CSV knowledge path must not mint trusted material unconditionally. + +``KnowledgeBaseBuilder._csv_documents`` stamped every row ``human_reviewed`` with +``review_status: reviewed`` regardless of whether anyone had reviewed it. Any CSV +with ``question``/``sql`` columns therefore entered the retrieval index as trusted +few-shot material, which inverts the three-tier verification model in +``domain/knowledge/governance.py`` where only an explicit human review counts as +trusted. ``SQLHistoryStore.import_success_stories`` already gated trust on a named +reviewer; this path did not. + +These tests exercise the real document builder, not a double. +""" + +from __future__ import annotations + +import tempfile +import unittest +from pathlib import Path + +from queryforge.infrastructure.storage.knowledge_base import KnowledgeBaseBuilder + + +def _write_csv(path: Path, header: str, rows: list[str]) -> Path: + path.write_text(header + "\n" + "\n".join(rows) + "\n", encoding="utf-8") + return path + + +class CsvKnowledgeTrustTest(unittest.TestCase): + def _documents(self, header: str, rows: list[str]) -> list: + with tempfile.TemporaryDirectory() as directory: + path = _write_csv(Path(directory) / "source.csv", header, rows) + return KnowledgeBaseBuilder._csv_documents(path) + + def test_rows_without_a_reviewer_are_not_marked_human_reviewed(self): + documents = self._documents( + "question,sql,evidence", + [ + 'How many anime are there?,"SELECT COUNT(*) AS n FROM dim_anime",checked', + 'How many studios?,"SELECT COUNT(*) AS n FROM dim_studio",checked', + ], + ) + self.assertEqual(len(documents), 2) + for document in documents: + metadata = document.metadata + with self.subTest(question=metadata["question"]): + # Execution success is not business correctness. + self.assertEqual( + metadata["verification_level"], "execution_success" + ) + self.assertEqual(metadata["review_status"], "draft") + self.assertIsNone(metadata["reviewer"]) + + def test_a_named_reviewer_earns_human_reviewed(self): + documents = self._documents( + "question,sql,reviewer", + [ + 'How many anime are there?,"SELECT COUNT(*) AS n FROM dim_anime",data-platform', + ], + ) + self.assertEqual(len(documents), 1) + metadata = documents[0].metadata + self.assertEqual(metadata["verification_level"], "human_reviewed") + self.assertEqual(metadata["review_status"], "reviewed") + self.assertEqual(metadata["reviewer"], "data-platform") + self.assertEqual(metadata["owner"], "data-platform") + + def test_reviewed_by_is_accepted_as_an_alias(self): + documents = self._documents( + "question,sql,reviewed_by", + ['How many genres?,"SELECT COUNT(*) AS n FROM dim_genre",analytics-lead'], + ) + metadata = documents[0].metadata + self.assertEqual(metadata["verification_level"], "human_reviewed") + self.assertEqual(metadata["reviewer"], "analytics-lead") + + def test_a_blank_reviewer_does_not_grant_trust(self): + documents = self._documents( + "question,sql,reviewer,owner", + ['How many users?,"SELECT COUNT(*) AS n FROM dim_user",,business'], + ) + metadata = documents[0].metadata + self.assertEqual(metadata["verification_level"], "execution_success") + self.assertEqual(metadata["review_status"], "draft") + # Falls back to the declared owner rather than losing it. + self.assertEqual(metadata["owner"], "business") + + def test_checked_in_sample_is_untrusted_until_a_reviewer_is_added(self): + """The shipped sample CSV has no reviewer column, so it must not be trusted.""" + sample = ( + Path(__file__).resolve().parents[1] + / "sample_data" + / "anime_streaming" + / "success_story.csv" + ) + self.assertTrue(sample.is_file(), "sample CSV is part of the repo") + documents = KnowledgeBaseBuilder._csv_documents(sample) + self.assertTrue(documents) + for document in documents: + self.assertEqual( + document.metadata["verification_level"], "execution_success" + ) + self.assertEqual(document.metadata["review_status"], "draft") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_model_budget.py b/tests/test_model_budget.py new file mode 100644 index 0000000..8e9ca69 --- /dev/null +++ b/tests/test_model_budget.py @@ -0,0 +1,233 @@ +"""feat-012: every model call is charged to the run's shared budget. + +Before this, the conversational path constructed no model budget at all. The tool +loop and the planner had one, but ``/ask`` did not, so a run's model spend and +wall-clock time were unbounded and ``BudgetLimits.model_deadline_ms`` had no +consumer anywhere in the codebase. + +These tests drive the real ``WorkflowRunner`` with a counting model provider, so +they exercise the actual decoration order (budget outside the observed provider) +and the real reservation lifecycle. +""" + +from __future__ import annotations + +import json +import sqlite3 +import tempfile +import unittest +from pathlib import Path + +from queryforge.core.config import Config +from queryforge.core.schemas.models import SqlTask +from queryforge.infrastructure.models.base import BaseModelProvider +from queryforge.orchestration.tools.budget import BudgetLimits, BudgetManager +from queryforge.orchestration.tools.specs import ToolBudgetError +from queryforge.workflow.budgeted_model import BudgetedModelProvider +from queryforge.workflow.workflow import WorkflowError +from queryforge.workflow.workflow_runner import WorkflowRunner + + +class CountingModel(BaseModelProvider): + """A real provider adapter that counts calls and records the deadline it saw. + + Subclasses ``BaseModelProvider`` (not a bare duck type) so it inherits the + ``generate_json`` / ``generate_text`` envelope the nodes actually call; a + duck-typed double that lacks them produces a confusing ``NoneType is not + callable`` deep in the provider chain. + """ + + provider = "counting" + model = "counting-1" + + def __init__(self) -> None: + self.calls = 0 + self.timeouts_seen: list[float | None] = [] + self.last_timeout: float | None = None + self.last_usage = None + + def generate_json(self, prompt: str, timeout: float | None = None): + # Overridden so the recorded deadline reflects what the run actually + # passed, rather than being reconstructed inside the base class. + self.timeouts_seen.append(timeout) + self.last_timeout = timeout + return super().generate_json(prompt, timeout=timeout) + + def generate_with_messages(self, messages, json_mode=False, timeout=None): + if not self.timeouts_seen: + # Direct generate_with_messages calls still need recording. + self.timeouts_seen.append(timeout) + self.calls += 1 + self.last_timeout = timeout + prompt = json.dumps(messages, ensure_ascii=False) + if "Select local QueryForge skills" in prompt: + payload = {"skills": [], "reason": "test"} + elif "Evaluate whether the SQL and result" in prompt: + payload = {"success": True, "strategy": "SUCCESS", "reason": "ok"} + else: + payload = {"sql": "SELECT 1 AS n", "explanation": "e", "tables_used": []} + + class _Usage: + prompt_tokens = 100 + completion_tokens = 20 + total_tokens = 120 + estimated = False + + self.last_usage = _Usage() + return json.dumps(payload) + + +class BudgetedModelProviderTest(unittest.TestCase): + def test_a_call_is_reserved_before_it_is_sent_and_settled_with_real_tokens(self): + budget = BudgetManager(limits={}) + inner = CountingModel() + provider = BudgetedModelProvider(inner, budget) + + provider.generate_with_messages([{"role": "user", "content": "hi"}]) + + self.assertEqual(inner.calls, 1) + snapshot = budget.snapshot() + self.assertEqual(snapshot["reservations"], 1) + # Settled to the provider's reported total, not the reservation. + self.assertEqual(snapshot["usage"]["max_estimated_tokens"], 120) + + def test_an_unknown_cost_is_not_recorded_as_a_free_call(self): + """A provider that reports nothing keeps its reservation charged.""" + + class SilentModel(CountingModel): + def generate_with_messages(self, messages, json_mode=False, timeout=None): + self.calls += 1 + self.last_usage = None + return json.dumps( + {"sql": "SELECT 1", "explanation": "e", "tables_used": []} + ) + + budget = BudgetManager(limits={}) + provider = BudgetedModelProvider(SilentModel(), budget) + provider.generate_with_messages([{"role": "user", "content": "hi"}]) + + usage = budget.snapshot()["usage"]["max_estimated_tokens"] + self.assertGreater(usage, 0, "an unreported cost must still be charged") + + def test_the_remaining_deadline_reaches_the_adapter(self): + """model_deadline_ms had no consumer before this feature.""" + inner = CountingModel() + inner.timeouts_seen = [] + budget = BudgetManager(limits={"model_deadline_ms": 30_000}) + provider = BudgetedModelProvider(inner, budget) + provider.generate_with_messages([{"role": "user", "content": "hi"}]) + + self.assertEqual(len(inner.timeouts_seen), 1) + timeout = inner.timeouts_seen[0] + self.assertIsNotNone(timeout, "the adapter must receive a deadline") + self.assertGreater(timeout, 0) + self.assertLessEqual(timeout, 30) + + def test_an_exhausted_call_budget_refuses_before_sending(self): + budget = BudgetManager(limits={"max_tool_calls": 1}) + inner = CountingModel() + refusals: dict = {} + provider = BudgetedModelProvider(inner, budget, refusal_sink=refusals) + + provider.generate_with_messages([{"role": "user", "content": "one"}]) + with self.assertRaises(ToolBudgetError): + provider.generate_with_messages([{"role": "user", "content": "two"}]) + + # The refusal happened *before* the second request was sent. + self.assertEqual(inner.calls, 1) + self.assertEqual(refusals["budget_refusal"]["limit"], "max_tool_calls") + self.assertIn("max_tool_calls", refusals["budget_refusal"]["reason"]) + + def test_an_expired_deadline_refuses_before_sending(self): + budget = BudgetManager(limits={"model_deadline_ms": 0}) + inner = CountingModel() + refusals: dict = {} + provider = BudgetedModelProvider(inner, budget, refusal_sink=refusals) + + with self.assertRaises(ToolBudgetError): + provider.generate_with_messages([{"role": "user", "content": "hi"}]) + self.assertEqual(inner.calls, 0, "an impossible call must not be sent") + self.assertEqual(refusals["budget_refusal"]["limit"], "model_deadline_ms") + + +class WorkflowBudgetTest(unittest.TestCase): + """The budget has to stop a real run, not just a synthetic provider call.""" + + def setUp(self) -> None: + self._directory = tempfile.TemporaryDirectory() + self.root = Path(self._directory.name) + self.database = self.root / "budget.sqlite" + with sqlite3.connect(self.database) as connection: + connection.execute("CREATE TABLE items (name TEXT)") + connection.execute("INSERT INTO items VALUES ('a')") + self.model = CountingModel() + + def tearDown(self) -> None: + self._directory.cleanup() + + def _run(self, **runner_kwargs): + runner = WorkflowRunner( + self._config(), + llm_factory=lambda _: self.model, + selected_skills=[], + **runner_kwargs, + ) + return runner.run( + SqlTask(question="List item names", database_path=str(self.database)) + ) + + def _config(self): + """Minimal explicit config, as tests/test_retry_workflow.py does. + + Loading the ambient config would point the semantic model at the sample + database and fail schema validation against this fixture database. + """ + return Config( + llm_provider="openai", + llm_api_key=None, + llm_model="offline", + llm_base_url=None, + database_path=str(self.database), + history_db_path=str(self.root / "history.db"), + orchestration_state_root=str(self.root / "runs"), + ) + + def test_a_small_call_budget_stops_the_run_and_records_why(self): + """A deliberately tiny budget must end the run mid-chain, with the reason. + + The workflow makes more than one model call (skill selection, then SQL + generation), so a single-call allowance is exhausted partway through. + """ + budget = BudgetManager(limits={"max_tool_calls": 1}) + with self.assertRaises(WorkflowError): + self._run(model_budget_manager=budget) + + snapshot = budget.snapshot() + self.assertGreaterEqual(snapshot["reservations"], 1) + self.assertIn("max_tool_calls", snapshot["exhausted"]) + self.assertLess( + self.model.calls, + 3, + "the refusal must stop further model calls, not merely report them", + ) + + def test_an_unbounded_run_still_completes(self): + """The default must not change existing behaviour.""" + output = self._run() + self.assertEqual(output["status"], "success") + self.assertGreaterEqual(self.model.calls, 1) + + def test_the_run_exposes_its_model_budget(self): + """A caller can see what a run spent, which run_context alone did not give.""" + budget = BudgetManager(limits={"model_deadline_ms": 60_000}) + output = self._run(model_budget_manager=budget) + self.assertEqual(output["status"], "success") + # Deadline reached the adapter on the real run, not only in a unit test. + self.assertTrue( + any(t is not None for t in self.model.timeouts_seen), + "the adapter must receive a deadline on a real run", + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_observability.py b/tests/test_observability.py index 717f467..7a08a6b 100644 --- a/tests/test_observability.py +++ b/tests/test_observability.py @@ -236,3 +236,47 @@ def test_ending_one_span_twice_records_it_once(self): if __name__ == "__main__": unittest.main() + + +class PerNodeLatencyBreakdownTest(unittest.TestCase): + """``by_node`` attributes model cost to the node that issued it. + + A run's wall clock is dominated by sequential model calls, so the run-level + total cannot say which step to optimise. Each model span is opened inside a node + span carrying the node name, which is enough to attribute the cost without new + instrumentation. + """ + + def test_model_cost_is_grouped_by_node(self): + from queryforge.core.observability import SpanRecorder + + from queryforge.core.observability import node_logging_context + + recorder = SpanRecorder("by_node_probe") + # Production opens model spans inside ``node_logging_context`` (see + # ``workflow._execute_observed_node``), which is what attributes them. + for node in ("gen_sql", "gen_sql", "reflect"): + with node_logging_context(node): + span = recorder.begin(f"{node}.model", "model") + recorder.end(span, status="success") + summary = recorder.latency_summary(end_to_end_ms=1000.0) + by_node = summary["by_node"] + self.assertIn("gen_sql", by_node) + self.assertIn("reflect", by_node) + self.assertEqual(by_node["gen_sql"]["model_calls"], 2) + self.assertEqual(by_node["reflect"]["model_calls"], 1) + # Ordered by model time, descending, so the biggest cost is first. + self.assertEqual(list(by_node), ["gen_sql", "reflect"]) + self.assertIn("prompt_tokens", by_node["gen_sql"]) + + def test_non_model_spans_are_excluded_from_the_node_breakdown(self): + from queryforge.core.observability import SpanRecorder + + from queryforge.core.observability import node_logging_context + + recorder = SpanRecorder("by_node_probe2") + with node_logging_context("execute_sql"): + span = recorder.begin("sql.execute", "sql") + recorder.end(span, status="success") + summary = recorder.latency_summary(end_to_end_ms=1.0) + self.assertEqual(summary["by_node"], {}) diff --git a/tests/test_retry_workflow.py b/tests/test_retry_workflow.py index 8b792c5..83c52da 100644 --- a/tests/test_retry_workflow.py +++ b/tests/test_retry_workflow.py @@ -168,17 +168,42 @@ def test_retry_limit_preserves_all_failed_attempts(self) -> None: SQLHistoryStore(self.config.history_db_path).list_entries(), [] ) - def test_need_user_review_stops_with_clear_reason(self) -> None: + def test_need_user_review_returns_a_clarification_instead_of_discarding_the_run( + self, + ) -> None: + """A NEED_USER_REVIEW verdict is a question, not a crash. + + This test previously asserted ``assertRaisesRegex(WorkflowError, "Human + review required")``. That behaviour was the defect (D-2): the model had + correctly identified an ambiguity, and the workflow answered by raising, + which produced **zero payload** and recorded the run as failed. The + verdict is now surfaced as a structured ``needs_clarification`` result, + which is the vocabulary the REST mapping, the event protocol, the gateway + wording and the evaluator already understand. + """ llm = RetryLLM("review") runner = WorkflowRunner( self.config, llm_factory=lambda _: llm, selected_skills=[], ) - with self.assertRaisesRegex(WorkflowError, "Human review required"): - runner.run( - SqlTask(question="List item names", database_path=str(self.database)) - ) + output = runner.run( + SqlTask(question="List item names", database_path=str(self.database)) + ) + self.assertEqual(output["status"], "needs_clarification") + self.assertEqual( + output["reason"], "The requested business meaning is ambiguous." + ) + self.assertEqual(output["strategy"], "NEED_USER_REVIEW") + self.assertEqual( + output["unresolved_questions"], + ["The requested business meaning is ambiguous."], + ) + # The run asked a question about meaning; the query itself still ran, so + # its SQL and rows must survive rather than being thrown away. + self.assertTrue(output["sql"]) + self.assertIn("row_count", output) + self.assertEqual(output["execution_errors"], []) if __name__ == "__main__": diff --git a/tests/test_run_outcomes.py b/tests/test_run_outcomes.py new file mode 100644 index 0000000..0f3da61 --- /dev/null +++ b/tests/test_run_outcomes.py @@ -0,0 +1,113 @@ +"""The terminal-outcome vocabulary and derivation (feat-008). + +Before ``queryforge.core.outcomes`` existed, the terminal status of a run was +derived in at least two places with three overlapping vocabularies. That is how +defect E-02 happened: the QA gate recorded a block, ``OrchestratorAgent`` then +re-derived the status from the workflow's earlier result dict and overwrote it to +``completed``, so ``state.json`` and the caller disagreed and both were locally +consistent. + +These tests pin the single derivation, the normalisation of every historical +spelling, and the rule that a recorded block outranks a stale result status. +""" + +from __future__ import annotations + +import unittest + +from queryforge.core.outcomes import ( + EVENT_OUTCOMES, + NON_REPLACEABLE, + TERMINAL_OUTCOMES, + derive_outcome, + is_terminal, + normalize_outcome, + to_event_outcome, +) + + +class OutcomeVocabularyTest(unittest.TestCase): + def test_every_canonical_outcome_maps_to_an_event_outcome(self): + for outcome in TERMINAL_OUTCOMES: + with self.subTest(outcome=outcome): + self.assertIn(outcome, EVENT_OUTCOMES) + self.assertIn(to_event_outcome(outcome), { + "success", "partial", "blocked", "failed", "cancelled", + }) + + def test_a_clarification_is_reported_as_blocked_to_a_streaming_client(self): + """The transport protocol distinguishes fewer outcomes than the engine. + + A clarification is not a success and not a failure; a streaming client sees + a blocked run carrying the reason. + """ + self.assertEqual(to_event_outcome("needs_clarification"), "blocked") + + def test_historical_spellings_normalise_onto_the_canonical_set(self): + expected = { + "success": "succeeded", + "completed": "succeeded", + "planned": "succeeded", + "ok": "succeeded", + "degraded": "partial", + "error": "failed", + "canceled": "cancelled", + "cancelled": "cancelled", + "needs_clarification": "needs_clarification", + "blocked": "blocked", + # Case and whitespace must not matter. + " COMPLETED ": "succeeded", + } + for raw, want in expected.items(): + with self.subTest(raw=raw): + self.assertEqual(normalize_outcome(raw), want) + + def test_non_terminal_and_unknown_values_are_not_terminal(self): + for raw in ("running", "created", "routing", "", None, " "): + with self.subTest(raw=raw): + self.assertIsNone(normalize_outcome(raw)) + self.assertFalse(is_terminal(raw)) + + def test_a_cancellation_is_non_replaceable(self): + """First-writer-wins covers every terminal outcome, not just some.""" + for outcome in TERMINAL_OUTCOMES: + self.assertIn(outcome, NON_REPLACEABLE) + + +class OutcomeDerivationTest(unittest.TestCase): + def test_a_recorded_block_outranks_a_stale_result_status(self): + """The E-02 rule: a gate that recorded why it blocked is not overwritten. + + The workflow's result dict is captured before the completion hook runs, so + it can still say "success" while a gate has already blocked the run. + """ + outcome = derive_outcome( + result_status="success", + blocked_reason="qa_report found severe data quality issue", + ) + self.assertEqual(outcome, "blocked") + + def test_cancellation_outranks_everything(self): + outcome = derive_outcome( + result_status="success", + blocked_reason="some gate", + cancelled=True, + ) + self.assertEqual(outcome, "cancelled") + + def test_result_status_is_normalised_not_passed_through(self): + self.assertEqual(derive_outcome(result_status="planned"), "succeeded") + self.assertEqual(derive_outcome(result_status="degraded"), "partial") + self.assertEqual( + derive_outcome(result_status="needs_clarification"), "needs_clarification" + ) + + def test_absent_or_unknown_status_defaults_to_succeeded(self): + """Explicit and testable, rather than an implicit ``or "success"``.""" + self.assertEqual(derive_outcome(), "succeeded") + self.assertEqual(derive_outcome(result_status=None), "succeeded") + self.assertEqual(derive_outcome(result_status="something-new"), "succeeded") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_semantic_sql_validator.py b/tests/test_semantic_sql_validator.py index b09b4e6..45c764a 100644 --- a/tests/test_semantic_sql_validator.py +++ b/tests/test_semantic_sql_validator.py @@ -975,3 +975,63 @@ def noop(state: Context) -> NodeResult: if __name__ == "__main__": unittest.main() + + +class DefaultFilterWithoutColumnTest(FixtureMixin, unittest.TestCase): + """E-06: a filter that parses but names no column must not crash ``validate``. + + ``_default_filter_expectations`` returns ``(None, None)`` for such a filter — + for example ``EXISTS(SELECT 1 FROM fact_sales.is_valid)`` — while its type + annotation claimed ``list``. The caller trusted the annotation and passed the + ``None`` to ``set(columns)``, raising ``TypeError`` out of ``validate()``. + That escapes the node's try block, so the semantic gate crashed outright + instead of returning a verdict, taking the run down with it. + """ + + def _validator_for(self, default_filters: list[str]): + import copy + + import yaml + + model = copy.deepcopy(SEMANTIC_MODEL) + model["metrics"][0]["default_filters"] = default_filters + path = self.root / "default_filter_model.yml" + path.write_text(yaml.safe_dump(model, sort_keys=False), encoding="utf-8") + semantic = SemanticModelLoader.load_and_validate( + path, self.schemas, "What are net sales?" + ) + context = Context( + task=SqlTask(question="What are net sales?", database_path=str(self.database)), + semantic_model=semantic, + metric_matches=SemanticModelLoader.match_metrics( + semantic.model, "What are net sales?" + ), + ) + return SemanticSQLValidator.for_context(context) + + def test_a_filter_naming_no_column_returns_a_verdict(self): + validator = self._validator_for( + ["EXISTS(SELECT 1 FROM fact_sales.is_valid)"] + ) + # Must not raise: this is the whole point of the regression. + result = validator.validate( + "SELECT SUM(s.amount) AS net_sales FROM fact_sales s" + ) + self.assertEqual(result.status, "violation") + self.assertIn("default_filter", result.rule_names) + self.assertIn("references no column", result.error_message()) + + # Note: the ``expected is None`` branch (a filter that does not even parse) is + # not reachable from a loaded semantic model — ``models.py`` rejects a filter + # without a qualified ``table.column`` reference at load time — so there is no + # test for it here. The reachable-``None`` case is the one above: a filter that + # parses and references a qualified column inside a subquery, yet contributes + # no column that the validator can look for in the SQL under test. + + def test_a_well_formed_filter_still_validates_normally(self): + """The guard must not disturb the ordinary path.""" + validator = self._validator_for(["fact_sales.is_valid = 1"]) + result = validator.validate( + "SELECT SUM(s.amount) AS net_sales FROM fact_sales s WHERE s.is_valid = 1" + ) + self.assertEqual(result.status, "passed") diff --git a/tests/test_workspace_root.py b/tests/test_workspace_root.py new file mode 100644 index 0000000..00319e2 --- /dev/null +++ b/tests/test_workspace_root.py @@ -0,0 +1,145 @@ +"""Workspace-root resolution (E-22). + +``PROJECT_ROOT`` used to be ``Path(__file__).resolve().parents[2]`` in six separate +modules. In a source checkout that is the repository root; after +``pip install queryforge`` it is ``site-packages``, so ``.queryforge/history.db``, +the trace directory, the chart output directory and the LanceDB path all moved into +the installed package — not the caller's data, and frequently not writable. + +These tests pin the *resolution order* rather than the checkout case, because the +checkout case is what every other test in this suite already exercises implicitly. +""" + +import os +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +from queryforge.core import paths +from queryforge.core.paths import ( + WORKSPACE_ROOT_ENV, + resolve_path, + workspace_root, +) + +PROJECT_ROOT = Path(__file__).resolve().parents[1] + + +class WorkspaceRootTest(unittest.TestCase): + def setUp(self): + paths.workspace_root.cache_clear() + self.addCleanup(paths.workspace_root.cache_clear) + + def test_checkout_is_found_by_walking_up_from_the_package(self): + """Without an override, the nearest checkout ancestor wins.""" + self.assertEqual(workspace_root(), PROJECT_ROOT) + self.assertTrue((workspace_root() / "pyproject.toml").is_file()) + + def test_override_wins_and_supports_relative_and_tilde_forms(self): + with tempfile.TemporaryDirectory() as d, patch.dict( + os.environ, {WORKSPACE_ROOT_ENV: d} + ): + paths.workspace_root.cache_clear() + self.assertEqual(workspace_root(), Path(d).resolve()) + + paths.workspace_root.cache_clear() + with patch.dict(os.environ, {WORKSPACE_ROOT_ENV: "."}): + self.assertEqual(workspace_root(), Path.cwd().resolve()) + + def test_installed_package_does_not_resolve_state_into_site_packages(self): + """The regression: no override + no checkout ancestor -> the cwd, not the package. + + ``site-packages`` is simulated by a package directory with no checkout + marker anywhere above it, which is exactly what makes ``parents[2]`` wrong + there. + """ + with tempfile.TemporaryDirectory() as package_parent, tempfile.TemporaryDirectory() as cwd: + installed = Path(package_parent) / "site-packages" / "queryforge" / "core" + installed.mkdir(parents=True) + with patch.dict(os.environ, {}, clear=False): + os.environ.pop(WORKSPACE_ROOT_ENV, None) + with patch.object(Path, "cwd", staticmethod(lambda: Path(cwd))): + resolved = paths._resolve_root(installed) + self.assertEqual(resolved, Path(cwd).resolve()) + # And the checkout case still wins over the cwd. + self.assertEqual( + paths._resolve_root(PROJECT_ROOT / "queryforge" / "core"), + PROJECT_ROOT, + ) + + def test_a_wrong_root_is_not_selected_for_a_non_checkout_install(self): + """Guard the marker list: a bare directory must not be mistaken for a checkout.""" + with tempfile.TemporaryDirectory() as package_parent: + installed = Path(package_parent) / "queryforge" / "core" + installed.mkdir(parents=True) + self.assertFalse(paths._is_checkout(installed)) + self.assertFalse(paths._is_checkout(Path(package_parent))) + self.assertTrue(paths._is_checkout(PROJECT_ROOT)) + + +class ResolvePathTest(unittest.TestCase): + def test_absolute_paths_pass_through_untouched(self): + absolute = Path(tempfile.gettempdir()) / "queryforge-absolute.db" + self.assertEqual(resolve_path(absolute), absolute) + + def test_relative_paths_resolve_against_the_workspace(self): + with tempfile.TemporaryDirectory() as d, patch.dict( + os.environ, {WORKSPACE_ROOT_ENV: d} + ): + paths.workspace_root.cache_clear() + self.addCleanup(paths.workspace_root.cache_clear) + self.assertEqual( + resolve_path(".queryforge/history.db"), Path(d).resolve() / ".queryforge/history.db" + ) + + +class ModuleRootsAgreeTest(unittest.TestCase): + def test_every_module_root_is_the_same_object_value(self): + """Six modules derived this independently; they must not drift again.""" + from queryforge.core.config import PROJECT_ROOT as config_root + from queryforge.core.observability import PROJECT_ROOT as observability_root + from queryforge.domain.domains import PROJECT_ROOT as domains_root + from queryforge.domain.semantic.builder import PROJECT_ROOT as builder_root + from queryforge.infrastructure.storage import knowledge_base, sql_history_store + from queryforge.infrastructure.storage import vector_store + from queryforge.workflow.node.visualization_node import ( + PROJECT_ROOT as visualization_root, + ) + + roots = { + "config": config_root, + "observability": observability_root, + "domains": domains_root, + "semantic.builder": builder_root, + "knowledge_base": knowledge_base.PROJECT_ROOT, + "sql_history_store": sql_history_store.PROJECT_ROOT, + "vector_store": vector_store.PROJECT_ROOT, + "visualization_node": visualization_root, + } + distinct = {str(value) for value in roots.values()} + self.assertEqual(distinct, {str(PROJECT_ROOT)}, roots) + + def test_state_defaults_live_under_the_workspace_root(self): + from queryforge.core.observability import DEFAULT_LOG_PATH, DEFAULT_TRACE_DIR + from queryforge.infrastructure.storage.sql_history_store import ( + DEFAULT_HISTORY_DB_PATH, + ) + from queryforge.workflow.node.visualization_node import ( + DEFAULT_CHART_OUTPUT_DIR, + ) + + for label, path in { + "log": DEFAULT_LOG_PATH, + "traces": DEFAULT_TRACE_DIR, + "history": DEFAULT_HISTORY_DB_PATH, + "charts": DEFAULT_CHART_OUTPUT_DIR, + }.items(): + with self.subTest(default=label): + self.assertTrue( + str(path).startswith(str(PROJECT_ROOT)), f"{label} -> {path}" + ) + + +if __name__ == "__main__": # pragma: no cover + unittest.main() diff --git a/web/README.md b/web/README.md index edfe00f..6dcae8f 100644 --- a/web/README.md +++ b/web/README.md @@ -33,7 +33,7 @@ If the API uses another address, copy `.env.example` to `.env.local` and change - **Data Domains** — create, select, and manage isolated business contexts. - **Overview** — active-domain readiness, governed metrics, contract health, and recent activity. -- **Data Sources** — domain-scoped SQLite, CSV, and Parquet onboarding. +- **Data Sources** — domain-scoped CSV and Parquet onboarding; QueryForge builds the governed SQLite database from the uploaded source data. - **Semantic Studio** — a required contract builder for identity, grain, dimensions, measures, metrics, relationships, Join Paths, policy, and quality. - **Ask & Analyze** — natural language to governed SQL with progress, results, diff --git a/web/app/api/studio/upload/route.ts b/web/app/api/studio/upload/route.ts index 973faa7..e4b8584 100644 --- a/web/app/api/studio/upload/route.ts +++ b/web/app/api/studio/upload/route.ts @@ -2,7 +2,15 @@ import { requireStudioUser, studioAuthMode } from "@/app/studio-auth"; import { ensureStudioSchema, getStudioBindings } from "@/db/runtime"; const MAX_FILE_BYTES = 25 * 1024 * 1024; -const ALLOWED_EXTENSIONS = new Set(["sqlite", "db", "csv", "parquet"]); +// Must match PublishService.ALLOWED_EXTENSIONS in +// queryforge/application/publish_service.py. This list previously also accepted +// "sqlite" and "db", which the Python side rejects outright, so uploading a +// database file always ended in pythonPublish.status = "failed" and an HTTP 422: +// a dead end advertised as a supported path. Publishing builds a governed +// database *from* source data, so a prebuilt database is not an accepted input and +// the UI says so up front instead of failing at the end of the wizard. +const ALLOWED_EXTENSIONS = new Set(["csv", "parquet"]); +const REJECTED_DATABASE_EXTENSIONS = new Set(["sqlite", "db", "sqlite3"]); const CSV_HEADER_READ_LIMIT = 256 * 1024; type UploadedSemanticContract = { @@ -284,9 +292,25 @@ export async function POST(request: Request) { } for (const file of files) { - if (!ALLOWED_EXTENSIONS.has(extension(file.name))) { + const fileExtension = extension(file.name); + if (REJECTED_DATABASE_EXTENSIONS.has(fileExtension)) { return Response.json( - { detail: `Unsupported file type: ${file.name}` }, + { + detail: + `${file.name}: database files cannot be published directly. Upload ` + + `the source data (CSV or Parquet) and QueryForge builds the governed ` + + `database from it.`, + }, + { status: 400 }, + ); + } + if (!ALLOWED_EXTENSIONS.has(fileExtension)) { + return Response.json( + { + detail: + `Unsupported file type: ${file.name}. Allowed: ` + + `${[...ALLOWED_EXTENSIONS].sort().join(", ")}.`, + }, { status: 400 }, ); } diff --git a/web/app/page.tsx b/web/app/page.tsx index b4030db..b052ee2 100644 --- a/web/app/page.tsx +++ b/web/app/page.tsx @@ -3955,7 +3955,7 @@ function UploadModal({ className="visually-hidden" type="file" multiple - accept=".sqlite,.db,.csv,.parquet" + accept=".csv,.parquet" onChange={handleFiles} /> {files.length > 0 && (