diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f186827..a39597e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,20 +2,148 @@ name: CI on: push: - branches: [main] + branches: [main, "release/**"] pull_request: +permissions: + contents: read + jobs: + # The quality gate: lint, tests with a 90% branch-coverage floor, and + # strict typing across the full supported Python matrix (spec §8). test: runs-on: ubuntu-latest strategy: + fail-fast: false matrix: - # 3.9 is the package floor (requires-python), 3.13 is current. - python-version: ["3.9", "3.13"] + python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"] steps: - uses: actions/checkout@v4 - uses: actions/setup-python@v5 with: python-version: ${{ matrix.python-version }} - - run: pip install -e ".[dev]" - - run: pytest -q + - name: Install with dev tooling + run: | + python -m pip install --upgrade pip + pip install -e ".[dev]" + - name: Ruff + run: ruff check inputguard tests eval + - name: Run tests with 90% branch-coverage floor + # Latency asserts are excluded here and gate only in the dedicated + # benchmark job: coverage tracing roughly triples per-call cost + # (observed on run 35277175054 — debug p50 21.14 ms vs 6.59 ms + # untraced), so gating latency under coverage double-gates the same + # budgets under different conditions. + run: | + pytest -q --cov=inputguard --cov-branch --cov-fail-under=90 --strict \ + --ignore=tests/test_benchmarks.py + - name: Strict typing (checks the shipped py.typed) + run: mypy --strict inputguard + + # Latency budgets are gates, not notes: a regression here is a production + # failure mode (the v0.2 unbounded 3.8 s scan). Measured p50/p99 print to + # the job log; budgets and their derivation live in tests/test_benchmarks.py. + benchmark: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + - name: Install with dev tooling + run: | + python -m pip install --upgrade pip + pip install -e ".[dev]" + - name: Latency budgets (p50/p99 recorded) + run: pytest -v -s tests/test_benchmarks.py + + # The zero-dependency promise, verified against the built artifact — not + # the source tree: wheel METADATA carries no Requires-Dist, a bare venv + # install pulls no third-party package, and the public API + CLI work from + # the installed wheel alone. The smoke steps run from a scratch directory + # so the repo's local `inputguard/` package can never shadow the wheel. + wheel-zero-dep: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + - name: Build wheel + run: | + python -m pip install --upgrade pip build + python -m build --wheel + - name: Wheel metadata declares no unconditional runtime dependencies + run: | + python - <<'PY' + import glob + import zipfile + + wheel = glob.glob("dist/*.whl")[0] + names = zipfile.ZipFile(wheel).namelist() + meta_name = next(n for n in names if n.endswith(".dist-info/METADATA")) + meta = zipfile.ZipFile(wheel).read(meta_name).decode() + requires = [ + line + for line in meta.splitlines() + if line.startswith("Requires-Dist:") + ] + unconditional = [ + line for line in requires if 'extra == "' not in line + ] + assert not unconditional, ( + "wheel METADATA declares unconditional runtime dependencies — " + f"the zero-dep promise is broken: {unconditional}" + ) + extra_gated = len(requires) + print( + f"{wheel}: 0 unconditional runtime dependencies " + f"({extra_gated} extra-gated dev-only entries)" + ) + PY + - name: Install wheel into a bare venv + run: | + python -m venv bare + ./bare/bin/pip install --upgrade pip + ./bare/bin/pip install dist/*.whl + - name: Public API works from the installed wheel + run: | + cd "$(mktemp -d)" + "$GITHUB_WORKSPACE/bare/bin/python" - <<'PY' + import json + + import inputguard + + assert inputguard.__version__ + guard = inputguard.InputGuard() + result = guard.analyze("make this faster", domain="coding") + assert isinstance(result.clarity_score, int) + assert 0 <= result.clarity_score <= 100 + payload = result.to_dict() + json.dumps(payload) + print(f"bare-venv analyze OK: {result.status} score={result.clarity_score}") + PY + - name: Bare venv contains no third-party packages + run: | + third_party=$(./bare/bin/pip list --format=freeze \ + | cut -d= -f1 | tr '[:upper:]' '[:lower:]' \ + | grep -v -E '^(inputguard|pip|setuptools)$' || true) + if [ -n "$third_party" ]; then + echo "Third-party packages present after wheel install: $third_party" + exit 1 + fi + echo "bare venv holds only inputguard (+ venv tooling)" + - name: CLI works from the installed wheel + run: | + cd "$(mktemp -d)" + INPUTGUARD="$GITHUB_WORKSPACE/bare/bin/inputguard" + "$INPUTGUARD" analyze "make this faster" --format json + set +e + "$INPUTGUARD" analyze "make this faster" --min-score 85 + code=$? + set -e + if [ "$code" -ne 1 ]; then + echo "expected exit 1 below the min-score floor, got $code" + exit 1 + fi + echo "CLI exit codes OK (0 on analysis, 1 below the min-score floor)" diff --git a/CHANGELOG.md b/CHANGELOG.md index 3cda490..55cf6c3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -42,6 +42,45 @@ in `docs/false-positive-benchmark.md`. carrying a run of 3+ consecutive uncovered-script letters (mixed English+Han). English prompts with loanwords, URLs, or name collisions are pinned unchanged by tests. +- CI as production gates (`.github/workflows/ci.yml`): a Python 3.9–3.13 + matrix running ruff, pytest with a 90% branch-coverage floor + (`--cov-branch --cov-fail-under=90 --strict`, latency asserts excluded + from the traced pass), and mypy `--strict` (the shipped `py.typed` is + now actually checked); a dedicated untraced latency-benchmark job with + budgets measured on this codebase (10k-char build-intent p50 ≤ 65 ms + including the catch-all re-run path, debug and ready paths ≤ 25 ms; a + 1.6 MB input bounded by the policy cap); and a zero-dependency wheel + gate that builds the wheel, installs it into a bare venv, and asserts + no third-party package is present. +- `inputguard` CLI (`inputguard analyze`) via a console-script entry + point: argparse, `--format json` emitting the same `to_dict()` + contract, and CI-friendly exit codes (0 ok, 1 below `--min-score`, + 2 usage error). +- Packaging metadata: 3.13 classifier, Development Status → 4 - Beta, + project URLs, and the explicit dev extras / tool config + (`[tool.ruff]`, `[tool.mypy]`, `[tool.pytest.ini_options]`). + +### Docs +- README rewritten for the v0.3 surface — public exports (including the + extension API), Policy customization, custom rule/domain registration, + follow-up questions, multilingual degradation, and CLI usage — with the + v0.2 example-drift failure mode closed structurally: every README + example is executed by `tests/test_readme_examples.py`, the embedded + `to_dict()` JSON is regenerated from live analyzer output, and the + export list, rule tables, gap vocabulary, and CLI output are checked + against the registry. +- Framework integration recipes: `docs/recipes/langchain.md` and + `docs/recipes/litellm.md`, copy-paste patterns over the `to_dict()` + serialization boundary. Framework packages stay out of core; the recipe + blocks are example-tested with framework stubs in + `tests/test_recipes.py`. +- False-positive benchmark documentation refreshed from a measured run of + `eval/measure_fp.py` on this release state: 116/121 overall, FP 3.3% / + FN 0.8%, zero false positives on true negatives and degradation rows. + The labeling-guide boundary target is stated honestly: 4 of 6 + fixture-style rows still flag (v0.2 baseline 6/6, target 0/6) through + the matcher's deliberate inflection tolerance — documented as an + eval-driven trade-off for a future release, not a label edit. ### Changed - All term matching now happens at word boundaries (`#6`) — detector and diff --git a/README.md b/README.md index 5a2b13c..b3254a6 100644 --- a/README.md +++ b/README.md @@ -3,10 +3,14 @@ Catch unclear inputs before they become bad AI outputs. -InputGuard is a pre-flight input clarity layer. It sits between a user's input and an LLM call. It detects vague, incomplete, or unspecific inputs before they reach the AI — saving the correction cycle that wastes time and tokens when the AI guesses wrong. +InputGuard is a pre-flight input clarity layer. It sits between a user's input and an LLM call. It detects vague, incomplete, or unspecific inputs before they reach the AI — saving the correction cycle that wastes time and tokens when the AI guesses wrong. When it finds gaps, it returns the questions that close them, ready to send back to the user. Zero LLM calls. Zero external dependencies. Pure local Python. +- **Deterministic rules you can extend** — register your own rules and analysis domains through the same typed API the built-ins use. +- **Calibration as data** — status bands, severity penalties, rule filters, and the input cap live in one frozen `Policy` object. +- **Honest by construction** — non-English input is reported as a limitation of the tool, never silently scored `ready`. + --- ## The problem it solves @@ -33,7 +37,7 @@ A non-technical user asks to "fix my code." The AI guesses at the problem, picks pip install inputguard ``` -Python 3.9+. No external dependencies. +Python 3.9+. No external dependencies — `pip install inputguard` imports nothing outside the standard library. --- @@ -45,21 +49,56 @@ from inputguard import InputGuard guard = InputGuard() result = guard.analyze("fix my code") -print(result.detected_intent) # 'debug' -print(result.status) # 'needs_clarification' -print(result.clarity_score) # 35 -print(result.gaps) # ['error description', 'expected vs actual behavior', 'code context'] -print(result.is_clear()) # False +assert result.detected_intent == "debug" +assert result.status == "needs_clarification" +assert result.clarity_score == 35 +assert result.gaps == ["error description", "expected vs actual behavior", "code context"] +assert result.is_clear() is False +``` + +Every gap comes with plain-English advice and the questions that close it: -for rec in result.recommendations: - print(rec["gap"]) - print(rec["what_is_missing"]) - print(rec["what_to_provide"]) - print(rec["why_it_matters"]) - print() +```python +for question in result.follow_ups: + print("-", question) + +assert len(result.follow_ups) == 3 +assert result.follow_ups[0].startswith("What is the exact error message") ``` -`analyze()` takes an optional `domain` argument, which defaults to `"coding"`. Phase 1 supports `"coding"` only. Intent detection is automatic — no extra parameters needed. +`analyze()` takes an optional `domain` argument, which defaults to `"coding"`. Three domains ship built in — `coding`, `writing`, and `data-analysis` — and you can register your own (see [Custom domains](#custom-domains)). + +--- + +## The public API + +Nine names; everything else is internal: + +```python +from inputguard import ( + AnalysisResult, + InputGuard, + Policy, + REGISTRY, + Rule, + RuleFinding, + __version__, + register_domain, + register_rule, +) +``` + +| Name | What it is | +|---|---| +| `InputGuard` | The engine: `InputGuard(mode="warning", policy=None)` | +| `AnalysisResult` | The frozen result of `analyze()` | +| `RuleFinding` | One rule's raw verdict: `code`, `message`, `severity`, `gap` | +| `Policy` | Frozen calibration: bands, penalties, filters, input cap | +| `Rule` | The protocol a custom rule implements | +| `register_rule` | Register one rule (decorator or call) | +| `register_domain` | Register an analysis domain: intent signals + rules | +| `REGISTRY` | The process-global registry (read during `analyze()`) | +| `__version__` | The installed version string | --- @@ -74,12 +113,12 @@ warn_guard = InputGuard(mode="warning") # Strict mode — blocks inputs that fall below the clarity threshold strict_guard = InputGuard(mode="strict") -result_warn = warn_guard.analyze("build a REST API") +result_warn = warn_guard.analyze("build a REST API") result_strict = strict_guard.analyze("build a REST API") -print(result_warn.status) # 'needs_clarification' -print(result_strict.status) # 'blocked' -print(result_warn.clarity_score == result_strict.clarity_score) # True +assert result_warn.status == "needs_clarification" +assert result_strict.status == "blocked" +assert result_warn.clarity_score == result_strict.clarity_score ``` The clarity score is mode-independent. Only the status threshold changes. @@ -96,6 +135,25 @@ Use `warning` when you want to surface gaps to the user without blocking. Use `s --- +## Follow-up questions + +Every gap maps to one or two templated clarifying questions — the sentence the user can answer verbatim. Questions dedupe and order with the gaps: + +```python +result = guard.analyze("make this faster") + +assert result.gaps == [ + "optimization target", + "performance baseline", + "optimization constraint", +] +assert result.follow_ups[0] == "Which function or module should get faster?" +``` + +A gap with no built-in question table entry — including gaps from your own custom rules — gets a documented fallback question, never silence. The same contract holds for recommendations. + +--- + ## Non-English input InputGuard's rules are English-language heuristics. Before any rule runs, a zero-dependency script probe (stdlib `unicodedata` only) classifies the input's script and language. Inputs the rules cannot assess take the explicit degraded path — an uncovered dominant script, Latin-script text recognized as French/Spanish/Portuguese by its function words, or a Latin-dominant input carrying a run of 3+ consecutive uncovered-script letters. In every case InputGuard says so instead of pretending: @@ -103,12 +161,12 @@ InputGuard's rules are English-language heuristics. Before any rule runs, a zero ```python result = guard.analyze("建造一个用户登录应用") -result.status # 'degraded' — never 'ready', in either mode -result.clarity_score # 80 (100 minus the degradation penalty) -result.detected_intent # 'undetermined' -result.detected_language # 'zh' (coarse, script-derived guess) -result.heuristic_coverage # 'none' -result.degradation_note # explains that rules were skipped and why +assert result.status == "degraded" +assert result.clarity_score == 80 # 100 minus the degradation penalty +assert result.detected_intent == "undetermined" +assert result.detected_language == "zh" +assert result.heuristic_coverage == "none" +assert result.degradation_note is not None ``` The rules are **skipped explicitly** — running English keyword rules on text they cannot assess would produce a silent, unearned verdict or spurious gaps invented out of the silence. A degraded result reports the literal `degraded` status in both modes: it is the tool reporting a language limitation of itself, not a judgment of the input's clarity (strict mode's banding would otherwise read as an ordinary critique). In v0.2 this input silently scored 100/ready; v0.3 refuses to assert a confidence it does not have. @@ -131,17 +189,170 @@ The degraded path covers three shapes of input the English rules cannot assess: ```python result = guard.analyze("Preciso de um aplicativo web com login de usuário e relatórios") -result.detected_language # 'pt' (function-word guess) -result.heuristic_coverage # 'none' -result.status # 'degraded' — never 'ready' -result.degradation_note # names Portuguese (pt) and explains the skip +assert result.detected_language == "pt" +assert result.heuristic_coverage == "none" +assert result.status == "degraded" +assert "Portuguese" in result.degradation_note +``` + +--- + +## Custom rules + +The extension contract is four members and one method. Built-in rules register through the exact same path — the API is exercised by all 31 built-in rules before anyone writes their own. + +```python +from typing import Optional + +from inputguard import RuleFinding, register_domain + +class CheckRollbackPlan: + id = "missing_rollback_plan" # unique across the registry + domain = "deploy" # the INTENT name this rule runs for + severity = "high" # "low" | "medium" | "high" — validated + gap = "rollback plan" # groups findings for scoring dedup + + def check(self, text: str) -> Optional[RuleFinding]: + # text arrives normalized: lowercased, whitespace-collapsed. + if "rollback" not in text and "roll back" not in text: + return RuleFinding( + code=self.id, + message="No rollback plan described.", + severity=self.severity, + gap=self.gap, + ) + return None +``` + +The rule above ships with the custom domain in the next section — `register_domain` registers rules through the same path `register_rule` uses, once the intents they name exist. The contracts a rule author accepts: + +- **`check` receives normalized text** (lowercased, whitespace-collapsed) and returns at most one `RuleFinding`, or `None` when the rule does not fire. +- **A `check` exception is never swallowed** — it aborts the `analyze()` call in flight, naming the rule and where it was registered. +- **Registration is permanent** for the process lifetime (there is no unregister), and `REGISTRY` is a process-global singleton shared by everything that imports inputguard. +- **Duplicate rule ids raise `ValueError`** at registration, as do unknown severities and rules naming an intent no registered domain declares. +- **Every gap deserves advice**: a gap with no recommendation or follow-up table entry gets a documented generic fallback — never an empty list and never silence. + +--- + +## Custom domains + +A domain is a named analysis scope: intent signals in priority order, plus its rules. The single intent with empty terms is the fallback for inputs no other intent matches. + +```python +register_domain( + "devops", + { + "deploy": ("deploy", "deployment", "release", "rollout", "ship", "shipping"), + "general": (), # fallback intent — inputs no deploy signal matches + }, + [CheckRollbackPlan], +) + +result = guard.analyze("deploy the new checkout service to production", domain="devops") + +assert result.detected_intent == "deploy" +assert result.gaps == ["rollback plan"] +assert result.clarity_score == 75 +assert result.status == "usable_with_warnings" +assert result.follow_ups == ["Can you add the rollback plan this request is missing?"] +``` + +Intent names are globally unique across domains — registering a domain whose intent another domain already declares raises `ValueError`, because rules dispatch on intent name alone. The same guard rejects duplicate domain names, so re-registering `devops` above raises too. + +--- + +## Policy: calibrate the guard + +Every scoring constant is data on a frozen, validated `Policy` — with the v0.2 values pinned as defaults, so existing behavior cannot drift. Two layers stay separate: per-rule severity decides *what fires*; the policy's bands decide *what happens*. + +| Field | Default | Controls | +|---|---|---| +| `ready_at` | `85` | score ≥ `ready_at` is `ready`, both modes | +| `usable_at` | `60` | warning-mode floor for `usable_with_warnings` | +| `strict_clarify_at` | `65` | strict-mode floor for `needs_clarification`; below it, strict blocks | +| `penalty_low` / `penalty_medium` / `penalty_high` | `5` / `15` / `25` | points per distinct gap, by highest severity | +| `min_words` | `3` | shorter input is never flagged as vague | +| `max_chars` | `10_000` | input cap — truncation is reported, never silent | +| `borderline_at` | `74` | near-miss band just below `ready_at` — the "worth one more pass" signal | +| `disabled_rules` | `frozenset()` | rule ids to skip entirely | +| `allow_patterns` | `()` | regex patterns; matching input is never flagged | + +```python +from inputguard import Policy + +# Disable a rule whose advice your product handles elsewhere — +# the disabled low-severity penalty comes back (55 -> 60) and the +# status flips out of needs_clarification. +tuned = InputGuard( + mode="warning", + policy=Policy(disabled_rules=frozenset({"missing_optimization_constraint"})), +) +result = tuned.analyze("make this faster") +assert result.clarity_score == 60 +assert result.status == "usable_with_warnings" + +# Allowlist internal input shapes — matching input is never flagged. +allowlisted = InputGuard(policy=Policy(allow_patterns=("^re:",))) +result = allowlisted.analyze("re: invoice numbering scheme question") +assert result.status == "ready" +assert result.gaps == [] +``` + +The cap bounds every analysis; a longer input is truncated and the result says so: + +```python +result = InputGuard().analyze("word " * 3000) +assert result.truncated is True +``` + +Policy is validated at construction — mis-ordered bands raise `ValueError` instead of silently distorting statuses — and a per-call `policy=` argument overrides the guard's for one `analyze()` call: + +```python +strict = InputGuard(mode="strict") + +default_bands = strict.analyze("make this faster") +assert default_bands.status == "blocked" # 55 < 65 + +loosened = strict.analyze("make this faster", policy=Policy(usable_at=50, strict_clarify_at=50)) +assert loosened.status == "needs_clarification" # 55 >= 50 ``` --- +## CLI + +A zero-dependency `argparse` front end over the same pipeline, installed as a console script: + +```bash +pip install inputguard +inputguard analyze "make this faster" --mode strict +``` + +```text +status: blocked +clarity: 55/100 +intent: optimization +domain: coding +missing: optimization target, performance baseline, optimization constraint +ask: Which function or module should get faster? +ask: How slow is it today, and what latency would be acceptable? +ask: What must not change while it gets faster — an interface, readability, behavior others depend on? +``` + +`--format json` emits the same `to_dict()` contract the library documents, and `--min-score` turns the CLI into a commit-hook / CI gate: + +```bash +inputguard analyze "$(cat prompt.txt)" --format json --min-score 85 +cat prompt.txt | inputguard analyze --stdin --min-score 85 +``` + +Exit codes: `0` analysis completed (score at or above `--min-score` when given), `1` score below the `--min-score` floor, `2` usage error (unknown flags, missing text, invalid domain or input). + +--- + ## How intent detection works -InputGuard automatically detects what kind of coding input it is receiving. No extra parameters needed. The same `.analyze()` call handles all five intent types. +Each domain detects intent over its own registered signals — insertion order is the priority chain, and the fallback intent catches what no signal matches. The coding domain detects five intents automatically; no extra parameters needed. | Intent | What it covers | Example input | |---|---|---| @@ -150,18 +361,20 @@ InputGuard automatically detects what kind of coding input it is receiving. No e | `optimization` | Performance, speed, refactoring | "make this function faster" | | `explanation` | Understanding code or concepts | "explain what this decorator does" | | `feature` | Adding to existing code | "add search to my existing React app" | +| `compose` *(writing)* | Essays, emails, documents, posts | "write a blog post" | +| `analysis` *(data analysis)* | Analyzing datasets, building reports | "analyze my sales data" | ```python -guard = InputGuard() - -guard.analyze("build a REST API").detected_intent # 'build' -guard.analyze("fix my code").detected_intent # 'debug' -guard.analyze("make this faster").detected_intent # 'optimization' -guard.analyze("explain how async works").detected_intent # 'explanation' -guard.analyze("add search to my existing app").detected_intent # 'feature' +assert guard.analyze("build a REST API").detected_intent == "build" +assert guard.analyze("fix my code").detected_intent == "debug" +assert guard.analyze("make this faster").detected_intent == "optimization" +assert guard.analyze("explain how async works").detected_intent == "explanation" +assert guard.analyze("add search to my existing app").detected_intent == "feature" +assert guard.analyze("write a blog post", domain="writing").detected_intent == "compose" +assert guard.analyze("analyze my sales data", domain="data-analysis").detected_intent == "analysis" ``` -When an input is ambiguous, debug always wins. "Fix this slow function" is a debug request, not optimization. Priority order is: debug → optimization → explanation → feature → build. +When a coding input is ambiguous, debug always wins. "Fix this slow function" is a debug request, not optimization. Priority order is: debug → optimization → explanation → feature → build. --- @@ -223,7 +436,7 @@ Requests like "add search to my existing app", "extend my current API with pagin ### Writing inputs -Requests like "write a blog post", "draft an email", "proofread my essay". Every gap carries its own follow-up questions, and each gap names one thing at a time — same one-gap-one-question discipline as the coding domains. +Requests like "write a blog post", "draft an email", "proofread my essay" — analyzed with `domain="writing"`. Every gap carries its own follow-up questions, and each gap names one thing at a time — same one-gap-one-question discipline as the coding domains. | Rule code | What it catches | Severity | |---|---|---| @@ -236,7 +449,7 @@ Requests like "write a blog post", "draft an email", "proofread my essay". Every ### Data-analysis inputs -Requests like "analyze my sales data", "build a dashboard", "report on this spreadsheet". +Requests like "analyze my sales data", "build a dashboard", "report on this spreadsheet" — analyzed with `domain="data-analysis"`. | Rule code | What it catches | Severity | |---|---|---| @@ -257,9 +470,9 @@ Requests like "analyze my sales data", "build a dashboard", "report on this spre |---|---|---| | `status` | `str` | One of `"ready"`, `"usable_with_warnings"`, `"needs_clarification"`, `"blocked"`, `"degraded"` (language limitation — see Non-English input) | | `clarity_score` | `int` | 0 to 100 | -| `detected_intent` | `str` | Which intent was detected: `build`, `debug`, `optimization`, `explanation`, `feature`, `compose`, or `analysis` | +| `detected_intent` | `str` | The detected intent: `build`, `debug`, `optimization`, `explanation`, `feature`, `compose`, `analysis`, or `undetermined` (degraded inputs) | | `gaps` | `List[str]` | Gap names, in the order rules fired | -| `recommendations` | `List[dict]` | One dict per gap (see next section) | +| `recommendations` | `List[dict]` | One dict per gap (see The gap vocabulary) | | `follow_ups` | `List[str]` | One or two clarifying questions per gap, ready to send back to the user | | `findings` | `List[RuleFinding]` | Raw rule findings (code, message, severity, gap) | | `interpretation_note` | `Optional[str]` | Set when the input is highly ambiguous (score < 50 or two or more high-severity findings) | @@ -268,7 +481,7 @@ Requests like "analyze my sales data", "build a dashboard", "report on this spre | `degradation_note` | `Optional[str]` | Set when rule analysis was skipped or limited by language coverage | | `borderline` | `bool` | `True` when the score sits in the near-miss band just below ready — worth one more pass | | `truncated` | `bool` | `True` when the input was capped at 10,000 characters (only the prefix was analyzed) | -| `score_breakdown` | `Optional[dict]` | Per-contribution score arithmetic, present when the policy exposes it | +| `score_breakdown` | `Optional[dict]` | Per-contribution score arithmetic (base, per-gap penalties, final) | Helpers: @@ -290,132 +503,190 @@ Example `result.to_dict()` for `guard.analyze("fix my code")`: "recommendations": [ { "gap": "error description", - "what_is_missing": "You haven't included the actual error message or exception.", - "what_to_provide": "Copy and paste the exact error message. For example: 'I'm getting TypeError: cannot read property of undefined on line 23'.", - "why_it_matters": "The exact wording tells the AI exactly what went wrong. Without it, the AI guesses and often fixes the wrong thing." + "what_is_missing": "You haven't shared the actual error message, exception, or output you're seeing.", + "what_to_provide": "Include the exact error message or exception you are seeing. Copy and paste it exactly as it appears. For example: 'I'm getting TypeError: cannot read property of undefined on line 23' or 'it throws a 500 Internal Server Error with message: connection refused'. The exact wording tells the AI exactly what went wrong.", + "why_it_matters": "Without the exact error, the AI has to guess what failure mode you're hitting. The wrong guess sends you down a fix path that doesn't apply to your actual problem." }, { "gap": "expected vs actual behavior", - "what_is_missing": "You haven't described what should happen vs what actually happens.", - "what_to_provide": "Describe both. For example: 'it should return a list of users but instead returns None every time'.", - "why_it_matters": "Without this, the AI is guessing what the problem is. It may fix something that was not broken." + "what_is_missing": "You haven't described what you expected to happen and what is actually happening.", + "what_to_provide": "Describe two things: what you expected to happen, and what actually happened. For example: 'I expected the function to return a list of users, but it returns an empty list every time' or 'the button should submit the form but nothing happens when I click it'. Without this, the AI is guessing what the problem is.", + "why_it_matters": "A bug is the gap between what you wanted and what happened. Without both sides, the AI cannot tell what counts as a fix." }, { "gap": "code context", - "what_is_missing": "You haven't pointed to the specific part of your code with the problem.", - "what_to_provide": "Name the language and the function. For example: 'this is a Python function called get_users()'.", - "why_it_matters": "The more specific you are, the more targeted the fix will be." + "what_is_missing": "You haven't pointed to a language, file, function, or snippet for the AI to look at.", + "what_to_provide": "Tell it which language you are using and point to the specific part of your code that has the problem. For example: 'this is a Python function called get_users()' or 'this is in my React component UserList.jsx on line 45'. The more specific you are, the more targeted the fix will be.", + "why_it_matters": "Without a code reference, the AI suggests generic fixes that may not apply to your actual code. Pointing to the exact location lets it propose a precise change." } ], + "follow_ups": [ + "What is the exact error message or exception you're seeing (copy it verbatim if you can)?", + "What did you expect to happen, and what actually happens instead?", + "Which file, function, or part of your code does the problem live in?" + ], "findings": [ - {"code": "missing_error_message", "message": "Debug request detected but no error message or exception described.", "severity": "high", "gap": "error description"}, - {"code": "missing_expected_vs_actual", "message": "No description of expected vs actual behavior provided.", "severity": "high", "gap": "expected vs actual behavior"}, - {"code": "missing_debug_code_context", "message": "No code context provided.", "severity": "medium", "gap": "code context"} + { + "code": "missing_error_message", + "message": "Debug request detected but no error message or exception described.", + "severity": "high", + "gap": "error description" + }, + { + "code": "missing_expected_vs_actual", + "message": "No description of expected vs actual behavior provided.", + "severity": "high", + "gap": "expected vs actual behavior" + }, + { + "code": "missing_debug_code_context", + "message": "No code context provided — no language, function name, or snippet referenced.", + "severity": "medium", + "gap": "code context" + } ], - "interpretation_note": "This input is ambiguous in multiple ways. Addressing each gap below before sending will prevent the AI from making assumptions that lead to the wrong output." + "interpretation_note": "This input is ambiguous in multiple ways. Addressing each gap below before sending will prevent the AI from making assumptions that lead to the wrong output.", + "detected_language": "en", + "heuristic_coverage": "full", + "degradation_note": null, + "borderline": false, + "truncated": false, + "score_breakdown": { + "base": 100, + "penalties": [ + { + "code": "missing_error_message", + "severity": "high", + "points": -25 + }, + { + "code": "missing_expected_vs_actual", + "severity": "high", + "points": -25 + }, + { + "code": "missing_debug_code_context", + "severity": "medium", + "points": -15 + } + ], + "final": 35 + } } ``` --- -## Recommendations +## The gap vocabulary + +Every built-in gap string, by intent. These strings are de-facto API — consumers switch on them — and every one has a recommendation and at least one follow-up question. + +- **Build:** `programming language`, `api structure`, `data model`, `integration specifics`, `authentication type`, `output format`, `task context` +- **Debug:** `error description`, `expected vs actual behavior`, `code context` +- **Optimization:** `optimization target`, `performance baseline`, `optimization constraint` +- **Explanation:** `code reference`, `explanation depth` +- **Feature:** `existing stack`, `feature scope`, `completion criteria` +- **Compose:** `audience`, `purpose`, `structure/format`, `source material`, `context`, `completeness` +- **Analysis:** `dataset/source`, `question/goal`, `output format`, `tooling`, `volume`, `reproducibility` Every entry in `result.recommendations` is a plain dict with four keys, all written for non-technical users: ```python +result = guard.analyze("fix my code") for rec in result.recommendations: + assert set(rec) == {"gap", "what_is_missing", "what_to_provide", "why_it_matters"} print(rec["gap"]) # which gap this addresses print(rec["what_is_missing"]) # plain English — what the user forgot print(rec["what_to_provide"]) # concrete example they can copy print(rec["why_it_matters"]) # what goes wrong if they skip it ``` -Gap names by intent type: - -- **Build:** `programming language`, `api structure`, `data model`, `integration specifics`, `authentication type`, `output format`, `task context` -- **Debug:** `error description`, `expected vs actual behavior`, `code context` -- **Optimization:** `optimization target`, `performance baseline`, `optimization constraint` -- **Explanation:** `code reference`, `explanation depth` -- **Feature:** `existing stack`, `feature scope`, `completion criteria` - --- ## Real-world examples ```python from inputguard import InputGuard + guard = InputGuard() # Build — vague -guard.analyze("build me an app") -# detected_intent: 'build' -# score: 60 -# status: usable_with_warnings -# gaps: ['programming language', 'output format'] +result = guard.analyze("build me an app") +assert result.detected_intent == "build" +assert result.clarity_score == 60 +assert result.status == "usable_with_warnings" +assert result.gaps == ["programming language", "output format"] # Debug — vague -guard.analyze("fix my code") -# detected_intent: 'debug' -# score: 35 -# status: needs_clarification -# gaps: ['error description', 'expected vs actual behavior', 'code context'] +result = guard.analyze("fix my code") +assert result.clarity_score == 35 +assert result.status == "needs_clarification" +assert result.gaps == ["error description", "expected vs actual behavior", "code context"] # Optimization — vague -guard.analyze("make this faster") -# detected_intent: 'optimization' -# score: 55 -# status: needs_clarification -# gaps: ['optimization target', 'performance baseline', 'optimization constraint'] +result = guard.analyze("make this faster") +assert result.clarity_score == 55 +assert result.status == "needs_clarification" +assert result.gaps == ["optimization target", "performance baseline", "optimization constraint"] # Build — fully specified -guard.analyze( +result = guard.analyze( "Build a REST API using FastAPI. " "Store users in PostgreSQL with fields: id, name, email. " "Expose GET /users and POST /users endpoints. " "Add JWT authentication." ) -# detected_intent: 'build' -# score: 100 -# status: ready -# gaps: [] +assert result.detected_intent == "build" +assert result.clarity_score == 100 +assert result.status == "ready" +assert result.gaps == [] # Debug — fully specified -guard.analyze( +result = guard.analyze( "Fix this Python function get_users() — it should return a list " "of user dicts but instead returns None. " "The error says: TypeError: NoneType is not iterable on line 45." ) -# detected_intent: 'debug' -# score: 100 -# status: ready -# gaps: [] +assert result.detected_intent == "debug" +assert result.clarity_score == 100 +assert result.status == "ready" +assert result.gaps == [] ``` --- ## Integration pattern -Drop it in front of your existing LLM call. Two minimal patterns: +Drop it in front of your existing LLM call: ```python from inputguard import InputGuard guard = InputGuard(mode="warning") +def call_your_llm(user_input: str) -> str: + return "..." # your real LLM call + def handle_user_input(user_input: str): result = guard.analyze(user_input) - if result.status in ("needs_clarification", "blocked"): - # Return feedback to the user before calling the LLM + if result.status in ("needs_clarification", "blocked", "degraded"): + # Surface the gaps and the questions that close them, + # instead of calling the LLM. `degraded` inputs arrive here + # in both modes; `blocked` only in strict mode. return { "status": result.status, "detected_intent": result.detected_intent, "gaps": result.gaps, + "follow_ups": result.follow_ups, "recommendations": result.recommendations, } # Input is clear enough — proceed to LLM return call_your_llm(user_input) + +payload = handle_user_input("fix my code") +assert payload["status"] == "needs_clarification" ``` For a hard gate, use `mode="strict"` and check `result.is_clear()`: @@ -428,13 +699,17 @@ def handle_user_input(user_input: str): if not result.is_clear(): return { "status": result.status, - "detected_intent": result.detected_intent, "gaps": result.gaps, - "recommendations": result.recommendations, + "follow_ups": result.follow_ups, } return call_your_llm(user_input) + +payload = handle_user_input("make this faster") +assert payload["status"] == "blocked" ``` +Framework-ready versions of this pattern — LangChain and LiteLLM, consuming `to_dict()` output — live in [docs/recipes/](docs/recipes/). The recipes are copy-paste patterns, not adapter packages: inputguard stays zero-dependency. + --- ## Package layout @@ -442,29 +717,27 @@ def handle_user_input(user_input: str): ``` inputguard/ ├── inputguard/ -│ ├── __init__.py -│ ├── analyzer.py -│ ├── detector.py -│ ├── recommender.py -│ ├── scorer.py -│ ├── types.py +│ ├── __init__.py # public exports +│ ├── analyzer.py # the analyze() pipeline +│ ├── cli.py # the inputguard console script +│ ├── detector.py # intent detection (coding signal chain) +│ ├── followups.py # per-gap clarifying questions +│ ├── language.py # script probe / multilingual degradation +│ ├── matching.py # shared word-boundary term matcher +│ ├── policy.py # Policy — calibration as data +│ ├── recommender.py # per-gap recommendations +│ ├── registry.py # Rule protocol, register_rule, register_domain +│ ├── scorer.py # scoring and status banding +│ ├── types.py # AnalysisResult, RuleFinding │ ├── py.typed -│ └── rules/ -│ ├── __init__.py -│ ├── coding.py -│ ├── debug.py -│ ├── optimization.py -│ ├── explanation.py -│ └── feature.py +│ └── rules/ # the 31 built-in rules +│ ├── coding.py # build, debug, optimization, explanation, feature +│ ├── writing.py # compose +│ └── data_analysis.py # analysis +├── eval/ # versioned 121-case clarity-evaluation set +├── docs/ # benchmark + framework recipes ├── tests/ -│ ├── test_coding.py -│ ├── test_detector.py -│ ├── test_debug.py -│ ├── test_optimization.py -│ ├── test_explanation.py -│ └── test_feature.py -├── pyproject.toml -└── README.md +└── pyproject.toml ``` --- @@ -473,9 +746,15 @@ inputguard/ ```bash pip install -e ".[dev]" -python -m pytest tests/ -v +python -m pytest tests/ +ruff check . +mypy # strict — the shipped py.typed is checked +python -m pytest tests/ --cov=inputguard --cov-branch --cov-fail-under=90 +python3 eval/measure_fp.py # clarity-evaluation benchmark (see docs/) ``` +CI runs the same gates on Python 3.9–3.13, plus a latency-benchmark job and a zero-dependency wheel check. + --- ## Publishing @@ -492,4 +771,4 @@ python -m twine upload dist/* MIT — see [LICENSE](LICENSE) for the full text. -Copyright © 2026 Nihanth Kalisetti. \ No newline at end of file +Copyright © 2026 Nihanth Kalisetti. diff --git a/docs/false-positive-benchmark.md b/docs/false-positive-benchmark.md index b2e02d4..f65d773 100644 --- a/docs/false-positive-benchmark.md +++ b/docs/false-positive-benchmark.md @@ -19,7 +19,7 @@ Last run: v0.3.0 release branch, 2026-09-17. | **overall** | **121** | **116** | **4** | **1** | **3.3%** | **0.8%** | Degradation honesty: a degradation note is present on 14 of 14 degradation -rows. Performance wall time (single pass): PF-001 22 ms, PF-002 22 ms — +rows. Performance wall time (single pass): PF-001 19 ms, PF-002 20 ms — PF-002 exercises the 10,000-character cap with the `truncated` flag set. Zero false positives on true negatives and zero on degradation rows is the @@ -27,21 +27,50 @@ load-bearing number: it says the English keyword rules fire only on English input, and that non-English input degrades instead of producing invented gaps. +## Fixture-style boundary target: 4 of 6, against v0.2's 6 of 6 + +The labeling guide (`art_XPvHhPeZ`) sets the release target: the six +fixture-style boundary rows (BD-002, BD-003, BD-004, BD-006, BD-013, +BD-014) must come back clean, with BD-012 — the genuine-debug-request +no-regression guard — still passing. The v0.2 baseline flagged all six +(substring matching read "fixture" as "fix"). + +Measured at this release: **2 of 6 pass cleanly (BD-003, BD-013), and +BD-012 passes**, but 4 of 6 still flag (BD-002, BD-004, BD-006, BD-014). +The 0/6 target is therefore **not met** — the honest read is "improved from +6/6 to 4/6 flagged, short of the 0/6 target." + +The mechanism is deliberate, verified in `inputguard/matching.py`: the +shared matcher kills the embedded-word false positives ("fixture" is no +longer "fix"), but it still absorbs common inflectional endings +(`-s`, `-ed`, `-ing`, `-er`, `-ly`, ...) so natural word forms keep firing +— "debugged" still matches "debug", "slowly" still matches "slow". The +class-of-hit the v0.2 coding matcher preserved was read as recall on +real inputs; these four labels were written against strict +boundary-only semantics and sit exactly on that trade-off. Closing them +means re-deciding that recall trade-off on the labeled set — an +eval-driven change for a future release, never a label edit. + +For the spec verification row "False positive eliminated", this run +records: v0.2 baseline 6/6 flagged; v0.3 measured 4/6 flagged with +BD-012 (no-regression guard) passing and zero false positives on all 37 +true negatives. + ## Residual mismatches (5) -The 5 unmatched rows are all pre-existing behavior on the coding domain, -known at label time and outside the v0.3 feature work: +The 5 unmatched rows, per the measured run at the top of this document: -- `TP-FEA-02` — false negative: `feature scope` gap not flagged; the row - comes back one status above expected (`usable_with_warnings` vs `ready`). -- `BD-002`, `BD-004`, `BD-006` — boundary rows: the detected intent - switches (build → optimization / debug) and the coding rules of the other - intent fire, adding spurious gaps. -- `BD-014` — boundary row with the same intent-adjacency shape. +- `TP-FEA-02` — false negative: the `feature scope` gap is not flagged; the + row lands one status above expected (`usable_with_warnings` vs `ready`). +- `BD-002`, `BD-004`, `BD-006`, `BD-014` — fixture-style boundary rows + still flagging through the matcher's deliberate inflection tolerance + (see the target section above): the detected intent flips + (build → optimization / debug) and the other intent's coding rules fire, + adding spurious gaps. -These are candidate labels to re-examine or intent-detector work for a -future release; per `eval/README.md`, labels are never edited to make a -measurement look better. +These are candidate re-evaluations of the inflection-recall trade-off, or +intent-detector work, for a future release; per `eval/README.md`, labels +are never edited to make a measurement look better. ## History within the release wave diff --git a/docs/recipes/langchain.md b/docs/recipes/langchain.md new file mode 100644 index 0000000..18640bb --- /dev/null +++ b/docs/recipes/langchain.md @@ -0,0 +1,82 @@ +# Recipe: InputGuard as a LangChain pre-flight gate + +InputGuard is framework-agnostic: it takes a string and returns a plain `to_dict()` payload, so wiring it into a LangChain chain is one `RunnableLambda` in front of your model call. There is no adapter package and no shared dependency — `inputguard` stays zero-dependency and imports nothing from `langchain-core`. + +> **Prerequisites:** `pip install inputguard` (and your own LangChain setup — inputguard installs none of it). This recipe targets langchain-core's runnable protocol; the same shape works for LCEL chains, agents, and LangGraph nodes. + +## 1 · The pre-flight payload (pure inputguard) + +This step is framework-free — a plain function returning the `to_dict()` contract your chain branches on. It is executed by CI (`tests/test_recipes.py`), including its assertions: + +```python +from inputguard import InputGuard + +guard = InputGuard() # warning mode: measure and flag, do not block + +def preflight(question: str) -> dict: + """Analyze input and return the to_dict() contract the chain routes on.""" + payload = guard.analyze(question).to_dict() + + if payload["status"] in ("needs_clarification", "blocked", "degraded"): + return { + "route": "clarify", + "question": question, + "gaps": payload["gaps"], + "follow_ups": payload["follow_ups"], + } + return {"route": "model", "question": question} + + +routed = preflight("fix my code") +assert routed["route"] == "clarify" +assert routed["gaps"] == ["error description", "expected vs actual behavior", "code context"] +assert routed["follow_ups"][0].startswith("What is the exact error message") + +specified = preflight( + "Fix this Python function get_users() — it should return a list " + "of user dicts but instead returns None. " + "The error says: TypeError: NoneType is not iterable on line 45." +) +assert specified["route"] == "model" +``` + +`degraded` rides the clarify route too: it is InputGuard reporting its own language limitation, not a judgment of the input — but the safest behavior is still to ask the user rather than run English-only analysis downstream. + +## 2 · Slot it into a chain + +The clarify path never touches the model — in the demo below, `call_your_model` is a `RuntimeError` to prove it: + +```python +from langchain_core.runnables import RunnableLambda + +def clarify_message(payload: dict) -> dict: + return { + **payload, + "message": "Before I can help, could you answer: " + " ".join(payload["follow_ups"]), + } + +def call_your_model(question: str) -> str: + # Replace with your real model invocation, e.g. ChatOpenAI(...).invoke(...) + raise RuntimeError("the clarify path must never reach the model") + +chain = RunnableLambda(preflight) | RunnableLambda( + lambda payload: clarify_message(payload) + if payload["route"] == "clarify" + else call_your_model(payload["question"]) +) + +result = chain.invoke("fix my code") +assert result["route"] == "clarify" +assert result["message"].startswith("Before I can help") +``` + +## Notes + +- **Warning vs strict:** in warning mode only `needs_clarification` and `degraded` arrive at the clarify branch. In strict mode `blocked` joins them — same route, more refusal. Pick via `InputGuard(mode=...)`; the payload contract does not change. +- **Where the score lives:** `payload["clarity_score"]` and `payload["score_breakdown"]` are on every result if you want to log the arithmetic or threshold on your side. +- **Follow-ups are ready to send:** `follow_ups` are plain sentences, deduped and gap-ordered — drop them into your clarification message verbatim. +- **Streaming/agents:** call `preflight()` before entering the agent loop; the payload is a plain dict, safe to attach to any run state. + +--- + +🔗 [Obvious Project](https://app.obvious.ai/p/inputguard-package-upgrade-plan-ZWAga4Fk) · 🧵 [Obvious Thread](https://app.obvious.ai/p/inputguard-package-upgrade-plan-ZWAga4Fk?thread=th_2ZBoxRug) diff --git a/docs/recipes/litellm.md b/docs/recipes/litellm.md new file mode 100644 index 0000000..bc55657 --- /dev/null +++ b/docs/recipes/litellm.md @@ -0,0 +1,87 @@ +# Recipe: InputGuard as a LiteLLM gate + +LiteLLM routes one OpenAI-shaped call to a hundred providers. InputGuard sits in front of that call: analyze the prompt first, and only forward to `litellm.completion(...)` when the payload says the input is worth the tokens. There is no adapter package and no shared dependency — `inputguard` stays zero-dependency and imports nothing from `litellm`. + +> **Prerequisites:** `pip install inputguard` (and your own LiteLLM setup — inputguard installs none of it). + +## 1 · The gate decision (pure inputguard) + +This step is framework-free — a strict-mode guard turns vague input into a refusal without a model call. It is executed by CI (`tests/test_recipes.py`), including its assertions: + +```python +from inputguard import InputGuard + +guard = InputGuard(mode="strict") # hard gate: refuse to forward vague input + +def guard_prompt(question: str) -> dict: + """Return {"ok": False, ...} to refuse, or {"ok": True, "prompt": ...} to forward.""" + payload = guard.analyze(question).to_dict() + + if payload["status"] != "ready": + return { + "ok": False, + "status": payload["status"], + "gaps": payload["gaps"], + "follow_ups": payload["follow_ups"], + } + return {"ok": True, "prompt": question} + + +blocked = guard_prompt("make this faster") +assert blocked["ok"] is False +assert blocked["status"] == "blocked" +assert blocked["gaps"] == [ + "optimization target", + "performance baseline", + "optimization constraint", +] +assert blocked["follow_ups"][0] == "Which function or module should get faster?" + +specified = guard_prompt( + "Fix this Python function get_users() — it should return a list " + "of user dicts but instead returns None. " + "The error says: TypeError: NoneType is not iterable on line 45." +) +assert specified["ok"] is True +``` + +A `degraded` payload also fails the gate (`"ready"` is required): non-English input gets an honest note back instead of a silent pass. + +## 2 · Wire it in front of the completion call + +The clarify path returns before `litellm.completion` is ever reached — in the CI run the stub raises if it is: + +```python +import litellm + +def answer(question: str) -> str: + decision = guard_prompt(question) + + if not decision["ok"]: + return ( + "I need more detail before I can help.\n" + "Missing: " + ", ".join(decision["gaps"]) + "\n" + "Could you answer: " + " ".join(decision["follow_ups"]) + ) + + response = litellm.completion( + model="gpt-4o-mini", # LiteLLM routes to 100+ providers behind this one call + messages=[{"role": "user", "content": decision["prompt"]}], + ) + return response.choices[0].message.content + + +reply = answer("make this faster") +assert reply.startswith("I need more detail") +``` + +## Notes + +- **Proxy deployments:** run the same check at the proxy edge — `guard_prompt` is a pure function, so a LiteLLM pre-call hook or a tiny FastAPI route in front of the proxy can reuse it verbatim. +- **Audit trail:** `payload["score_breakdown"]` gives you the per-gap arithmetic for logs; `payload["detected_intent"]` and `payload["detected_language"]` are useful routing metadata. +- **Softer rollout:** start with `InputGuard()` in warning mode and log `"ok": False` cases without refusing; flip to `mode="strict"` once you trust the measured false-positive rate (see [docs/false-positive-benchmark.md](../false-positive-benchmark.md)). +- **Cost math:** every refused call is a completion you never paid for; the gate costs one local, dependency-free `analyze()` pass. + +--- + +🔗 [Obvious Project](https://app.obvious.ai/p/inputguard-package-upgrade-plan-ZWAga4Fk) · 🧵 [Obvious Thread](https://app.obvious.ai/p/inputguard-package-upgrade-plan-ZWAga4Fk?thread=th_2ZBoxRug) diff --git a/inputguard/cli.py b/inputguard/cli.py new file mode 100644 index 0000000..0565439 --- /dev/null +++ b/inputguard/cli.py @@ -0,0 +1,134 @@ +"""The ``inputguard`` command-line interface (spec art_bTvdPdJS §5). + +A zero-dependency argparse front end over the same ``InputGuard.analyze`` +pipeline the library exposes. JSON output is the result's own ``to_dict()`` +contract — the CLI adds no fields and renames none. + +Exit codes (the CI/commit-hook contract): + +- ``0`` — analysis completed; with ``--min-score N``, the score is at or + above the floor. +- ``1`` — analysis completed but the clarity score is below the + ``--min-score`` floor. Use with ``--min-score`` to gate commits/PRs. +- ``2`` — usage error: unknown flags, missing text, or an invalid value + (unknown domain, empty input) reported by the pipeline. + +Examples:: + + inputguard analyze "make this faster" + inputguard analyze "make this faster" --mode strict + inputguard analyze "$(cat prompt.txt)" --format json --min-score 85 + cat prompt.txt | inputguard analyze --stdin --min-score 85 +""" + +from __future__ import annotations + +import argparse +import json +import sys +from typing import Optional, Sequence + +from inputguard import AnalysisResult, InputGuard + +EXIT_OK = 0 +EXIT_BELOW_FLOOR = 1 +EXIT_USAGE = 2 + + +def _build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + prog="inputguard", + description="Pre-flight LLM inputs for clarity before inference.", + ) + subparsers = parser.add_subparsers(dest="command", required=True) + analyze = subparsers.add_parser( + "analyze", + help="Analyze input text and report its clarity verdict.", + ) + analyze.add_argument( + "text", + nargs="?", + help="The input text to analyze; omit when --stdin is given.", + ) + analyze.add_argument( + "--stdin", + action="store_true", + help="Read the input text from stdin instead of the TEXT argument.", + ) + analyze.add_argument( + "--domain", + default="coding", + help="Registered domain to analyze against (default: coding).", + ) + analyze.add_argument( + "--mode", + choices=("warning", "strict"), + default="warning", + help="Status banding mode (default: warning).", + ) + analyze.add_argument( + "--format", + choices=("text", "json"), + default="text", + help="Output format (default: text). json emits the to_dict() contract.", + ) + analyze.add_argument( + "--min-score", + type=int, + default=None, + metavar="N", + help=( + "Exit 1 when the clarity score is below N — a ready-made " + "commit-hook / CI gate." + ), + ) + return parser + + +def _render_text(result: AnalysisResult, domain: str) -> str: + """Human-readable verdict, one fact per line (spec §5 example shape).""" + lines = [ + f"status: {result.status}", + f"clarity: {result.clarity_score}/100", + f"intent: {result.detected_intent}", + f"domain: {domain}", + ] + if result.gaps: + lines.append(f"missing: {', '.join(result.gaps)}") + for question in result.follow_ups: + lines.append(f"ask: {question}") + if result.degradation_note is not None: + lines.append(f"note: {result.degradation_note}") + return "\n".join(lines) + + +def main(argv: Optional[Sequence[str]] = None) -> int: + """Run one analysis; returns the process exit code (see module docstring).""" + parser = _build_parser() + args = parser.parse_args(argv) + + text = sys.stdin.read() if args.stdin else args.text + if not text: + parser.error("provide the text to analyze as TEXT, or pass --stdin") + + guard = InputGuard(mode=args.mode) + try: + result = guard.analyze(text, domain=args.domain) + except ValueError as exc: + # Invalid domain / empty-after-normalization input: a usage error, + # not a crash — the same ValueError contract the library documents. + print(f"inputguard: {exc}", file=sys.stderr) + return EXIT_USAGE + + if args.format == "json": + print(json.dumps(result.to_dict(), indent=2)) + else: + print(_render_text(result, args.domain)) + + if args.min_score is not None and result.clarity_score < args.min_score: + return EXIT_BELOW_FLOOR + return EXIT_OK + + +if __name__ == "__main__": # pragma: no cover — manual invocation convenience + sys.exit(main()) diff --git a/inputguard/detector.py b/inputguard/detector.py index a86fe96..bb62444 100644 --- a/inputguard/detector.py +++ b/inputguard/detector.py @@ -70,7 +70,7 @@ def normalize(text: str) -> str: return re.sub(r"\s+", " ", text.strip().lower()) -def _contains_any(text: str, terms) -> bool: +def _contains_any(text: str, terms: Iterable[str]) -> bool: # v0.3: word-boundary matching shared with the rule modules — "fixture" # is no longer read as the debug signal "fix" (probe P1). return contains_any(text, terms) diff --git a/inputguard/language.py b/inputguard/language.py index caabe0b..ca9ddfe 100644 --- a/inputguard/language.py +++ b/inputguard/language.py @@ -47,7 +47,7 @@ import re import unicodedata from dataclasses import dataclass -from typing import Dict, Optional +from typing import Dict, FrozenSet, Optional, Set __all__ = [ "COVERAGE_FULL", @@ -228,7 +228,7 @@ # of 3+ letters is a word or clause the English heuristics cannot read. _UNCOVERED_BLOCK_DEGRADES_AT = 3 -_LATIN_FUNCTION_WORDS: Dict[str, frozenset] = { +_LATIN_FUNCTION_WORDS: Dict[str, FrozenSet[str]] = { "en": frozenset({ "a", "an", "the", "and", "or", "but", "if", "then", "of", "to", "in", "on", "at", "by", "for", "with", "from", "into", "over", "under", @@ -357,7 +357,7 @@ def _latin_language_of(sample: str) -> Optional[str]: Ties between firing languages resolve alphabetically, like the script histogram's dominant-script tie-break. """ - distinct: Dict[str, set] = {} + distinct: Dict[str, Set[str]] = {} for raw in sample.translate(_APOSTROPHES).lower().split(): token = _EDGE_TRIM.sub("", raw) if not token: diff --git a/inputguard/recommender.py b/inputguard/recommender.py index b36b7a4..a5f8ed3 100644 --- a/inputguard/recommender.py +++ b/inputguard/recommender.py @@ -3,7 +3,7 @@ from typing import Dict, List -_RECOMMENDATIONS: Dict[str, dict] = { +_RECOMMENDATIONS: Dict[str, Dict[str, str]] = { "programming language": { "gap": "programming language", "what_is_missing": "You haven't told it which programming language or technology to use.", @@ -183,7 +183,7 @@ } -def _fallback_recommendation(gap: str) -> dict: +def _fallback_recommendation(gap: str) -> Dict[str, str]: """Generic four-key advice for a gap with no curated entry. The documented fallback (spec art_bTvdPdJS §6): unknown gaps keep their @@ -198,13 +198,13 @@ def _fallback_recommendation(gap: str) -> dict: } -def get_recommendations(gaps: List[str]) -> List[dict]: +def get_recommendations(gaps: List[str]) -> List[Dict[str, str]]: """Build one four-key recommendation per gap, in input order. A gap without a curated entry gets the documented fallback (see :func:`_fallback_recommendation`) — never a silent drop. """ - out: List[dict] = [] + out: List[Dict[str, str]] = [] for gap in gaps: entry = _RECOMMENDATIONS.get(gap) out.append(dict(entry) if entry is not None else _fallback_recommendation(gap)) diff --git a/inputguard/registry.py b/inputguard/registry.py index 6261bf2..dd27943 100644 --- a/inputguard/registry.py +++ b/inputguard/registry.py @@ -121,7 +121,10 @@ def check(self, text: str) -> Optional[RuleFinding]: def _instantiate(rule_cls: type) -> Rule: try: - return rule_cls() + # Bare ``type`` construction is typed Any; the annotation pins the + # protocol so strict mode sees a Rule, not Any. + instance: Rule = rule_cls() + return instance except TypeError as exc: raise TypeError( f"Cannot register rule class {rule_cls.__name__!r}: it must be " diff --git a/inputguard/rules/__init__.py b/inputguard/rules/__init__.py index 0a6f434..c5f350b 100644 --- a/inputguard/rules/__init__.py +++ b/inputguard/rules/__init__.py @@ -36,6 +36,14 @@ "run_optimization_rules", "run_explanation_rules", "run_feature_rules", + # Backward-compat re-exports of the v0.2 signal tables (INTENT_SIGNALS is + # consumed by the coding registration above; the per-intent tables stay + # importable for code that read them from this module in v0.2). + "INTENT_SIGNALS", + "DEBUG_SIGNALS", + "EXPLANATION_SIGNALS", + "FEATURE_SIGNALS", + "OPTIMIZATION_SIGNALS", ] # The coding domain: intent signals in strict priority order (the single diff --git a/inputguard/rules/coding.py b/inputguard/rules/coding.py index 2b3b9ff..75b5369 100644 --- a/inputguard/rules/coding.py +++ b/inputguard/rules/coding.py @@ -1,7 +1,7 @@ from __future__ import annotations import re -from typing import List, Optional, Set +from typing import Iterable, List, Optional, Set from inputguard.matching import contains_any from inputguard.registry import register_rule @@ -111,7 +111,7 @@ def _normalize(text: str) -> str: return re.sub(r"\s+", " ", text.lower()).strip() -def _contains_any(text: str, terms) -> bool: +def _contains_any(text: str, terms: Iterable[str]) -> bool: # v0.3: word-boundary matching via the shared matcher. token_fallback # keeps the v0.2 coding-rule behavior for multiword terms ("def ", # "sign in with"): their words may appear non-adjacent. diff --git a/inputguard/rules/debug.py b/inputguard/rules/debug.py index 7431f15..cd79ec7 100644 --- a/inputguard/rules/debug.py +++ b/inputguard/rules/debug.py @@ -1,7 +1,7 @@ from __future__ import annotations import re -from typing import List, Optional +from typing import Iterable, List, Optional from inputguard.detector import DEBUG_SIGNALS from inputguard.matching import contains_any @@ -43,7 +43,7 @@ def _normalize(text: str) -> str: return re.sub(r"\s+", " ", text.strip().lower()) -def _contains_any(text: str, terms) -> bool: +def _contains_any(text: str, terms: Iterable[str]) -> bool: # v0.3: word-boundary matching via the shared matcher (probe P1 fix). return contains_any(text, terms) diff --git a/inputguard/rules/explanation.py b/inputguard/rules/explanation.py index 69498de..9610b5e 100644 --- a/inputguard/rules/explanation.py +++ b/inputguard/rules/explanation.py @@ -1,7 +1,7 @@ from __future__ import annotations import re -from typing import List, Optional +from typing import Iterable, List, Optional from inputguard.detector import EXPLANATION_SIGNALS from inputguard.matching import contains_any @@ -37,7 +37,7 @@ def _normalize(text: str) -> str: return re.sub(r"\s+", " ", text.strip().lower()) -def _contains_any(text: str, terms) -> bool: +def _contains_any(text: str, terms: Iterable[str]) -> bool: # v0.3: word-boundary matching via the shared matcher (probe P1 fix). return contains_any(text, terms) diff --git a/inputguard/rules/feature.py b/inputguard/rules/feature.py index 51b6948..891653d 100644 --- a/inputguard/rules/feature.py +++ b/inputguard/rules/feature.py @@ -1,7 +1,7 @@ from __future__ import annotations import re -from typing import List, Optional +from typing import Iterable, List, Optional from inputguard.detector import FEATURE_SIGNALS from inputguard.matching import contains_any @@ -27,7 +27,7 @@ "oauth", "jwt", "session", "api key", "role-based", "rbac", "upload", "download", "preview", "thumbnail", "dashboard", "chart", "graph", "table", "export", - "search by", "filter by", "sort by", "group by", + "search by", "group by", } COMPLETION_CRITERIA_SIGNALS = { @@ -44,7 +44,7 @@ def _normalize(text: str) -> str: return re.sub(r"\s+", " ", text.strip().lower()) -def _contains_any(text: str, terms) -> bool: +def _contains_any(text: str, terms: Iterable[str]) -> bool: # v0.3: word-boundary matching via the shared matcher (probe P1 fix). return contains_any(text, terms) diff --git a/inputguard/rules/optimization.py b/inputguard/rules/optimization.py index fdfc587..e6eda25 100644 --- a/inputguard/rules/optimization.py +++ b/inputguard/rules/optimization.py @@ -1,7 +1,7 @@ from __future__ import annotations import re -from typing import List, Optional +from typing import Iterable, List, Optional from inputguard.detector import OPTIMIZATION_SIGNALS from inputguard.matching import contains_any @@ -43,7 +43,7 @@ def _normalize(text: str) -> str: return re.sub(r"\s+", " ", text.strip().lower()) -def _contains_any(text: str, terms) -> bool: +def _contains_any(text: str, terms: Iterable[str]) -> bool: # v0.3: word-boundary matching via the shared matcher (probe P1 fix). return contains_any(text, terms) diff --git a/inputguard/rules/writing.py b/inputguard/rules/writing.py index a0b8007..87a992e 100644 --- a/inputguard/rules/writing.py +++ b/inputguard/rules/writing.py @@ -397,7 +397,7 @@ def check(self, text: str) -> Optional[RuleFinding]: # The single fallback intent: every writing-domain input is a composition # task. Exactly one empty-terms intent is what the registry requires. -WRITING_SIGNALS: Tuple[Tuple[str, tuple], ...] = ( +WRITING_SIGNALS: Tuple[Tuple[str, Tuple[str, ...]], ...] = ( ("compose", ()), ) diff --git a/inputguard/types.py b/inputguard/types.py index 73bde72..85316c5 100644 --- a/inputguard/types.py +++ b/inputguard/types.py @@ -27,7 +27,7 @@ class AnalysisResult: clarity_score: int detected_intent: str gaps: List[str] = field(default_factory=list) - recommendations: List[dict] = field(default_factory=list) + recommendations: List[Dict[str, Any]] = field(default_factory=list) findings: List[RuleFinding] = field(default_factory=list) interpretation_note: Optional[str] = None # v0.3, additive: templated clarifying questions, one or two per gap, @@ -48,7 +48,7 @@ class AnalysisResult: truncated: bool = False score_breakdown: Optional[Dict[str, Any]] = None - def to_dict(self) -> dict: + def to_dict(self) -> Dict[str, Any]: return { "status": self.status, "clarity_score": self.clarity_score, diff --git a/pyproject.toml b/pyproject.toml index 93c94f3..8a3e547 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -14,7 +14,7 @@ keywords = [ "agents", "rag", "input-guard", "prompt-quality" ] classifiers = [ - "Development Status :: 3 - Alpha", + "Development Status :: 4 - Beta", "Intended Audience :: Developers", "Operating System :: OS Independent", "Programming Language :: Python :: 3", @@ -22,13 +22,48 @@ classifiers = [ "Programming Language :: Python :: 3.10", "Programming Language :: Python :: 3.11", "Programming Language :: Python :: 3.12", + "Programming Language :: Python :: 3.13", ] +[project.urls] +Homepage = "https://github.com/nihanthnaidu007/Input_Guard" +Repository = "https://github.com/nihanthnaidu007/Input_Guard" +Changelog = "https://github.com/nihanthnaidu007/Input_Guard/blob/main/CHANGELOG.md" +Issues = "https://github.com/nihanthnaidu007/Input_Guard/issues" + [project.optional-dependencies] -dev = ["build>=1.2", "pytest>=8", "twine>=5"] +dev = [ + "build>=1.2", + "mypy>=1.8", + "pytest>=8", + "pytest-cov>=5", + "ruff>=0.5", + "twine>=5", +] + +[project.scripts] +inputguard = "inputguard.cli:main" [tool.setuptools.packages.find] include = ["inputguard*"] [tool.setuptools.package-data] inputguard = ["py.typed"] + +[tool.pytest.ini_options] +testpaths = ["tests"] + +[tool.ruff] +line-length = 100 + +[tool.ruff.lint] +select = ["E", "F", "W", "B"] +# E501: the codebase's prose-bearing table lines are exempt; the CI run +# treats ruff as the gate with this exact policy. +ignore = ["E501"] + +[tool.mypy] +strict = true +files = ["inputguard"] +# No python_version pin: modern mypy rejects 3.9 as a target, and the CI +# matrix already type-checks under each supported interpreter. diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..c8459ea --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,33 @@ +"""Shared test fixtures. + +The registry-isolation fixture lives here so every module that mutates the +module-level ``REGISTRY`` (the contract-guard suite, the policy suite) uses +one canonical snapshot/restore implementation instead of reaching into +private registry state ad hoc. +""" + +from __future__ import annotations + +import pytest + +from inputguard.registry import REGISTRY + + +@pytest.fixture +def registry_isolation(): + """Snapshot the registry around tests that mutate it. + + Registration is permanent by design (there is no unregister), so tests + that register probe rules/domains restore the snapshot afterwards to keep + the process-global registry unpolluted for the rest of the suite. + """ + rules_before = dict(REGISTRY._rules) + domains_before = dict(REGISTRY._domains) + origins_before = dict(REGISTRY._origins) + yield REGISTRY + REGISTRY._rules.clear() + REGISTRY._rules.update(rules_before) + REGISTRY._domains.clear() + REGISTRY._domains.update(domains_before) + REGISTRY._origins.clear() + REGISTRY._origins.update(origins_before) diff --git a/tests/test_benchmarks.py b/tests/test_benchmarks.py new file mode 100644 index 0000000..e300346 --- /dev/null +++ b/tests/test_benchmarks.py @@ -0,0 +1,165 @@ +"""Latency budgets for the documented analysis path (spec art_bTvdPdJS §8). + +Every budget below is derived from measured numbers on this codebase, not +guesses. Two measurement environments inform them: + +- **Sandbox, untraced** (Python 3.13, 10,000-character inputs, 60 samples + after warmup, ``time.perf_counter``): + + build-intent (vague, C1 path): p50 21.94 ms / p95 26.09 / p99 27.30 + debug-intent (with findings): p50 6.59 ms / p95 8.06 / p99 8.07 + ready path (score 100): p50 5.62 ms / p95 5.88 / p99 6.03 + 1.6 MB input under the cap: 17.1 ms (truncated=True) + +- **CI runners, coverage-traced** (the first benchmark push observed, run + 35277175054): debug p50 21.14 ms, ready p50 20.35 ms — coverage tracing + roughly triples per-call cost, and runner hardware is slower than the + sandbox. That observation is why these tests are excluded from the + coverage pass (the ``test`` job) and gate only in the dedicated + benchmark job, untraced. + +What the measured path includes — deliberately: + +- **C1 double-run (adversarial review art_hC18m78C):** the build path runs + ``InsufficientContextRule.check``, which re-executes the six built-in build + checks plus the intent-detail check inside itself because the Rule protocol + is stateless — a ~2x multiplier on the busiest path. The build input below + is intentionally vague so this rule is always part of the measured cost; + any latency budget for build-intent inputs must account for it. +- **Word-boundary matching (PR #6):** ``contains_any`` matches at word + boundaries instead of raw substring scans; its measured delta is part of + every number above (it replaced the v0.2 substring matcher everywhere). +while still catching the failure class this guards against — algorithmic +regressions (accidental O(n^2), a lost length cap) blow past these budgets +by an order of magnitude, as the v0.2 unbounded 3.8 s scan would. The hard +gate is p50 (a stable statistic under CI noise), with a p95 +sustained-regression guard at 2.5x the p50 budget. Single-sample maxima +are NOT asserted — CI runners are noisy and a one-off slow sample must not +fail a merge. p50/p99 are printed so the CI benchmark job records them per +spec ("p50/p99 recorded in CI"). +""" + +from __future__ import annotations + +import time + +from inputguard import InputGuard + +# 10k characters: the default Policy.max_chars cap — the documented worst +# case for a single analyze() call. +_CAP = 10_000 +_SAMPLES = 60 +_WARMUP = 5 + +# Measured p50 21.94 ms untraced (including the C1 double-run); CI-traced +# observation stayed under 40 ms. Budget ~3x -> 65 ms; p95 guard 2.5x. +BUILD_P50_BUDGET_MS = 65.0 + +# Measured p50 6.59 ms untraced; 21.14 ms coverage-traced on CI runners. +# Budget ~3.8x untraced -> 25 ms; p95 guard 2.5x. +DEBUG_P50_BUDGET_MS = 25.0 + +# Measured p50 5.62 ms untraced; 20.35 ms coverage-traced on CI runners. +# Budget ~4.4x untraced -> 25 ms; p95 guard 2.5x. +READY_P50_BUDGET_MS = 25.0 + + +def _pad(text: str, filler: str, n: int = _CAP) -> str: + """Extend ``text`` to exactly ``n`` characters with neutral filler prose. + + The filler carries no rule triggers (no error/build/feature vocabulary); + only the seed text decides which intent and findings fire. + """ + reps = (n - len(text)) // len(filler) + 1 + return (text + " " + (filler * reps))[:n] + + +# Build-intent, deliberately vague: InsufficientContextRule is in the build +# ruleset, so this path pays the C1 re-run of seven checks inside check(). +_BUILD_VAGUE = _pad( + "build me an app", + "please and then the thing should be there somehow ", +) + +# Debug-intent with findings: a trigger with unresolved satisfies. +_DEBUG_GAPS = _pad( + "my python code is throwing an error when the users log in, " + "the app is just wrong", + "the function receives the arguments and the error happens again " + "when running it ", +) + +# Ready path: a fully-specified debug request (eval corpus TN-DBG-01 shape) +# padded to the cap — every trigger-and-satisfy pair resolves, score 100. +_READY = _pad( + "getting a TypeError in my Python function, expected a list but got None, " + "the function should return an empty list instead of crashing, " + "stack trace shows the failure happens on the iteration step", + "the function receives the arguments described above and returns the " + "value it should return; the behavior matches the description given ", +) + + +def _measure(input_text: str) -> tuple[float, float, float]: + """Return (p50, p95, p99) in milliseconds over _SAMPLES analyzed calls.""" + guard = InputGuard() + for _ in range(_WARMUP): + guard.analyze(input_text) + samples: list[float] = [] + for _ in range(_SAMPLES): + start = time.perf_counter() + guard.analyze(input_text) + samples.append((time.perf_counter() - start) * 1000.0) + samples.sort() + p50 = samples[len(samples) // 2] + p95 = samples[int(len(samples) * 0.95)] + p99 = samples[min(len(samples) - 1, int(len(samples) * 0.99))] + return p50, p95, p99 + + +def test_build_intent_p50_within_budget() -> None: + """C1 path: vague 10k build input, InsufficientContextRule re-running the + build checks inside check() — the documented worst-case busy path.""" + p50, p95, p99 = _measure(_BUILD_VAGUE) + print( + f"\n[benchmark] build-intent 10k: p50={p50:.2f}ms p95={p95:.2f}ms " + f"p99={p99:.2f}ms (budget p50<={BUILD_P50_BUDGET_MS}ms)" + ) + assert p50 <= BUILD_P50_BUDGET_MS, f"build p50 {p50:.2f}ms > {BUILD_P50_BUDGET_MS}ms" + assert p95 <= BUILD_P50_BUDGET_MS * 2.5, f"build p95 {p95:.2f}ms > sustained guard" + + +def test_debug_intent_p50_within_budget() -> None: + p50, p95, p99 = _measure(_DEBUG_GAPS) + print( + f"\n[benchmark] debug-intent 10k: p50={p50:.2f}ms p95={p95:.2f}ms " + f"p99={p99:.2f}ms (budget p50<={DEBUG_P50_BUDGET_MS}ms)" + ) + assert p50 <= DEBUG_P50_BUDGET_MS, f"debug p50 {p50:.2f}ms > {DEBUG_P50_BUDGET_MS}ms" + assert p95 <= DEBUG_P50_BUDGET_MS * 2.5, f"debug p95 {p95:.2f}ms > sustained guard" + + +def test_ready_path_p50_within_budget() -> None: + p50, p95, p99 = _measure(_READY) + print( + f"\n[benchmark] ready-path 10k: p50={p50:.2f}ms p95={p95:.2f}ms " + f"p99={p99:.2f}ms (budget p50<={READY_P50_BUDGET_MS}ms)" + ) + assert p50 <= READY_P50_BUDGET_MS, f"ready p50 {p50:.2f}ms > {READY_P50_BUDGET_MS}ms" + assert p95 <= READY_P50_BUDGET_MS * 2.5, f"ready p95 {p95:.2f}ms > sustained guard" + + +def test_unbounded_input_is_bounded_by_cap() -> None: + """The v0.2 failure mode (1.6 MB input taking ~3.8 s of linear scans) must + stay dead: the Policy cap bounds the scan and the result says so.""" + big = _pad(_BUILD_VAGUE, "more of the same neutral prose ", n=1_600_000) + assert len(big) == 1_600_000 + guard = InputGuard() + start = time.perf_counter() + result = guard.analyze(big) + elapsed_ms = (time.perf_counter() - start) * 1000.0 + print(f"\n[benchmark] 1.6MB input with cap: {elapsed_ms:.1f}ms, truncated={result.truncated}") + assert result.truncated is True + # 1s ceiling: 100x headroom over the measured capped path (~10 ms) while + # still catching any regression to the v0.2 unbounded 3.8 s behavior. + assert elapsed_ms < 1000.0, f"1.6MB input took {elapsed_ms:.1f}ms — the cap is not bounding the scan" diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..91824c8 --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,83 @@ +"""Tests for the ``inputguard`` CLI (spec art_bTvdPdJS §5). + +Covers the documented exit-code contract (0 = ok / 1 = below the +``--min-score`` floor / 2 = usage error), the text rendering shape, and that +``--format json`` emits exactly the result's ``to_dict()`` contract — the +CLI adds no fields and renames none. +""" + +from __future__ import annotations + +import io +import json + +import pytest + +from inputguard import InputGuard +from inputguard.cli import EXIT_BELOW_FLOOR, EXIT_OK, EXIT_USAGE, main + +VAGUE = "make this faster" # optimization intent, needs clarification +READY = ( + "getting a TypeError in my Python function, expected a list but got None" +) # eval corpus TN-DBG-01 shape: score 100 + + +def test_text_output_contains_verdict_lines(capsys: pytest.CaptureFixture[str]) -> None: + assert main(["analyze", VAGUE]) == EXIT_OK + out = capsys.readouterr().out + assert "status: needs_clarification" in out + assert "clarity: " in out + assert "intent: optimization" in out + assert "domain: coding" in out + assert "missing: " in out + assert "ask: " in out # the follow-up questions surface on the CLI too + + +def test_json_output_is_exactly_to_dict(capsys: pytest.CaptureFixture[str]) -> None: + assert main(["analyze", VAGUE, "--format", "json"]) == EXIT_OK + parsed = json.loads(capsys.readouterr().out) + expected = InputGuard().analyze(VAGUE).to_dict() + assert parsed == expected + + +def test_min_score_below_floor_exits_one() -> None: + assert main(["analyze", VAGUE, "--min-score", "85"]) == EXIT_BELOW_FLOOR + + +def test_min_score_at_or_above_floor_exits_zero() -> None: + assert main(["analyze", READY, "--min-score", "85"]) == EXIT_OK + + +def test_missing_text_is_a_usage_error() -> None: + with pytest.raises(SystemExit) as exc_info: + main(["analyze"]) + assert exc_info.value.code == EXIT_USAGE + + +def test_unknown_domain_is_a_usage_error_not_a_traceback( + capsys: pytest.CaptureFixture[str], +) -> None: + assert main(["analyze", VAGUE, "--domain", "legal"]) == EXIT_USAGE + err = capsys.readouterr().err + assert "inputguard:" in err + assert "legal" in err + + +def test_stdin_input(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr("sys.stdin", io.StringIO(VAGUE)) + assert main(["analyze", "--stdin"]) == EXIT_OK + + +def test_strict_mode_flag_is_accepted() -> None: + assert main(["analyze", READY, "--mode", "strict", "--min-score", "85"]) == EXIT_OK + + +def test_degraded_note_appears_in_text_output(capsys: pytest.CaptureFixture[str]) -> None: + # A Chinese build request (spec probe P2): the language probe must mark + # the result degraded, and the CLI must show that note instead of a + # silent ready. + degraded_input = "建造一个用户登录应用" + assert main(["analyze", degraded_input]) == EXIT_OK + out = capsys.readouterr().out + assert "note: " in out + assert "status: ready" not in out diff --git a/tests/test_followups.py b/tests/test_followups.py index eaffd2b..1d7d6c1 100644 --- a/tests/test_followups.py +++ b/tests/test_followups.py @@ -73,7 +73,7 @@ def test_every_template_renders_with_unknown_slot_failing_loudly(): # Rendering every template against empty slots exercises the render path: # a typo'd slot name raises (loud table bug); a known-but-unfilled slot # skips the template (documented behavior); plain templates pass through. - for gap, questions in _FOLLOW_UP_QUESTIONS.items(): + for _gap, questions in _FOLLOW_UP_QUESTIONS.items(): for question in questions: rendered = _render(question, slots={}) if "{" in question: @@ -368,8 +368,13 @@ def run(text: str) -> None: small_ms = _wall_ms(run, "a" * 1_000) large_ms = _wall_ms(run, "a" * 10_000) assert large_ms < 40 * max(small_ms, 0.05) - # And both stay fast in absolute terms — 1.3 s at 10 K was the old number. - assert large_ms < 50.0 + # And both stay fast in absolute terms — 1.3 s at 10 K was the old + # number. The bound is a single wall-clock sample, so it carries CI + # headroom: coverage-traced runs on slow runners measured up to ~51 ms + # (a bare 50 ms bound flaked there, run 35277834661), while the + # quadratic regression this guards against is 1.3 s — 5x+ margin even + # at 250 ms. + assert large_ms < 250.0 def test_dataset_file_extraction_dotted_filler_stays_linear_from_1k_to_10k(): @@ -379,7 +384,9 @@ def run(text: str) -> None: small_ms = _wall_ms(run, "a." * 500) large_ms = _wall_ms(run, "a." * 5_000) assert large_ms < 40 * max(small_ms, 0.05) - assert large_ms < 50.0 + # Same CI headroom rationale as above: single sample, traced runners + # measured ~51 ms, regression is 1.3 s. + assert large_ms < 250.0 def test_dataset_file_extraction_is_case_insensitive_as_before(): diff --git a/tests/test_policy.py b/tests/test_policy.py index 0fc1622..23db685 100644 --- a/tests/test_policy.py +++ b/tests/test_policy.py @@ -17,7 +17,8 @@ from inputguard.policy import SEVERITIES from inputguard.registry import KNOWN_SEVERITIES, register_rule from inputguard.scorer import require_known_severity -from test_registry import _test_rule, registry_isolation # noqa: F401 — pytest fixture +from test_registry import _test_rule # noqa: F401 — shared rule builder +# registry_isolation resolves as a pytest fixture from tests/conftest.py. # Observed v0.2 / release-branch behavior (runtime probes), reused as fixtures. VAGUE_INPUT = "do something now" # build intent, one high finding: insufficient_context diff --git a/tests/test_readme_examples.py b/tests/test_readme_examples.py new file mode 100644 index 0000000..2d813bd --- /dev/null +++ b/tests/test_readme_examples.py @@ -0,0 +1,164 @@ +"""README anti-drift gate: every example in README.md is executed. + +The v0.2 README shipped ``to_dict()`` examples that drifted from the code +within two releases (survey art_CnghyDxp §5.6) — the first docs a new user +read were wrong. This module makes that failure mode a test failure: + +1. Every ```python fenced block in README.md runs, in document order, in + one fresh subprocess. The README's code blocks carry real ``assert`` + statements for their documented values, so a stale value fails here. + The subprocess also isolates the extension examples' process-global + registry registrations from the modules this suite runs afterwards. +2. The embedded ```json block must equal the analyzer's live ``to_dict()`` + output for the same input — regenerated docs, never hand-copied. +3. The rule tables, gap-name lists, export list, and CLI output block must + match the live package: rule ids, severities, and gap strings per + intent, straight from the registry. +""" + +from __future__ import annotations + +import json +import re +import subprocess +import sys +from pathlib import Path +from typing import List + +import inputguard +from inputguard import REGISTRY + +README_PATH = Path(__file__).resolve().parent.parent / "README.md" + +_FENCE = re.compile(r"```([a-z]*)\n(.*?)```", re.DOTALL) + +# "What gets checked" section header -> the intent whose rules it documents. +_SECTION_INTENTS = [ + ("### Build inputs", "build"), + ("### Debug inputs", "debug"), + ("### Optimization inputs", "optimization"), + ("### Explanation inputs", "explanation"), + ("### Feature inputs", "feature"), + ("### Writing inputs", "compose"), + ("### Data-analysis inputs", "analysis"), +] + +# Bold label in the gap-vocabulary bullets -> intent id. +_GAP_LABELS = { + "Build": "build", + "Debug": "debug", + "Optimization": "optimization", + "Explanation": "explanation", + "Feature": "feature", + "Compose": "compose", + "Analysis": "analysis", +} + + +def _blocks(lang: str) -> List[str]: + """The README's fenced blocks in one language, in document order.""" + text = README_PATH.read_text(encoding="utf-8") + found = [match.group(2) for match in _FENCE.finditer(text) if match.group(1) == lang] + assert found, f"no {lang!r} blocks found in README.md — extraction rotted" + return found + + +def test_python_blocks_run_in_document_order() -> None: + blocks = _blocks("python") + # If this count drops, block extraction rotted — fix the extraction, not the docs. + assert len(blocks) >= 10 + script = "\n\n".join(blocks) + proc = subprocess.run( + [sys.executable, "-c", script], + capture_output=True, + text=True, + timeout=120, + ) + assert proc.returncode == 0, ( + "A README example failed — README and code have drifted.\n" + f"--- stdout ---\n{proc.stdout}\n--- stderr ---\n{proc.stderr}" + ) + + +def test_to_dict_json_block_matches_live_output() -> None: + blocks = _blocks("json") + assert len(blocks) == 1, "README should embed exactly one to_dict() JSON example" + documented = json.loads(blocks[0]) + live = inputguard.InputGuard().analyze("fix my code").to_dict() + assert documented == live + + +def test_public_exports_block_matches_package() -> None: + import_blocks = [b for b in _blocks("python") if "from inputguard import (" in b] + assert len(import_blocks) == 1, "README should show the public exports exactly once" + names = re.findall(r"^\s{4}(\w+),?$", import_blocks[0], re.MULTILINE) + assert set(names) == set(inputguard.__all__) + for name in names: + assert hasattr(inputguard, name), f"README export {name!r} does not exist" + + +def test_rule_tables_match_registry() -> None: + text = README_PATH.read_text(encoding="utf-8") + lines = text.splitlines() + for header, intent in _SECTION_INTENTS: + at = lines.index(header) # a vanished section raises — that is the failure + documented = {} + for line in lines[at + 1 :]: + if line.startswith("#"): + break + if line.startswith("| `"): + cells = [cell.strip() for cell in line.strip().strip("|").split("|")] + documented[cells[0].strip("`")] = cells[-1] + expected = {r.id: r.severity for r in REGISTRY.rules() if r.domain == intent} + assert documented == expected, ( + f"README rule table for {intent!r} drifted from the registry" + ) + registry_intents = {r.domain for r in REGISTRY.rules()} + assert registry_intents == {intent for _, intent in _SECTION_INTENTS}, ( + "a registry intent has no README rule-table section" + ) + + +def test_gap_name_lists_match_registry() -> None: + text = README_PATH.read_text(encoding="utf-8") + section = text.split("## The gap vocabulary", 1)[1].split("\n## ", 1)[0] + documented = {} + for line in section.splitlines(): + match = re.match(r"- \*\*(\w+):?\*\*\s*(.*)", line) + if not match: + continue + intent = _GAP_LABELS[match.group(1)] + documented[intent] = sorted(re.findall(r"`([^`]+)`", match.group(2))) + assert set(documented) == set(_GAP_LABELS.values()), "a gap list is missing from README" + for intent, gaps in documented.items(): + expected = sorted( + {r.gap for r in REGISTRY.rules() if r.domain == intent and r.gap is not None} + ) + assert gaps == expected, f"README gap list for {intent!r} drifted from the registry" + + +def test_cli_examples_behave_as_documented() -> None: + proc = subprocess.run( + [sys.executable, "-m", "inputguard.cli", "analyze", "make this faster", "--mode", "strict"], + capture_output=True, + text=True, + timeout=60, + ) + assert proc.returncode == 0 + text_blocks = _blocks("text") + assert len(text_blocks) == 1, "README should embed exactly one CLI output block" + assert proc.stdout.strip() == text_blocks[0].strip(), ( + "README CLI output drifted from the real CLI" + ) + + gated = subprocess.run( + [sys.executable, "-m", "inputguard.cli", "analyze", "build a REST API", "--min-score", "85"], + capture_output=True, + text=True, + timeout=60, + ) + assert gated.returncode == 1, "--min-score below the floor must exit 1" + + bash_blocks = _blocks("bash") + assert any("pip install inputguard" in block for block in bash_blocks) + assert any("inputguard analyze" in block for block in bash_blocks) diff --git a/tests/test_recipes.py b/tests/test_recipes.py new file mode 100644 index 0000000..8776fdb --- /dev/null +++ b/tests/test_recipes.py @@ -0,0 +1,88 @@ +"""docs/recipes example gate: the recipe code blocks execute. + +Each recipe is split into a framework-free block (pure inputguard, runs +verbatim) and a wiring block whose only framework usage is the runnable/ +completion idiom. The tests run every block in order inside a fresh +subprocess, with minimal stubs for the framework imports — proving the +inputguard-side contract the recipe relies on, without adding framework +dependencies to this repository. +""" + +from __future__ import annotations + +import re +import subprocess +import sys +from pathlib import Path + +RECIPES_DIR = Path(__file__).resolve().parent.parent / "docs" / "recipes" +_FENCE = re.compile(r"```python\n(.*?)```", re.DOTALL) + +# Minimal langchain-core runnable protocol: RunnableLambda wraps a callable, +# `|` composes pipelines left-to-right, .invoke() runs the chain. +_LANGCHAIN_STUB = """ +import types as _types + +class _Pipeline: + def __init__(self, steps): self._steps = steps + def __or__(self, other): return _Pipeline(self._steps + [other]) + def invoke(self, value): + for step in self._steps: + value = step.invoke(value) + return value + +class RunnableLambda: + def __init__(self, fn): self._fn = fn + def invoke(self, value): return self._fn(value) + def __or__(self, other): return _Pipeline([self, other]) + +_langchain_core = _types.ModuleType("langchain_core") +_runnables = _types.ModuleType("langchain_core.runnables") +_runnables.RunnableLambda = RunnableLambda +_langchain_core.runnables = _runnables +sys.modules["langchain_core"] = _langchain_core +sys.modules["langchain_core.runnables"] = _runnables +""" + +# The clarify path must return before the completion call — the stub raises +# if the recipe wiring ever reaches the model on the demo input. +_LITELLM_STUB = """ +import types as _types + +def _completion(*args, **kwargs): + raise AssertionError("litellm.completion called on the clarify path") + +litellm = _types.ModuleType("litellm") +litellm.completion = _completion +sys.modules["litellm"] = litellm +""" + +_RUNNERS = { + "langchain.md": _LANGCHAIN_STUB, + "litellm.md": _LITELLM_STUB, +} + + +def test_both_recipe_pages_exist_and_use_to_dict() -> None: + for name in _RUNNERS: + page = RECIPES_DIR / name + assert page.is_file(), f"missing recipe page docs/recipes/{name}" + assert "to_dict()" in page.read_text(encoding="utf-8") + + +def test_recipe_blocks_execute_in_document_order() -> None: + for name, stub_setup in _RUNNERS.items(): + page = RECIPES_DIR / name + blocks = _FENCE.findall(page.read_text(encoding="utf-8")) + assert len(blocks) >= 2, f"{name}: expected preflight + wiring blocks" + script = "import sys\n" + stub_setup + "\n" + "\n\n".join(blocks) + proc = subprocess.run( + [sys.executable, "-c", script], + capture_output=True, + text=True, + timeout=120, + ) + assert proc.returncode == 0, ( + f"A docs/recipes/{name} example failed.\n" + f"--- stdout ---\n{proc.stdout}\n--- stderr ---\n{proc.stderr}" + ) diff --git a/tests/test_registry_contract_guards.py b/tests/test_registry_contract_guards.py new file mode 100644 index 0000000..1ec0cac --- /dev/null +++ b/tests/test_registry_contract_guards.py @@ -0,0 +1,268 @@ +"""Registry contract-guard tests (adversarial review art_hC18m78C). + +The three tests the review found missing, as a dedicated CI-enforced suite: + +1. Duplicate rule ids within one ``register_domain`` call raise ``ValueError`` + (review C3/probe P2 — the high-severity second rule used to vanish + silently) and nothing mutates on rejection. +2. A rule whose ``check`` signature cannot be called as ``check(self, text)`` + is rejected *at registration* with an actionable message (review B1/N3/ + probe P1 — a mismatched rule used to pass registration and detonate mid- + ``analyze()`` with a ``TypeError`` on arbitrary user input). +3. A rule that raises inside ``check()`` surfaces as a ``RuntimeError`` + carrying the rule's id and registration origin, with the original + exception chained (review C2/P4 — exceptions used to escape raw, with no + attribution and no recovery path). + +These complement the remediation tests in ``test_registry.py`` (which covers +the ``register_rule`` path); this module is the contract-guards gate the CI +matrix runs on every push, and also pins the ``register_domain`` road. +""" + +from __future__ import annotations + +import pytest + +from inputguard import InputGuard, RuleFinding +from inputguard.registry import REGISTRY, register_domain, register_rule + + +# --- 1 · same-call duplicate rule ids raise, nothing mutates ---------------- + + +class _ProbeRule: + """Probe rule with a configurable id, firing on the word ``widget``.""" + + domain = "probe thing" + severity = "high" + gap = "probe gap" + + def __init__(self, rule_id: str) -> None: + self.id = rule_id + + def check(self, text: str): + if "widget" in text: + return RuleFinding( + code=self.id, + message="probe rule fired", + severity=self.severity, + gap=self.gap, + ) + return None + + +_PROBE_SIGNALS = {"probe thing": ("widget",), "probe review": ()} + + +def test_same_call_duplicate_rule_ids_raise_and_mutate_nothing(registry_isolation): + """Review C3: duplicate ids inside one register_domain call must raise. + + The original hole: the pre-check compared only against *already + registered* rules, so a same-call duplicate was silently skipped in the + mutation loop — the second rule disappeared without an error. The error + must name the duplicated id, and a rejected batch must leave the registry + untouched ("nothing mutates unless every check passes"). + """ + with pytest.raises(ValueError, match="register_domain call.*'dup_x'") as exc_info: + register_domain( + "probe_dup", + _PROBE_SIGNALS, + rules=[_ProbeRule("dup_x"), _ProbeRule("dup_x")], + ) + + # The message names the offending id so the author can fix it in one look. + assert "dup_x" in str(exc_info.value) + + # No partial mutation: neither the domain nor either rule registered. + assert "probe_dup" not in REGISTRY.domain_names() + assert "dup_x" not in REGISTRY.rule_ids() + + # A corrected call through the same path works — the guard blocks + # duplicates, not the batch API itself. + register_domain("probe_dup_fixed", _PROBE_SIGNALS, rules=[_ProbeRule("solo_rule")]) + result = InputGuard().analyze("widget please", domain="probe_dup_fixed") + assert "solo_rule" in [f.code for f in result.findings] + + +def test_duplicate_hidden_among_distinct_ids_raises(registry_isolation): + """A duplicate buried in a batch of otherwise-distinct ids still raises, + and the message names exactly the duplicated id.""" + with pytest.raises(ValueError, match="'twin_b'") as exc_info: + register_domain( + "probe_dup_mixed", + _PROBE_SIGNALS, + rules=[ + _ProbeRule("unique_a"), + _ProbeRule("twin_b"), + _ProbeRule("twin_b"), + _ProbeRule("unique_c"), + ], + ) + # The distinct ids are not blamed. + assert "unique_a" not in str(exc_info.value) + assert "unique_c" not in str(exc_info.value) + assert "probe_dup_mixed" not in REGISTRY.domain_names() + assert "unique_a" not in REGISTRY.rule_ids() + assert "twin_b" not in REGISTRY.rule_ids() + + +# --- 2 · wrong check() signature is a registration error -------------------- + + +def test_extra_parameter_check_rejected_at_registration(registry_isolation): + """Review B1: the retired check(self, text, intent) shape must be a + registration error, not a mid-analyze TypeError. The message is + actionable: it names the rule, the one-positional-argument dispatch, and + the expected check(self, text) signature.""" + probe_signals = {"probe sig": ("gadget",), "probe sig review": ()} + + class ExtraArgRule: + id = "probe_extra_arg" + domain = "probe sig" + severity = "low" + gap = None + + def check(self, text, intent): # noqa: ARG001 — the retired shape + return None + + with pytest.raises(TypeError) as exc_info: + register_domain("probe_sig", probe_signals, rules=[ExtraArgRule()]) + + message = str(exc_info.value) + assert "probe_extra_arg" in message + assert "check(self, text)" in message + assert "probe_extra_arg" not in REGISTRY.rule_ids() + assert "probe_sig" not in REGISTRY.domain_names() + + +def test_zero_parameter_check_rejected_at_registration(registry_isolation): + """check() with no parameters cannot bind the normalized text — reject at + registration.""" + + class NoArgRule: + id = "probe_noarg" + domain = "debug" + severity = "low" + gap = None + + def check(self): + return None + + with pytest.raises(TypeError, match="probe_noarg") as exc_info: + register_rule(NoArgRule()) + assert "check(self, text)" in str(exc_info.value) + assert "probe_noarg" not in REGISTRY.rule_ids() + + +def test_keyword_only_check_rejected_at_registration(registry_isolation): + """check(self, *, text) cannot be called positionally — the analyzer + dispatches rule.check(normalized), so keyword-only rejects too.""" + + class KeywordOnlyRule: + id = "probe_kwonly" + domain = "debug" + severity = "low" + gap = None + + def check(self, *, text): + return None + + with pytest.raises(TypeError, match="probe_kwonly"): + register_rule(KeywordOnlyRule()) + assert "probe_kwonly" not in REGISTRY.rule_ids() + + +def test_defaulted_check_is_accepted_and_fires(registry_isolation): + """check(self, text="...") binds one positional argument, so it satisfies + the dispatch contract — the gate rejects what cannot be called, not + signatures it merely dislikes.""" + + class DefaultedRule: + id = "probe_defaulted" + domain = "debug" + severity = "low" + gap = None + + def check(self, text=""): + if "fix the bug" in text: + return RuleFinding( + code=self.id, + message="Defaulted-signature rule fired.", + severity=self.severity, + gap=self.gap, + ) + return None + + register_rule(DefaultedRule()) + result = InputGuard().analyze("fix the bug in my app") + assert "probe_defaulted" in [f.code for f in result.findings] + + +def test_spec_signature_rule_is_accepted_and_fires(registry_isolation): + """Positive control for the arity gate: check(self, text) — the pinned + spec's signature (art_bTvdPdJS §1) — registers cleanly and the analyzer + dispatches it. The gate rejects deviations, never the documented shape.""" + + class SpecRule: + id = "probe_spec_signature" + domain = "debug" + severity = "medium" + gap = "error description" + + def check(self, text: str): + if "fix the bug" in text: + return RuleFinding( + code=self.id, + message="Spec-signature rule fired.", + severity=self.severity, + gap=self.gap, + ) + return None + + register_rule(SpecRule()) + result = InputGuard().analyze("fix the bug in my app") + assert "probe_spec_signature" in [f.code for f in result.findings] + + +# --- 3 · dispatch exceptions carry rule attribution -------------------------- + + +def test_raising_rule_surfaces_id_origin_and_cause_via_register_domain( + registry_isolation, +): + """Review C2/P4: an exception inside check() aborts analyze() with the + rule's id, its registration origin, and the original traceback chained. + Tested through the register_domain path — test_registry.py covers the + register_rule path — so both registration roads get the same loud + attribution.""" + + class ExplodingRule: + id = "probe_boom_domain_rule" + domain = "probe boom thing" + severity = "high" + gap = None + + def check(self, text: str): + raise ZeroDivisionError("division by zero in probe rule") + + register_domain( + "probe_boom", + {"probe boom thing": ("widget",), "probe boom review": ()}, + rules=[ExplodingRule()], + ) + + with pytest.raises(RuntimeError) as exc_info: + InputGuard().analyze("widget please", domain="probe_boom") + + message = str(exc_info.value) + # Rule id, exception type, and registration origin all named. + assert "probe_boom_domain_rule" in message + assert "ZeroDivisionError" in message + assert "registered at" in message + # The original exception is chained, not swallowed. + assert isinstance(exc_info.value.__cause__, ZeroDivisionError) + assert "division by zero in probe rule" in str(exc_info.value.__cause__) + # The origin names the file that registered the rule — this test module, + # not inputguard internals (registration happened via register_domain + # above, so the recorded call site is the register_domain line here). + assert __file__ in message