Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
138 changes: 133 additions & 5 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,20 +2,148 @@ name: CI

on:
push:
branches: [main]
branches: [main, "release/**"]
pull_request:

permissions:
contents: read

jobs:
# The quality gate: lint, tests with a 90% branch-coverage floor, and
# strict typing across the full supported Python matrix (spec §8).
test:
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
# 3.9 is the package floor (requires-python), 3.13 is current.
python-version: ["3.9", "3.13"]
python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: ${{ matrix.python-version }}
- run: pip install -e ".[dev]"
- run: pytest -q
- name: Install with dev tooling
run: |
python -m pip install --upgrade pip
pip install -e ".[dev]"
- name: Ruff
run: ruff check inputguard tests eval
- name: Run tests with 90% branch-coverage floor
# Latency asserts are excluded here and gate only in the dedicated
# benchmark job: coverage tracing roughly triples per-call cost
# (observed on run 35277175054 — debug p50 21.14 ms vs 6.59 ms
# untraced), so gating latency under coverage double-gates the same
# budgets under different conditions.
run: |
pytest -q --cov=inputguard --cov-branch --cov-fail-under=90 --strict \
--ignore=tests/test_benchmarks.py
- name: Strict typing (checks the shipped py.typed)
run: mypy --strict inputguard

# Latency budgets are gates, not notes: a regression here is a production
# failure mode (the v0.2 unbounded 3.8 s scan). Measured p50/p99 print to
# the job log; budgets and their derivation live in tests/test_benchmarks.py.
benchmark:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: "3.13"
- name: Install with dev tooling
run: |
python -m pip install --upgrade pip
pip install -e ".[dev]"
- name: Latency budgets (p50/p99 recorded)
run: pytest -v -s tests/test_benchmarks.py

# The zero-dependency promise, verified against the built artifact — not
# the source tree: wheel METADATA carries no Requires-Dist, a bare venv
# install pulls no third-party package, and the public API + CLI work from
# the installed wheel alone. The smoke steps run from a scratch directory
# so the repo's local `inputguard/` package can never shadow the wheel.
wheel-zero-dep:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: "3.13"
- name: Build wheel
run: |
python -m pip install --upgrade pip build
python -m build --wheel
- name: Wheel metadata declares no unconditional runtime dependencies
run: |
python - <<'PY'
import glob
import zipfile

wheel = glob.glob("dist/*.whl")[0]
names = zipfile.ZipFile(wheel).namelist()
meta_name = next(n for n in names if n.endswith(".dist-info/METADATA"))
meta = zipfile.ZipFile(wheel).read(meta_name).decode()
requires = [
line
for line in meta.splitlines()
if line.startswith("Requires-Dist:")
]
unconditional = [
line for line in requires if 'extra == "' not in line
]
assert not unconditional, (
"wheel METADATA declares unconditional runtime dependencies — "
f"the zero-dep promise is broken: {unconditional}"
)
extra_gated = len(requires)
print(
f"{wheel}: 0 unconditional runtime dependencies "
f"({extra_gated} extra-gated dev-only entries)"
)
PY
- name: Install wheel into a bare venv
run: |
python -m venv bare
./bare/bin/pip install --upgrade pip
./bare/bin/pip install dist/*.whl
- name: Public API works from the installed wheel
run: |
cd "$(mktemp -d)"
"$GITHUB_WORKSPACE/bare/bin/python" - <<'PY'
import json

import inputguard

assert inputguard.__version__
guard = inputguard.InputGuard()
result = guard.analyze("make this faster", domain="coding")
assert isinstance(result.clarity_score, int)
assert 0 <= result.clarity_score <= 100
payload = result.to_dict()
json.dumps(payload)
print(f"bare-venv analyze OK: {result.status} score={result.clarity_score}")
PY
- name: Bare venv contains no third-party packages
run: |
third_party=$(./bare/bin/pip list --format=freeze \
| cut -d= -f1 | tr '[:upper:]' '[:lower:]' \
| grep -v -E '^(inputguard|pip|setuptools)$' || true)
if [ -n "$third_party" ]; then
echo "Third-party packages present after wheel install: $third_party"
exit 1
fi
echo "bare venv holds only inputguard (+ venv tooling)"
- name: CLI works from the installed wheel
run: |
cd "$(mktemp -d)"
INPUTGUARD="$GITHUB_WORKSPACE/bare/bin/inputguard"
"$INPUTGUARD" analyze "make this faster" --format json
set +e
"$INPUTGUARD" analyze "make this faster" --min-score 85
code=$?
set -e
if [ "$code" -ne 1 ]; then
echo "expected exit 1 below the min-score floor, got $code"
exit 1
fi
echo "CLI exit codes OK (0 on analysis, 1 below the min-score floor)"
17 changes: 17 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,23 @@ in `docs/false-positive-benchmark.md`.
carrying a run of 3+ consecutive uncovered-script letters (mixed
English+Han). English prompts with loanwords, URLs, or name collisions
are pinned unchanged by tests.
- CI as production gates (`.github/workflows/ci.yml`): a Python 3.9–3.13
matrix running ruff, pytest with a 90% branch-coverage floor
(`--cov-branch --cov-fail-under=90 --strict`, latency asserts excluded
from the traced pass), and mypy `--strict` (the shipped `py.typed` is
now actually checked); a dedicated untraced latency-benchmark job with
budgets measured on this codebase (10k-char build-intent p50 ≤ 65 ms
including the catch-all re-run path, debug and ready paths ≤ 25 ms; a
1.6 MB input bounded by the policy cap); and a zero-dependency wheel
gate that builds the wheel, installs it into a bare venv, and asserts
no third-party package is present.
- `inputguard` CLI (`inputguard analyze`) via a console-script entry
point: argparse, `--format json` emitting the same `to_dict()`
contract, and CI-friendly exit codes (0 ok, 1 below `--min-score`,
2 usage error).
- Packaging metadata: 3.13 classifier, Development Status → 4 - Beta,
project URLs, and the explicit dev extras / tool config
(`[tool.ruff]`, `[tool.mypy]`, `[tool.pytest.ini_options]`).

### Changed
- All term matching now happens at word boundaries (`#6`) — detector and
Expand Down
134 changes: 134 additions & 0 deletions inputguard/cli.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,134 @@
"""The ``inputguard`` command-line interface (spec art_bTvdPdJS §5).

A zero-dependency argparse front end over the same ``InputGuard.analyze``
pipeline the library exposes. JSON output is the result's own ``to_dict()``
contract — the CLI adds no fields and renames none.

Exit codes (the CI/commit-hook contract):

- ``0`` — analysis completed; with ``--min-score N``, the score is at or
above the floor.
- ``1`` — analysis completed but the clarity score is below the
``--min-score`` floor. Use with ``--min-score`` to gate commits/PRs.
- ``2`` — usage error: unknown flags, missing text, or an invalid value
(unknown domain, empty input) reported by the pipeline.

Examples::

inputguard analyze "make this faster"
inputguard analyze "make this faster" --mode strict
inputguard analyze "$(cat prompt.txt)" --format json --min-score 85
cat prompt.txt | inputguard analyze --stdin --min-score 85
"""

from __future__ import annotations

import argparse
import json
import sys
from typing import Optional, Sequence

from inputguard import AnalysisResult, InputGuard

EXIT_OK = 0
EXIT_BELOW_FLOOR = 1
EXIT_USAGE = 2


def _build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
prog="inputguard",
description="Pre-flight LLM inputs for clarity before inference.",
)
subparsers = parser.add_subparsers(dest="command", required=True)
analyze = subparsers.add_parser(
"analyze",
help="Analyze input text and report its clarity verdict.",
)
analyze.add_argument(
"text",
nargs="?",
help="The input text to analyze; omit when --stdin is given.",
)
analyze.add_argument(
"--stdin",
action="store_true",
help="Read the input text from stdin instead of the TEXT argument.",
)
analyze.add_argument(
"--domain",
default="coding",
help="Registered domain to analyze against (default: coding).",
)
analyze.add_argument(
"--mode",
choices=("warning", "strict"),
default="warning",
help="Status banding mode (default: warning).",
)
analyze.add_argument(
"--format",
choices=("text", "json"),
default="text",
help="Output format (default: text). json emits the to_dict() contract.",
)
analyze.add_argument(
"--min-score",
type=int,
default=None,
metavar="N",
help=(
"Exit 1 when the clarity score is below N — a ready-made "
"commit-hook / CI gate."
),
)
return parser


def _render_text(result: AnalysisResult, domain: str) -> str:
"""Human-readable verdict, one fact per line (spec §5 example shape)."""
lines = [
f"status: {result.status}",
f"clarity: {result.clarity_score}/100",
f"intent: {result.detected_intent}",
f"domain: {domain}",
]
if result.gaps:
lines.append(f"missing: {', '.join(result.gaps)}")
for question in result.follow_ups:
lines.append(f"ask: {question}")
if result.degradation_note is not None:
lines.append(f"note: {result.degradation_note}")
return "\n".join(lines)


def main(argv: Optional[Sequence[str]] = None) -> int:
"""Run one analysis; returns the process exit code (see module docstring)."""
parser = _build_parser()
args = parser.parse_args(argv)

text = sys.stdin.read() if args.stdin else args.text
if not text:
parser.error("provide the text to analyze as TEXT, or pass --stdin")

guard = InputGuard(mode=args.mode)
try:
result = guard.analyze(text, domain=args.domain)
except ValueError as exc:
# Invalid domain / empty-after-normalization input: a usage error,
# not a crash — the same ValueError contract the library documents.
print(f"inputguard: {exc}", file=sys.stderr)
return EXIT_USAGE

if args.format == "json":
print(json.dumps(result.to_dict(), indent=2))
else:
print(_render_text(result, args.domain))

if args.min_score is not None and result.clarity_score < args.min_score:
return EXIT_BELOW_FLOOR
return EXIT_OK


if __name__ == "__main__": # pragma: no cover — manual invocation convenience
sys.exit(main())
2 changes: 1 addition & 1 deletion inputguard/detector.py
Original file line number Diff line number Diff line change
Expand Up @@ -70,7 +70,7 @@ def normalize(text: str) -> str:
return re.sub(r"\s+", " ", text.strip().lower())


def _contains_any(text: str, terms) -> bool:
def _contains_any(text: str, terms: Iterable[str]) -> bool:
# v0.3: word-boundary matching shared with the rule modules — "fixture"
# is no longer read as the debug signal "fix" (probe P1).
return contains_any(text, terms)
Expand Down
6 changes: 3 additions & 3 deletions inputguard/language.py
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,7 @@
import re
import unicodedata
from dataclasses import dataclass
from typing import Dict, Optional
from typing import Dict, FrozenSet, Optional, Set

__all__ = [
"COVERAGE_FULL",
Expand Down Expand Up @@ -228,7 +228,7 @@
# of 3+ letters is a word or clause the English heuristics cannot read.
_UNCOVERED_BLOCK_DEGRADES_AT = 3

_LATIN_FUNCTION_WORDS: Dict[str, frozenset] = {
_LATIN_FUNCTION_WORDS: Dict[str, FrozenSet[str]] = {
"en": frozenset({
"a", "an", "the", "and", "or", "but", "if", "then", "of", "to", "in",
"on", "at", "by", "for", "with", "from", "into", "over", "under",
Expand Down Expand Up @@ -357,7 +357,7 @@ def _latin_language_of(sample: str) -> Optional[str]:
Ties between firing languages resolve alphabetically, like the script
histogram's dominant-script tie-break.
"""
distinct: Dict[str, set] = {}
distinct: Dict[str, Set[str]] = {}
for raw in sample.translate(_APOSTROPHES).lower().split():
token = _EDGE_TRIM.sub("", raw)
if not token:
Expand Down
8 changes: 4 additions & 4 deletions inputguard/recommender.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
from typing import Dict, List


_RECOMMENDATIONS: Dict[str, dict] = {
_RECOMMENDATIONS: Dict[str, Dict[str, str]] = {
"programming language": {
"gap": "programming language",
"what_is_missing": "You haven't told it which programming language or technology to use.",
Expand Down Expand Up @@ -183,7 +183,7 @@
}


def _fallback_recommendation(gap: str) -> dict:
def _fallback_recommendation(gap: str) -> Dict[str, str]:
"""Generic four-key advice for a gap with no curated entry.

The documented fallback (spec art_bTvdPdJS §6): unknown gaps keep their
Expand All @@ -198,13 +198,13 @@ def _fallback_recommendation(gap: str) -> dict:
}


def get_recommendations(gaps: List[str]) -> List[dict]:
def get_recommendations(gaps: List[str]) -> List[Dict[str, str]]:
"""Build one four-key recommendation per gap, in input order.

A gap without a curated entry gets the documented fallback
(see :func:`_fallback_recommendation`) — never a silent drop.
"""
out: List[dict] = []
out: List[Dict[str, str]] = []
for gap in gaps:
entry = _RECOMMENDATIONS.get(gap)
out.append(dict(entry) if entry is not None else _fallback_recommendation(gap))
Expand Down
5 changes: 4 additions & 1 deletion inputguard/registry.py
Original file line number Diff line number Diff line change
Expand Up @@ -121,7 +121,10 @@ def check(self, text: str) -> Optional[RuleFinding]:

def _instantiate(rule_cls: type) -> Rule:
try:
return rule_cls()
# Bare ``type`` construction is typed Any; the annotation pins the
# protocol so strict mode sees a Rule, not Any.
instance: Rule = rule_cls()
return instance
except TypeError as exc:
raise TypeError(
f"Cannot register rule class {rule_cls.__name__!r}: it must be "
Expand Down
Loading
Loading