diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..77495ee --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,30 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + branches: [main] + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + node-version: [20, 22] + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-node@v7 + with: + node-version: ${{ matrix.node-version }} + cache: npm + - name: Install dependencies + run: npm ci + - name: Build and test + run: npm test + - name: Verify package contents + run: npm pack --dry-run diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..2517162 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,46 @@ +# Contributing to API Inspector + +API Inspector is designed to remain auditable, provider-agnostic, and free of runtime dependencies. Contributions should preserve those properties. + +## Prerequisites + +- Node.js 20 or 22 +- npm 10 or newer + +## Local Verification + +```bash +npm ci +npm test +npm pack --dry-run +``` + +`npm test` builds the TypeScript source and runs deterministic tests with Node's built-in test runner. Required tests must not call a paid provider or depend on a live API key. + +## Design Boundaries + +- Provider-specific request and response translation belongs in `src/providers/`. +- Reusable audit behavior belongs in `src/checks/`. +- Pipeline order and short-circuit behavior belong in `src/pipeline.ts`. +- Trust-score changes belong in `src/scoring.ts` and must include explicit scoring tests. +- CLI parsing must remain dependency-free and should be covered in `test/args.test.mjs`. + +## Adding a Provider + +Implement the shared `ProviderClient` interface, export the adapter through `src/providers/index.ts`, and add deterministic tests with stubbed responses. Document provider-specific limitations. Never commit real credentials or recorded responses containing account data. + +## Pull Requests + +Keep pull requests focused and include: + +- the problem or audit gap; +- the behavior changed; +- tests added or updated; +- commands run and their results; +- any credential, privacy, or compatibility implications. + +Run `git diff --check` before requesting review. + +## Security Issues + +Do not open a public issue for a vulnerability or exposed credential. Follow [SECURITY.md](./SECURITY.md). diff --git a/README.md b/README.md index 9064731..de3949c 100644 --- a/README.md +++ b/README.md @@ -1,273 +1,149 @@ # API Inspector -> **Built by [Tokenta](https://tokenta.space)** — Open-source toolkit for verifying LLM API providers. +API Inspector is a dependency-free TypeScript CLI for auditing LLM API endpoints. It compares advertised model access with observable endpoint behavior and produces an explainable trust report for OpenAI, Anthropic, Gemini, and OpenAI-compatible gateways. -[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) -[![Node](https://img.shields.io/badge/node-%3E=20-3c873a.svg)](https://nodejs.org) -[![Status](https://img.shields.io/badge/status-alpha-orange.svg)](#roadmap) +[![CI](https://github.com/Tokenta/api-inspector/actions/workflows/ci.yml/badge.svg)](https://github.com/Tokenta/api-inspector/actions/workflows/ci.yml) +[![License: MIT](https://img.shields.io/badge/license-MIT-yellow.svg)](./LICENSE) +[![Node.js](https://img.shields.io/badge/Node.js-20%20or%2022-3c873a.svg)](https://nodejs.org/) -Detect model identity mismatches, rate-limit discrepancies, quota anomalies, and performance claims of any OpenAI / Anthropic / Gemini / OpenAI-compatible LLM endpoint — in under a minute, from your terminal, with zero runtime dependencies. +Status: alpha. Treat the report as engineering evidence, not a security or purchasing guarantee. ---- +## Checks -## Why API Inspector - -Most developers no longer buy AI access from official providers. They buy from resellers, marketplaces, Discord sellers, Telegram channels, generic API hubs, or "OpenAI-compatible" gateways. A typical purchase looks like this: - -| Seller claims | Reality (often) | +| Check | What it observes | | --- | --- | -| `gpt-4o` | Routed to **Kimi K2** or **DeepSeek-V3** behind a translation prompt | -| 100 RPM | 10 RPM in practice; 429s after the third request | -| 128K context | Hard fails above 32K | -| $100 of credit | $0.12 left on a shared account | -| Stable model | Silently swapped between providers per-request | - -There is currently no easy way for buyers to **independently verify** any of this. API Inspector is that tool. - -It runs the same kind of verification pipeline Tokenta uses internally to grade listings on its trusted-API marketplace, packaged as a standalone CLI you can run against any endpoint you control or are evaluating. - ---- - -## Features +| Reachability | Whether the configured host responds | +| Authentication and discovery | Whether the key is accepted and which models the endpoint advertises | +| Claim comparison | Exact, family-level, absent, or unknown model-name match | +| Latency | p50 and p95 over a configurable set of small requests | +| Context probe | The largest attempted prompt accepted by the endpoint | +| Fingerprint | Multi-prompt heuristics and whether they contradict the claimed model | +| Rate-limit transparency | Standard rate-limit headers returned by the endpoint | +| Trust score | Fixed, inspectable weights from `src/scoring.ts` | -- **Model identity fingerprinting** — multi-prompt vendor detection that flags the most common model swaps (GPT to Kimi, Claude to DeepSeek, Gemini to Qwen, etc.). -- **Authentication and quota probe** — confirms the key is live and surfaces remaining-quota signals where providers expose them. -- **Model discovery vs claim** — compares the seller's claimed model name to what the endpoint actually advertises. -- **Effective context window probe** — sends canary tokens at increasing prompt sizes (1K → 32K by default, up to 128K with `--full-context`). -- **Latency benchmark** — p50 and p95 over a configurable burst. -- **Rate-limit transparency check** — inspects standard `x-ratelimit-*` headers for resold-endpoint stripping. -- **Explainable trust score** — every contribution to the 0–100 score is tied to a specific check; no opaque ML. -- **Provider plug-ins** — built-in clients for OpenAI, Anthropic, Google Gemini, plus any OpenAI-compatible custom endpoint. -- **CI-friendly** — `--json` flag emits a machine-readable report for build pipelines and audits. -- **Zero runtime dependencies** — ships only the TypeScript compiler and `@types/node` as dev deps. +All required project tests are deterministic and run without provider credentials. Live endpoint behavior depends on the provider, account, model, network, and gateway configuration at inspection time. ---- +## Install From Source -## Install - -### Run with `npx` (no install) - -```bash -npx @tokenta/api-inspector verify -``` - -Or run the latest from GitHub directly: - -```bash -npx github:Tokenta/api-inspector verify -``` - -### Global install - -```bash -npm install -g @tokenta/api-inspector -api-inspector verify -``` - -### From source +The supported installation method for the current alpha is a source checkout. The package is not yet published to the npm registry. ```bash git clone https://github.com/Tokenta/api-inspector.git cd api-inspector -npm install -npm run build -node dist/cli.js verify +npm ci +npm test +node dist/cli.js --version ``` -Requires Node.js 20 or newer. - ---- +Requires Node.js 20 or 22. ## Usage -### Interactive +Show help: ```bash -api-inspector verify -``` - -``` -Tokenta Inspector -Independent verification for any LLM API provider. - -Select provider: - 1. OpenAI (api.openai.com) - 2. Anthropic (api.anthropic.com) - 3. Google Gemini (generativelanguage.googleapis.com) - 4. Custom (OpenAI-compat) (your reseller / gateway) -> 1 -Enter API key (input hidden): -> ******************** -Claimed model (optional, e.g. gpt-4o): -> gpt-4o +node dist/cli.js --help ``` -### Non-interactive +Audit an OpenAI endpoint using an environment variable: ```bash -api-inspector verify \ +API_INSPECTOR_KEY="your-test-key" \ + node dist/cli.js verify \ --provider openai \ - --key "$OPENAI_API_KEY" \ --claimed gpt-4o ``` -### Audit a reseller / gateway +Audit an OpenAI-compatible gateway: ```bash -api-inspector verify \ +API_INSPECTOR_KEY="your-test-key" \ + node dist/cli.js verify \ --provider custom \ --base-url https://gateway.example.com/v1 \ - --key "sk-..." \ --claimed gpt-4o ``` -### CI-friendly JSON +Emit JSON for a build pipeline or an audit record: ```bash -api-inspector verify --provider openai --key "$KEY" --json > report.json +API_INSPECTOR_KEY="your-test-key" \ + node dist/cli.js verify \ + --provider openai \ + --json > report.json ``` -### All options +On PowerShell, set the key with `$env:API_INSPECTOR_KEY="your-test-key"` before running the command. + +### Options | Flag | Description | | --- | --- | -| `--provider ` | `openai` \| `anthropic` \| `gemini` \| `custom` | -| `--key ` | API key (or set `API_INSPECTOR_KEY`) | -| `--claimed ` | Model name advertised by the seller | -| `--base-url ` | Override base URL (required for `custom`) | -| `--quick` | Skip context-window probe | -| `--full-context` | Probe up to 128K context | -| `--latency-samples ` | Latency probe sample count (default 5) | -| `--timeout ` | Per-request timeout (default 30000) | -| `--json` | Emit JSON instead of a TTY report | +| `--provider ` | `openai`, `anthropic`, `gemini`, or `custom` | +| `--key ` | API key; `API_INSPECTOR_KEY` is safer for routine use | +| `--claimed ` | Model name advertised by the seller or gateway | +| `--base-url ` | Base URL override; required for `custom` | +| `--quick` | Skip the context-window probe | +| `--full-context` | Extend attempted context sizes up to 128K | +| `--latency-samples ` | Number of latency samples; default 5 | +| `--timeout ` | Per-request timeout; default 30000 | +| `--json` | Emit a machine-readable report | | `--no-color` | Disable ANSI color output | | `-h`, `--help` | Show help | | `-v`, `--version` | Show version | ---- - -## Sample output - -``` -API Inspector Report -Tool v0.1.0 Started 2026-06-07T01:42:11Z - -Provider api.openai.com -Base URL https://api.openai.com/v1 -Claimed gpt-4o -Detected OpenAI (gpt) (confidence 100%) -Context 32K observed -Latency p50 412 ms p95 780 ms -Rate limit 10000 req/min (header reported) - -Pipeline - [OK] Endpoint reachability api.openai.com responded with HTTP 401 - [OK] Authentication Key accepted by api.openai.com - [OK] Model discovery gpt-4o is advertised (catalog: 67 models) - [OK] Latency benchmark p50 412 ms, p95 780 ms (5/5 successful) - [OK] Context window probe Confirmed at 32K tokens - [OK] Fingerprint analysis Detected OpenAI (confidence 100%) - consistent with claimed "gpt-4o" - [OK] Rate-limit headers 6 rate-limit headers exposed - - Trust score: 95 / 100 [VERIFIED] - -Built by Tokenta - https://tokenta.space -Source: https://github.com/Tokenta/api-inspector -``` - -A failing audit looks like this: - -``` -Pipeline - [OK] Endpoint reachability gateway.example.com responded with HTTP 200 - [OK] Authentication Key accepted by gateway.example.com - [!!] Model discovery "gpt-4o" is NOT advertised by this endpoint (3 models listed) - [OK] Latency benchmark p50 1840 ms, p95 5420 ms (5/5 successful) - [!!] Context window probe Confirmed at 4K tokens - [XX] Fingerprint analysis Detected DeepSeek (confidence 100%) but seller claimed "gpt-4o". Likely model swap. - [!!] Rate-limit headers No standard rate-limit headers exposed (transparency concern on resold endpoints) - - Trust score: 28 / 100 [HIGH RISK] -``` - ---- - -## Pipeline +## Explainable Scoring -1. **Endpoint reachability** — TLS handshake, host responds. -2. **Authentication** — `GET /models` (or equivalent) with the supplied key. -3. **Model discovery** — compare advertised models to the seller's claim. Grades: `exact`, `family`, `absent`. -4. **Latency benchmark** — N small chat requests, p50 / p95 of round-trip latency. -5. **Context window probe** — canary tokens at 1K → 32K (`--full-context` extends to 128K). -6. **Fingerprint analysis** — multi-prompt vendor identification with confidence and claim-mismatch flagging. -7. **Rate-limit transparency** — inspects standard `x-ratelimit-*` headers. -8. **Trust score** — explainable additive aggregation across the seven checks. +The score is a fixed additive model, not a learned classifier: -See `src/scoring.ts` for the exact weights — the math is meant to be auditable, not a black box. +| Evidence | Maximum contribution | +| --- | ---: | +| Authentication | 20 | +| Reachability | 5 | +| Model discovery and claim match | 20 | +| Fingerprint and claim consistency | 25 | +| Context probe | 10 | +| Latency probe | 10 | +| Rate-limit transparency | 10 | ---- +Scores at or above 80 are labeled `VERIFIED`, scores from 60 through 79 are `SUSPECT`, and lower scores are `HIGH_RISK`. A fingerprint contradiction applies a penalty. Read [src/scoring.ts](./src/scoring.ts) and [the scoring tests](./test/scoring.test.mjs) for the exact contract. ## Development ```bash -git clone https://github.com/Tokenta/api-inspector.git -cd api-inspector -npm install -npm run build -node dist/cli.js verify +npm ci +npm test +npm pack --dry-run ``` -Project layout: +Project structure: -``` +```text src/ - cli.ts CLI entry, arg parsing, interactive prompts - pipeline.ts Orchestrates checks in order - scoring.ts Trust score weighting (auditable, no ML) - report.ts TTY report renderer - types.ts Shared types and interfaces - providers/ OpenAI / Anthropic / Gemini / Custom clients - checks/ Reachability, auth+discovery, latency, context, - fingerprint, rate-limit - util/ ANSI colors, fetch wrapper, args, prompt + cli.ts CLI entry and validation + pipeline.ts Ordered inspection orchestration + scoring.ts Explainable trust-score weights + providers/ Provider request and response adapters + checks/ Independent evidence checks + util/ Argument, HTTP, prompt, and color helpers +test/ + args.test.mjs + scoring.test.mjs ``` -Pull requests are welcome. Promising contribution areas: - -- New provider plug-ins (Mistral, Together, Groq, Bedrock, Vertex, Cohere). -- Additional fingerprint heuristics (tokenizer behavior, knowledge-cutoff probes, deterministic-trap prompts). -- Streaming verification mode. -- Cryptographic proof receipts for tamper-evident audit trails. - ---- +See [CONTRIBUTING.md](./CONTRIBUTING.md) for provider boundaries and testing rules. Report vulnerabilities through [SECURITY.md](./SECURITY.md), not a public issue. ## Roadmap -- [ ] Streaming verification mode (`--stream`) -- [ ] Provider plug-ins: Mistral, Together, Groq, Bedrock, Vertex, Cohere -- [ ] Burst-test mode for true rate-limit measurement -- [ ] Tokenizer-divergence fingerprint -- [ ] Cryptographic proof receipts (tamper-evident) -- [ ] `pip install api-inspector` (Python wrapper) -- [ ] Homebrew formula - ---- +- Streaming verification mode +- Additional provider adapters with deterministic fixtures +- Burst testing for observed rate-limit behavior +- Tokenizer-divergence fingerprint experiments +- Tamper-evident audit receipts +- A versioned npm release after the alpha contract is stable ## License -MIT — see [LICENSE](./LICENSE). - ---- - -## Need a verified AI marketplace? - -API Inspector is part of the **[Tokenta](https://tokenta.space)** open-source ecosystem: - -- **api-inspector** — this repo. Verify any LLM endpoint from your terminal. -- **model-fingerprints** — public dataset of vendor signatures (coming soon). -- **provider-benchmark** — reproducible benchmarks across providers (coming soon). -- **tokenta-platform** — the trusted AI API marketplace (in beta). - -Sellers on Tokenta are continuously verified using the same pipeline that ships in this CLI. If you are tired of model swaps, ghost quotas, and ratelimit lies, follow the project on GitHub and **stay tuned for Tokenta — coming soon.** +MIT. See [LICENSE](./LICENSE). -> Powered by the Tokenta Verification Engine. +Built by [Tokenta](https://tokenta.space). diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..9be8c87 --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,25 @@ +# Security Policy + +## Supported Code + +Security fixes are evaluated against the current default branch. API Inspector is an alpha-stage auditing tool and should not be the sole basis for a purchasing, security, or compliance decision. + +## Reporting a Vulnerability + +Do not open a public issue for a vulnerability, leaked API key, or response containing private account data. + +Use GitHub's private vulnerability reporting for this repository when available. Otherwise email `hello@tokenta.space` with the subject `API Inspector security report` and include the affected module or commit, minimal reproduction steps, observed impact, and any suggested mitigation. + +Use a synthetic endpoint and placeholder credentials whenever possible. Revoke any live key that may have been exposed before sending a report. + +## Credential Handling + +API keys are accepted as command-line or environment input and are used for direct requests to the selected endpoint. Command-line arguments may be visible to other local processes or retained in shell history, so environment variables are safer for routine use: + +```bash +API_INSPECTOR_KEY="your-test-key" node dist/cli.js verify --provider openai +``` + +Do not paste credentials into issues, pull requests, CI logs, screenshots, or saved JSON reports. Review custom base URLs before sending a key; a custom endpoint receives the supplied credential. + +This repository does not publish a response-time or remediation-time guarantee. diff --git a/package.json b/package.json index ac510b7..ae07ba4 100644 --- a/package.json +++ b/package.json @@ -14,6 +14,7 @@ ], "scripts": { "build": "tsc -p .", + "test": "npm run build && node --test", "prepare": "npm run build", "start": "node dist/cli.js", "verify": "node dist/cli.js verify" diff --git a/src/util/args.ts b/src/util/args.ts index ee4eb37..aaed9d1 100644 --- a/src/util/args.ts +++ b/src/util/args.ts @@ -14,6 +14,8 @@ export interface ParsedArgs { positionals: string[]; } +const BOOLEAN_FLAGS = new Set(["quick", "full-context", "json", "no-color"]); + export function parseArgs(argv: string[]): ParsedArgs { const args = argv.slice(2); const flags: Record = {}; @@ -42,6 +44,10 @@ export function parseArgs(argv: string[]): ParsedArgs { continue; } const key = token.slice(2); + if (BOOLEAN_FLAGS.has(key)) { + flags[key] = true; + continue; + } const next = args[i + 1]; if (next != null && !next.startsWith("-")) { flags[key] = next; diff --git a/test/args.test.mjs b/test/args.test.mjs new file mode 100644 index 0000000..559f760 --- /dev/null +++ b/test/args.test.mjs @@ -0,0 +1,66 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { flagBool, flagNumber, flagString, parseArgs } from "../dist/util/args.js"; + +test("parses the verify command and mixed flag forms", () => { + const result = parseArgs([ + "node", + "api-inspector", + "verify", + "--provider", + "custom", + "--base-url=https://gateway.example.com/v1", + "--claimed", + "gpt-4o", + "--json", + "audit-label", + ]); + + assert.equal(result.command, "verify"); + assert.deepEqual(result.flags, { + provider: "custom", + "base-url": "https://gateway.example.com/v1", + claimed: "gpt-4o", + json: true, + }); + assert.deepEqual(result.positionals, ["audit-label"]); +}); + +test("normalizes help and version shorthands", () => { + const result = parseArgs(["node", "api-inspector", "-h", "-v"]); + + assert.equal(result.command, null); + assert.equal(flagBool(result.flags, "help"), true); + assert.equal(flagBool(result.flags, "version"), true); +}); + +test("reads string, boolean, and numeric flag values", () => { + const result = parseArgs([ + "node", + "api-inspector", + "verify", + "--provider", + "openai", + "--quick=true", + "--latency-samples", + "7", + ]); + + assert.equal(flagString(result.flags, "provider"), "openai"); + assert.equal(flagBool(result.flags, "quick"), true); + assert.equal(flagNumber(result.flags, "latency-samples"), 7); + assert.equal(flagString(result.flags, "missing"), undefined); +}); + +test("rejects non-numeric values through the numeric accessor", () => { + const result = parseArgs([ + "node", + "api-inspector", + "verify", + "--timeout", + "not-a-number", + ]); + + assert.equal(flagNumber(result.flags, "timeout"), undefined); +}); diff --git a/test/scoring.test.mjs b/test/scoring.test.mjs new file mode 100644 index 0000000..e95c3b5 --- /dev/null +++ b/test/scoring.test.mjs @@ -0,0 +1,69 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { computeTrustScore } from "../dist/scoring.js"; + +function check(id, status) { + return { id, label: id, status }; +} + +test("awards a verified verdict when every evidence check passes", () => { + const result = computeTrustScore({ + checks: [ + check("reachability", "pass"), + check("auth", "pass"), + check("fingerprint", "pass"), + check("context", "pass"), + check("latency", "pass"), + check("rate-limit", "pass"), + ], + matchGrade: "exact", + fingerprintConfidence: 100, + fingerprintMismatch: false, + }); + + assert.deepEqual(result, { score: 100, verdict: "VERIFIED" }); +}); + +test("applies the fingerprint mismatch penalty even when other checks pass", () => { + const result = computeTrustScore({ + checks: [ + check("reachability", "pass"), + check("auth", "pass"), + check("fingerprint", "fail"), + check("context", "pass"), + check("latency", "pass"), + check("rate-limit", "pass"), + ], + matchGrade: "exact", + fingerprintConfidence: 100, + fingerprintMismatch: true, + }); + + assert.deepEqual(result, { score: 65, verdict: "SUSPECT" }); +}); + +test("keeps weak or unavailable evidence in the high-risk range", () => { + const result = computeTrustScore({ + checks: [ + check("reachability", "pass"), + check("auth", "pass"), + check("fingerprint", "warn"), + check("context", "warn"), + check("latency", "warn"), + check("rate-limit", "warn"), + ], + matchGrade: "absent", + }); + + assert.deepEqual(result, { score: 43, verdict: "HIGH_RISK" }); +}); + +test("clamps a mismatch-only score at zero", () => { + const result = computeTrustScore({ + checks: [check("fingerprint", "fail")], + fingerprintMismatch: true, + }); + + assert.deepEqual(result, { score: 0, verdict: "HIGH_RISK" }); +});