From 5d345fc875e357daf6d9ac12e3cc85c29f3cf90a Mon Sep 17 00:00:00 2001 From: Yash-Chindam Date: Sun, 30 Aug 2026 09:27:03 +0530 Subject: [PATCH 1/5] feat: expose authenticated OpenAI-compatible gateway --- src/llm_router/app.py | 164 +++++++++++++++++++++++++++++++++++++ src/llm_router/backends.py | 34 ++++++++ 2 files changed, 198 insertions(+) create mode 100644 src/llm_router/app.py create mode 100644 src/llm_router/backends.py diff --git a/src/llm_router/app.py b/src/llm_router/app.py new file mode 100644 index 0000000..76d6a2c --- /dev/null +++ b/src/llm_router/app.py @@ -0,0 +1,164 @@ +import hashlib +import secrets +import time +import uuid +from collections.abc import AsyncIterator +from contextlib import asynccontextmanager + +from fastapi import Depends, FastAPI, Header, HTTPException, Request, Response, status +from fastapi.responses import JSONResponse + +from llm_router.admission import ( + AdmissionController, + AdmissionRejectedError, + QuotaExceededError, + SlidingWindowQuota, +) +from llm_router.backends import InferenceBackend, MockInferenceBackend +from llm_router.config import Settings, get_settings +from llm_router.models import ( + ChatCompletionChoice, + ChatCompletionRequest, + ChatCompletionResponse, + ChatMessage, + Usage, +) +from llm_router.routing import NoEligibleModelError, Router, default_model_profiles + + +def create_app( + settings: Settings | None = None, + *, + backend: InferenceBackend | None = None, +) -> FastAPI: + runtime_settings = settings or get_settings() + router = Router( + profiles=default_model_profiles(), + external_fallback_enabled=runtime_settings.external_fallback_enabled, + ) + admission = AdmissionController( + runtime_settings.max_concurrency, + runtime_settings.admission_timeout_seconds, + ) + quota = SlidingWindowQuota(runtime_settings.quota_requests_per_minute) + inference_backend = backend or MockInferenceBackend() + + @asynccontextmanager + async def lifespan(app: FastAPI) -> AsyncIterator[None]: + app.state.ready = True + yield + app.state.ready = False + + app = FastAPI( + title="Local LLM Inference Router", + version="0.1.0", + lifespan=lifespan, + ) + + async def authenticate(authorization: str | None = Header(default=None)) -> str: + prefix = "Bearer " + if authorization is None or not authorization.startswith(prefix): + raise HTTPException( + status_code=status.HTTP_401_UNAUTHORIZED, + detail="missing bearer token", + headers={"WWW-Authenticate": "Bearer"}, + ) + token = authorization.removeprefix(prefix) + if not any( + secrets.compare_digest(token, candidate) + for candidate in runtime_settings.accepted_api_keys + ): + raise HTTPException( + status_code=status.HTTP_401_UNAUTHORIZED, + detail="invalid bearer token", + headers={"WWW-Authenticate": "Bearer"}, + ) + return hashlib.sha256(token.encode()).hexdigest() + + @app.exception_handler(NoEligibleModelError) + async def no_model_handler(_: Request, error: NoEligibleModelError) -> JSONResponse: + return JSONResponse(status_code=422, content={"error": {"message": str(error)}}) + + @app.exception_handler(AdmissionRejectedError) + async def admission_handler(_: Request, error: AdmissionRejectedError) -> JSONResponse: + return JSONResponse( + status_code=503, + headers={"Retry-After": "1"}, + content={"error": {"message": str(error), "type": "overloaded"}}, + ) + + @app.exception_handler(QuotaExceededError) + async def quota_handler(_: Request, error: QuotaExceededError) -> JSONResponse: + return JSONResponse( + status_code=429, + headers={"Retry-After": "60"}, + content={"error": {"message": str(error), "type": "quota_exceeded"}}, + ) + + @app.get("/healthz") + async def health() -> dict[str, str]: + return {"status": "healthy"} + + @app.get("/readyz") + async def readiness(request: Request) -> dict[str, str]: + if not getattr(request.app.state, "ready", False): + raise HTTPException(status_code=503, detail="not ready") + return {"status": "ready"} + + @app.get("/v1/models", dependencies=[Depends(authenticate)]) + async def models() -> dict[str, object]: + visible = [ + { + "id": profile.id, + "object": "model", + "owned_by": "local" if profile.local else "external-policy", + "revision": profile.revision, + "healthy": profile.healthy, + } + for profile in router.profiles + if profile.local or runtime_settings.external_fallback_enabled + ] + return {"object": "list", "data": visible} + + @app.post("/v1/chat/completions", response_model=ChatCompletionResponse) + async def chat_completions( + payload: ChatCompletionRequest, + response: Response, + subject: str = Depends(authenticate), + ) -> ChatCompletionResponse: + await quota.consume(subject) + decision = router.select(payload) + async with admission.slot(): + result = await inference_backend.generate(payload, decision) + + response.headers["X-Route-Model"] = decision.profile.id + response.headers["X-Route-Revision"] = decision.profile.revision + response.headers["X-Route-Reason"] = decision.reason + return ChatCompletionResponse( + id=f"chatcmpl-{uuid.uuid4().hex}", + created=int(time.time()), + model=decision.profile.id, + choices=[ + ChatCompletionChoice( + message=ChatMessage(role="assistant", content=result.text), + finish_reason="length" if result.finish_reason == "length" else "stop", + ) + ], + usage=Usage( + prompt_tokens=result.prompt_tokens, + completion_tokens=result.completion_tokens, + total_tokens=result.prompt_tokens + result.completion_tokens, + ), + routing={ + "model_revision": decision.profile.revision, + "task": decision.task.value, + "reason": decision.reason, + "score": decision.score, + "candidate_count": decision.candidate_count, + }, + ) + + return app + + +app = create_app() diff --git a/src/llm_router/backends.py b/src/llm_router/backends.py new file mode 100644 index 0000000..a092f78 --- /dev/null +++ b/src/llm_router/backends.py @@ -0,0 +1,34 @@ +from dataclasses import dataclass +from typing import Protocol + +from llm_router.models import ChatCompletionRequest, RouteDecision + + +@dataclass(frozen=True) +class BackendResult: + text: str + prompt_tokens: int + completion_tokens: int + finish_reason: str = "stop" + + +class InferenceBackend(Protocol): + async def generate( + self, request: ChatCompletionRequest, decision: RouteDecision + ) -> BackendResult: ... + + +class MockInferenceBackend: + """Deterministic backend used until vLLM deployments are configured.""" + + async def generate( + self, request: ChatCompletionRequest, decision: RouteDecision + ) -> BackendResult: + response = f"[{decision.profile.id}] accepted {decision.task.value} request" + prompt_tokens = max(1, len(request.prompt) // 4) + completion_tokens = max(1, len(response) // 4) + return BackendResult( + text=response, + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + ) From 587b4c184ab8ba02a2fe28a35704a7ea8979f5bb Mon Sep 17 00:00:00 2001 From: Yash-Chindam Date: Sun, 30 Aug 2026 09:27:04 +0530 Subject: [PATCH 2/5] test: add gateway integration coverage --- tests/integration/test_api.py | 91 +++++++++++++++++++++++++++++++++++ 1 file changed, 91 insertions(+) create mode 100644 tests/integration/test_api.py diff --git a/tests/integration/test_api.py b/tests/integration/test_api.py new file mode 100644 index 0000000..72cc4ae --- /dev/null +++ b/tests/integration/test_api.py @@ -0,0 +1,91 @@ +from fastapi.testclient import TestClient + +from llm_router.app import create_app +from llm_router.config import Settings + + +def build_client(*, quota: int = 10, external: bool = False) -> TestClient: + settings = Settings( + api_keys="integration-key", + quota_requests_per_minute=quota, + external_fallback_enabled=external, + ) + return TestClient(create_app(settings)) + + +def test_health_readiness_and_model_catalog() -> None: + with build_client() as client: + assert client.get("/healthz").json() == {"status": "healthy"} + assert client.get("/readyz").json() == {"status": "ready"} + assert client.get("/v1/models").status_code == 401 + response = client.get("/v1/models", headers={"Authorization": "Bearer integration-key"}) + assert response.status_code == 200 + model_ids = {model["id"] for model in response.json()["data"]} + assert "small-specialist" in model_ids + assert "approved-external-fallback" not in model_ids + + +def test_openai_compatible_completion_contains_route_attribution() -> None: + with build_client() as client: + response = client.post( + "/v1/chat/completions", + headers={"Authorization": "Bearer integration-key"}, + json={ + "model": "auto", + "messages": [{"role": "user", "content": "Extract customer fields"}], + "max_tokens": 64, + "routing": {"privacy": "restricted"}, + }, + ) + assert response.status_code == 200 + payload = response.json() + assert payload["object"] == "chat.completion" + assert payload["model"] == "small-specialist" + assert payload["routing"]["model_revision"] + assert payload["routing"]["task"] == "extraction" + assert response.headers["x-route-model"] == "small-specialist" + assert response.headers["x-route-revision"] == payload["routing"]["model_revision"] + + +def test_validation_and_quota_errors_are_explicit() -> None: + with build_client(quota=1) as client: + headers = {"Authorization": "Bearer integration-key"} + invalid = client.post( + "/v1/chat/completions", + headers=headers, + json={"messages": [{"role": "user", "content": "hello"}], "stream": True}, + ) + assert invalid.status_code == 422 + first = client.post( + "/v1/chat/completions", + headers=headers, + json={"messages": [{"role": "user", "content": "hello"}]}, + ) + second = client.post( + "/v1/chat/completions", + headers=headers, + json={"messages": [{"role": "user", "content": "hello again"}]}, + ) + assert first.status_code == 200 + assert second.status_code == 429 + assert second.headers["retry-after"] == "60" + assert second.json()["error"]["type"] == "quota_exceeded" + + +def test_external_model_requires_public_data_and_explicit_opt_in() -> None: + with build_client(external=True) as client: + response = client.post( + "/v1/chat/completions", + headers={"Authorization": "Bearer integration-key"}, + json={ + "model": "approved-external-fallback", + "messages": [{"role": "user", "content": "Analyze deeply"}], + "routing": { + "privacy": "public", + "allow_external_fallback": True, + "task": "reasoning", + }, + }, + ) + assert response.status_code == 200 + assert response.json()["model"] == "approved-external-fallback" From d872b265fd46af54f1aaee4d32dcc8cce4e25fb5 Mon Sep 17 00:00:00 2001 From: Yash-Chindam Date: Sun, 30 Aug 2026 09:27:04 +0530 Subject: [PATCH 3/5] build: package gateway as non-root container --- .dockerignore | 10 ++++++++++ Dockerfile | 25 +++++++++++++++++++++++++ 2 files changed, 35 insertions(+) create mode 100644 .dockerignore create mode 100644 Dockerfile diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..b612216 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,10 @@ +.git +.github +.venv +__pycache__ +*.pyc +node_modules +playwright-report +test-results +tests + diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..35f1a0c --- /dev/null +++ b/Dockerfile @@ -0,0 +1,25 @@ +FROM python:3.13-slim AS builder + +WORKDIR /build +COPY pyproject.toml README.md ./ +COPY src ./src +RUN python -m pip wheel --no-cache-dir --wheel-dir /wheels . + +FROM python:3.13-slim AS runtime + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + ROUTER_ENVIRONMENT=production + +RUN useradd --create-home --uid 10001 appuser +COPY --from=builder /wheels /wheels +RUN python -m pip install --no-cache-dir /wheels/* && rm -rf /wheels + +USER appuser +WORKDIR /app +EXPOSE 8000 +HEALTHCHECK --interval=30s --timeout=3s --start-period=5s --retries=3 \ + CMD python -c "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/healthz', timeout=2)" + +CMD ["uvicorn", "llm_router.app:app", "--host", "0.0.0.0", "--port", "8000"] + From 9d5893700db218b26beec89fc3cdf85e55899500 Mon Sep 17 00:00:00 2001 From: Yash-Chindam Date: Sun, 30 Aug 2026 09:27:27 +0530 Subject: [PATCH 4/5] docs: document gateway usage and runtime policy --- README.md | 48 ++++++++++++++++++++++++++++++++++++++++++++---- 1 file changed, 44 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index cc7b847..54e0f8b 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,8 @@ # Production Local-LLM Inference & Routing Platform -Policy-aware routing components for a production local-model inference platform. +An OpenAI-compatible control plane for routing requests across local model tiers. The +current inference backend is deterministic for CI and is replaceable by Ray Serve and +vLLM deployments. The complete architecture and design targets are documented in [`02-production-local-llm-inference-routing-platform.md`](02-production-local-llm-inference-routing-platform.md). @@ -12,11 +14,49 @@ Requires Python 3.11 or newer. ```bash python -m venv .venv python -m pip install -e ".[dev]" +python -m uvicorn llm_router.app:app --app-dir src --reload +``` + +The development bearer token is `dev-key`. Override it in every shared environment. +Production startup rejects that development key. + +```bash +ROUTER_API_KEYS="replace-me" python -m uvicorn llm_router.app:app --app-dir src +``` + +Example request: + +```bash +curl http://127.0.0.1:8000/v1/chat/completions \ + -H "Authorization: Bearer dev-key" \ + -H "Content-Type: application/json" \ + -d '{"model":"auto","messages":[{"role":"user","content":"Extract invoice fields"}],"routing":{"privacy":"restricted"}}' +``` + +Every response records the selected model, immutable revision, inferred task, candidate +count, policy score, and route reason. + +## Verification + +```bash ruff format --check . ruff check . mypy -pytest tests/unit --cov=llm_router --cov-report=term-missing +pytest tests/unit tests/integration --cov=llm_router --cov-report=term-missing +docker build -t local-llm-router:dev . ``` -This first delivery slice contains deterministic routing, privacy restrictions, quotas, -and bounded admission. API ingress and model-serving adapters are delivered separately. +## Runtime settings + +All settings use the `ROUTER_` prefix. + +| Variable | Default | Purpose | +|---|---:|---| +| `ROUTER_API_KEYS` | `dev-key` | Comma-separated bearer tokens. | +| `ROUTER_MAX_CONCURRENCY` | `32` | Maximum in-flight requests. | +| `ROUTER_ADMISSION_TIMEOUT_SECONDS` | `0.25` | Time allowed to wait for capacity. | +| `ROUTER_QUOTA_REQUESTS_PER_MINUTE` | `120` | Per-token sliding-window quota. | +| `ROUTER_EXTERNAL_FALLBACK_ENABLED` | `false` | Operator gate for external fallback. | + +External routing also requires public data and request-level opt-in. Private and restricted +requests are never eligible for an external route. From 3c6d9abe80d84424547ac29623be8d86a753f3ab Mon Sep 17 00:00:00 2001 From: Yash-Chindam Date: Sun, 30 Aug 2026 09:27:45 +0530 Subject: [PATCH 5/5] ci: add integration and container release gates --- .github/workflows/ci.yml | 29 ++++++++++++++++++++++++++++- 1 file changed, 28 insertions(+), 1 deletion(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b22c043..6fb6f6f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -26,8 +26,35 @@ jobs: - run: ruff format --check . - run: ruff check . - run: mypy - - run: pytest tests/unit --cov=llm_router --cov-report=term-missing --cov-report=xml + - run: pytest tests/unit tests/integration --cov=llm_router --cov-report=term-missing --cov-report=xml - uses: actions/upload-artifact@v4 with: name: coverage path: coverage.xml + + integration: + name: Integration + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + cache: pip + - run: python -m pip install -e ".[dev]" + - run: pytest tests/integration + + container: + name: Release image validation + runs-on: ubuntu-latest + needs: [unit, integration] + steps: + - uses: actions/checkout@v4 + - uses: docker/setup-buildx-action@v3 + - uses: docker/build-push-action@v6 + with: + context: . + push: false + tags: local-llm-router:${{ github.sha }} + cache-from: type=gha + cache-to: type=gha,mode=max